雨澄润 发表于 2026-9-26 14:40:18

修正爬虫爬取内容不准确

本帖最后由 雨澄润 于 2026-9-26 16:44 编辑

这是一个爬虫程序,但爬虫爬取的内容与用户输入内容无关,希望能有大神能够帮忙解答一下。
有人能帮忙的话,真的万分感谢!!!

import tkinter as tk
from tkinter import messagebox, scrolledtext
import requests
from bs4 import BeautifulSoup
import sqlite3
import urllib.parse
import json
import time

# ==================== 数据库操作(完全保留原有逻辑) ====================
DB_NAME = "quotes.db"

def init_db():
    """初始化数据库,创建表;若旧表缺少 keyword 列则自动补上"""
    conn = sqlite3.connect(DB_NAME)
    cursor = conn.cursor()
    cursor.execute(
      "CREATE TABLE IF NOT EXISTS quotes ("
      "id INTEGER PRIMARY KEY AUTOINCREMENT, "
      "text TEXT, "
      "author TEXT, "
      "keyword TEXT)"
    )
    cursor.execute("PRAGMA table_info(quotes)")
    columns = for row in cursor.fetchall()]
    if "keyword" not in columns:
      cursor.execute("ALTER TABLE quotes ADD COLUMN keyword TEXT")
    conn.commit()
    conn.close()


def save_to_db(data_list, keyword=""):
    """将爬取的数据批量存入数据库"""
    if not data_list:
      return 0
    conn = sqlite3.connect(DB_NAME)
    cursor = conn.cursor()
    cursor.executemany(
      "INSERT INTO quotes (text, author, keyword) VALUES (?, ?, ?)",
      [(item["text"], item["author"], keyword) for item in data_list]
    )
    conn.commit()
    conn.close()
    return len(data_list)


def get_from_db():
    """从数据库读取所有数据"""
    conn = sqlite3.connect(DB_NAME)
    cursor = conn.cursor()
    cursor.execute("SELECT text, author, keyword FROM quotes")
    rows = cursor.fetchall()
    conn.close()
    return rows


# ==================== 统一请求头(模拟真实浏览器) ====================
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
    "Upgrade-Insecure-Requests": "1",
}


# ==================== 源1:古诗文网(API接口,返回JSON) ====================
def crawl_from_gushiwen_api(keyword):
    """
    古诗文网 - 通过API接口获取名句(JSON格式,无需HTML解析)
    优势:结构稳定、中文无乱码、返回数据干净
    """
    results = []
    api_url = "https://api.gushiwen.cn/v3/gshiwang/shiju/search"
   
    params = {
      "key": keyword,
      "page": 1,
      "size": 20,
    }
   
    try:
      resp = requests.get(api_url, headers=HEADERS, params=params, timeout=15)
      # 使用 resp.text(自动编码检测),不手动 decode
      resp.encoding = resp.apparent_encoding or 'utf-8'
      data = resp.json()
      
      if data and "data" in data and data["data"]:
            for item in data["data"]:
                text = item.get("title", "") or item.get("content", "")
                author = item.get("author", "") or item.get("dynasty", "") + item.get("author", "")
                if text and len(text.strip()) > 2:
                  results.append({
                        "text": text.strip(),
                        "author": author.strip() if author else "佚名"
                  })
    except Exception as e:
      raise Exception(f"古诗文网API请求失败: {str(e)}")
   
    return results


# ==================== 源2:古诗文网(HTML页面备用) ====================
def crawl_from_gushiwen_html(keyword):
    """
    古诗文网 - 通过HTML页面搜索备用
    使用多个CSS选择器策略提高兼容性
    """
    results = []
    search_url = "https://so.gushiwen.cn/search/"
   
    params = {
      "value": keyword,
    }
   
    try:
      resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
      # 关键修复:使用 resp.text 自动处理编码,避免中文乱码
      resp.encoding = resp.apparent_encoding or 'utf-8'
      html = resp.text
      
      soup = BeautifulSoup(html, "html.parser")
      
      # 策略1:尝试新的选择器 div.content > div.left > div.song > div.son > p
      items = soup.select("div.content div.left div.song div.son p a")
      if len(items) >= 2:
            for i in range(0, len(items) - 1, 2):
                text = items.get_text(strip=True)
                author = items.get_text(strip=True) if i + 1 < len(items) else ""
                if text and len(text.strip()) > 2:
                  results.append({"text": text.strip(), "author": author.strip()})
      
      # 策略2:尝试另一种选择器
      if not results:
            items = soup.select("div.sons p a")
            if len(items) >= 2:
                for i in range(0, len(items) - 1, 2):
                  text = items.get_text(strip=True)
                  author = items.get_text(strip=True) if i + 1 < len(items) else ""
                  if text and len(text.strip()) > 2:
                        results.append({"text": text.strip(), "author": author.strip()})
      
      # 策略3:尝试 div.left > div.son
      if not results:
            items = soup.select("div.left div.son p a")
            if len(items) >= 2:
                for i in range(0, len(items) - 1, 2):
                  text = items.get_text(strip=True)
                  author = items.get_text(strip=True) if i + 1 < len(items) else ""
                  if text and len(text.strip()) > 2:
                        results.append({"text": text.strip(), "author": author.strip()})
                        
    except Exception as e:
      raise Exception(f"古诗文网HTML请求失败: {str(e)}")
   
    return results


# ==================== 源3:百度汉语(成语/词语释义) ====================
def crawl_from_baidu_hanyu(keyword):
    """
    百度汉语 - 搜索成语/词语释义
    作为古诗文网的备用源
    """
    results = []
    search_url = "https://dict.baidu.com/s"
   
    params = {
      "wd": keyword,
    }
   
    try:
      resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
      resp.encoding = resp.apparent_encoding or 'utf-8'
      html = resp.text
      
      soup = BeautifulSoup(html, "html.parser")
      
      # 尝试获取百度汉语的释义内容
      # 策略:查找包含释义的div
      definitions = soup.select("div.vc-paragraph-inner, div.vc-content, div.result")
      for div in definitions:
            text = div.get_text(strip=True)
            if text and len(text) > 5 and keyword in text:
                results.append({"text": text[:200], "author": "百度汉语"})
      
      # 如果上面的选择器没匹配到,尝试更宽泛的搜索
      if not results:
            all_texts = soup.find_all(string=True)
            for t in all_texts:
                t_stripped = t.strip()
                if keyword in t_stripped and len(t_stripped) > 10:
                  results.append({"text": t_stripped[:200], "author": "百度汉语"})
                  if len(results) >= 5:
                        break
                        
    except Exception as e:
      raise Exception(f"百度汉语请求失败: {str(e)}")
   
    return results


# ==================== 源4:一言API(随机名句) ====================
def crawl_from_hitokoto():
    """
    一言API - 获取随机名句(不依赖关键词,作为兜底源)
    """
    results = []
    urls = [
      "https://v1.hitokoto.cn/?c=a",# 诗词古典
      "https://v1.hitokoto.cn/?c=f",# 哲学思想
      "https://v1.hitokoto.cn/?c=d",# 原创
    ]
   
    for url in urls:
      try:
            resp = requests.get(url, headers=HEADERS, timeout=10)
            resp.encoding = resp.apparent_encoding or 'utf-8'
            data = resp.json()
            if "hitokoto" in data and data["hitokoto"]:
                results.append({
                  "text": data["hitokoto"],
                  "author": data.get("from", "") or data.get("creator", "佚名")
                })
      except:
            continue
   
    return results


# ==================== 主爬虫函数(多源fallback + 关键词验证) ====================
def crawl_quotes(keyword=""):
    """
    多源爬虫主函数:
    1. 有关键词:优先古诗文网API -> HTML备用 -> 百度汉语
    2. 无关键词:返回一言API随机名句
    3. 自动编码检测,避免中文乱码
    4. 关键词相关性验证,确保返回内容与搜索词相关
    5. 去重、异常捕获
    """
    keyword = keyword.strip()
    results = []
    seen = set()
   
    def add_result(item):
      """添加结果并去重"""
      text = item.get("text", "").strip()
      author = item.get("author", "").strip()
      if not text or len(text) < 3:
            return
      key = (text[:50], author[:20])
      if key not in seen:
            seen.add(key)
            results.append({"text": text, "author": author})
   
    def filter_by_keyword(items, kw):
      """过滤与关键词相关的结果"""
      if not kw:
            return items
      filtered = []
      kw_lower = kw.lower()
      for item in items:
            text = item.get("text", "")
            author = item.get("author", "")
            # 只要文本或作者中包含关键词就算相关
            if kw_lower in text.lower() or kw_lower in author.lower():
                filtered.append(item)
            elif len(filtered) < 3:
                # 允许部分匹配的结果(防止严格过滤导致无结果)
                filtered.append(item)
      return filtered if filtered else items# 兜底:至少返回一些结果
   
    try:
      if keyword:
            # === 有关键词:多源搜索 ===
            
            # 源1:古诗文网API
            try:
                api_results = crawl_from_gushiwen_api(keyword)
                for item in api_results:
                  add_result(item)
            except Exception:
                pass
            
            # 源2:古诗文网HTML备用
            if len(results) < 3:
                try:
                  html_results = crawl_from_gushiwen_html(keyword)
                  for item in html_results:
                        add_result(item)
                except Exception:
                  pass
            
            # 源3:百度汉语
            if len(results) < 3:
                try:
                  baidu_results = crawl_from_baidu_hanyu(keyword)
                  for item in baidu_results:
                        add_result(item)
                except Exception:
                  pass
            
            # 关键词相关性过滤
            results = filter_by_keyword(results, keyword)
            
      else:
            # === 无关键词:返回随机名句 ===
            results = crawl_from_hitokoto()
            
            # 如果一言API也失败,尝试古诗文网默认名句
            if not results:
                try:
                  api_results = crawl_from_gushiwen_api("")
                  for item in api_results:
                        add_result(item)
                  results = api_results
                except Exception:
                  pass
                  
    except Exception as e:
      raise Exception(f"爬虫主流程异常: {str(e)}")
   
    # 最终去重
    final_results = []
    final_seen = set()
    for item in results:
      key = item["text"][:50]
      if key not in final_seen:
            final_seen.add(key)
            final_results.append(item)
   
    return final_results


# ==================== GUI界面 ====================
init_db()
root = tk.Tk()
root.title("简易AI 8.0【多源爬虫 · 中文关键词搜索 · 修复版】")
root.geometry("750x620")

# ---------- 功能一:用户输入获取区 ----------
frame_input = tk.LabelFrame(root, text="用户输入获取(支持中文,如:理想、青春、奋斗、爱情)", padx=10, pady=10)
frame_input.pack(padx=10, pady=5, fill="x")
entry = tk.Entry(frame_input, width=45)
entry.pack(side="left", padx=5)


def get_input():
    content = entry.get()
    if content:
      messagebox.showinfo("输入结果", f"你输入了:{content}")
    else:
      messagebox.showwarning("提示", "输入框为空!")


btn_get = tk.Button(frame_input, text="获取输入", command=get_input)
btn_get.pack(side="left", padx=5)

# ---------- 功能二:爬虫爬取与存储区 ----------
frame_crawl = tk.LabelFrame(root, text="爬虫爬取与存储(多源fallback机制)", padx=10, pady=10)
frame_crawl.pack(padx=10, pady=5, fill="both", expand=True)
btn_frame = tk.Frame(frame_crawl)
btn_frame.pack(pady=5)
btn_crawl = tk.Button(btn_frame, text="爬取并存储", width=14)
btn_crawl.pack(side="left", padx=5)
btn_show = tk.Button(btn_frame, text="查看数据库", width=14)
btn_show.pack(side="left", padx=5)

tk.Label(frame_crawl, text="爬取结果:", font=("微软雅黑", 10, "bold")).pack(anchor="w", padx=5)

# 使用 scrolledtext 替代 Text,支持自动滚动
text_area = scrolledtext.ScrolledText(frame_crawl, width=85, height=20, font=("微软雅黑", 10))
text_area.pack(padx=5, pady=5, fill="both", expand=True)

# ---------- 功能三:退出按钮 ----------
frame_exit = tk.Frame(root)
frame_exit.pack(pady=10)


def on_exit():
    root.destroy()


btn_exit = tk.Button(frame_exit, text="退出", command=on_exit)
btn_exit.pack()


# ==================== 按钮回调函数 ====================
def on_crawl():
    user_input = entry.get().strip()
    btn_crawl.config(state="disabled", text="爬取中...")
    root.update()
   
    # 清空结果区
    text_area.delete("1.0", tk.END)
   
    # 显示搜索信息
    if user_input:
      text_area.insert(tk.END, f"===== 搜索关键词:{user_input} =====\n")
      text_area.insert(tk.END, f"===== 多源搜索策略:古诗文网API → 古诗文网HTML → 百度汉语 =====\n\n")
    else:
      text_area.insert(tk.END, "===== 未输入关键词,抓取随机名句 =====\n")
      text_area.insert(tk.END, f"===== 源:一言API + 古诗文网 =====\n\n")
   
    # 执行爬取(带超时和异常处理)
    try:
      data = crawl_quotes(keyword=user_input)
    except Exception as e:
      text_area.insert(tk.END, f"【错误】爬虫执行失败:{str(e)}\n")
      text_area.insert(tk.END, "请检查网络连接后重试。\n")
      messagebox.showerror("爬取出错", f"爬虫执行失败:{str(e)}\n请检查网络连接!")
      btn_crawl.config(state="normal", text="爬取并存储")
      return
   
    # 处理结果
    if data:
      count = save_to_db(data, keyword=user_input)
      text_area.insert(tk.END, f"===== 共爬取并存储 {count} 条数据到数据库 =====\n\n")
      for i, item in enumerate(data, 1):
            text_area.insert(tk.END, f"【{i}】{item['text']}\n")
            text_area.insert(tk.END, f"    —— {item['author']}\n\n")
      
      # 根据结果数量给出不同反馈
      if len(data) >= 5:
            messagebox.showinfo("成功", f"爬取完成!共获取 {count} 条数据(多源合并),已存储到数据库。")
      elif len(data) >= 1:
            messagebox.showinfo("部分成功", f"爬取完成!共获取 {count} 条数据,已存储到数据库。\n如结果不够精确,请尝试更换关键词。")
      else:
            messagebox.showwarning("结果有限", "未获取到足够数据,请检查网络或更换关键词!")
    else:
      # 无结果时的智能提示
      if user_input:
            text_area.insert(tk.END, f"未找到与「{user_input}」直接匹配的内容。\n")
            text_area.insert(tk.END, "可能原因:\n")
            text_area.insert(tk.END, "1. 网络连接不稳定,部分数据源未返回结果\n")
            text_area.insert(tk.END, "2. 关键词较为生僻,尝试更通用的词语\n")
            text_area.insert(tk.END, "3. 建议尝试:理想、青春、奋斗、思念、爱情、人生、月亮、花等常见主题\n")
            messagebox.showwarning("无结果",
                f"未找到与「{user_input}」匹配的内容。\n"
                "建议:尝试更通用的关键词,如「理想」「青春」「奋斗」「思念」等。\n"
                "请检查网络连接后重试。")
      else:
            text_area.insert(tk.END, "未能抓取到名句,请检查网络连接后重试。\n")
            messagebox.showwarning("无结果", "未抓取到任何数据,请检查网络连接!")
   
    btn_crawl.config(state="normal", text="爬取并存储")


def on_show():
    rows = get_from_db()
    if not rows:
      messagebox.showwarning("提示", "数据库为空,请先爬取数据!")
      return
    text_area.delete("1.0", tk.END)
    text_area.insert(tk.END, f"===== 数据库中共有 {len(rows)} 条数据 =====\n\n")
    for i, (text, author, keyword) in enumerate(rows, 1):
      tag_info = f"(搜索关键词:{keyword})" if keyword else ""
      text_area.insert(tk.END, f"【{i}】{text}\n")
      text_area.insert(tk.END, f"    —— {author} {tag_info}\n\n")


btn_crawl.config(command=on_crawl)
btn_show.config(command=on_show)

root.mainloop()

FishC 发表于 2026-9-26 14:40:33

没有代码只能先判断方向。
爬取内容和输入无关,常见原因一般是:请求没带上关键词、参数名写错、GET/POST 用错、搜索结果是 AJAX 动态加载、解析选择器选错。

按这个顺序排查:

[*]打印最终请求地址和参数,看关键词有没有真的发出去。
[*]浏览器按 F12,打开 Network,搜索一次,看真实请求是 GET 还是 POST、参数名是什么。
[*]用相同 headers、params 或 data 重新请求。
[*]打印 r.url、r.status_code、r.text[:500]。
[*]如果是 AJAX,直接请求数据接口,或用 Selenium/Playwright。
[*]检查解析用的选择器是否对应搜索结果区域。


通用模板:

import requests

keyword = "用户输入"
url = "搜索页地址"   # 换成真实搜索地址
params = {
    "q": keyword,   # 参数名必须按 F12 里的真实名字改
}

headers = {
    "User-Agent": "Mozilla/5.0 ..."
}

r = requests.get(url, params=params, headers=headers, timeout=10)

print("最终URL:", r.url)
print("状态码:", r.status_code)
print(r.text[:500])


如果真实请求是 POST,把 params 改成 data:

r = requests.post(url, data={"q": keyword}, headers=headers, timeout=10)


解析:

from bs4 import BeautifulSoup

soup = BeautifulSoup(r.text, "html.parser")
items = soup.select("搜索结果条目的选择器")
for item in items:
    print(item.get_text(strip=True))


最后:把完整代码、输入关键词、实际爬到的内容、期望内容,以及 F12 里搜索请求的参数名发出来,我帮你直接改。

以上回复来自 FishC AI 小助理,如未能正确解答您的问题,请继续追问。

雨澄润 发表于 2026-9-26 16:47:32

FishC 发表于 2026-9-26 14:40
没有代码只能先判断方向。
爬取内容和输入无关,常见原因一般是:请求没带上关键词、参数名写错、GET/PO ...

追问:代码如下


import tkinter as tk
from tkinter import messagebox, scrolledtext
import requests
from bs4 import BeautifulSoup
import sqlite3
import urllib.parse
import json
import time

# ==================== 数据库操作(完全保留原有逻辑) ====================
DB_NAME = "quotes.db"

def init_db():
    """初始化数据库,创建表;若旧表缺少 keyword 列则自动补上"""
    conn = sqlite3.connect(DB_NAME)
    cursor = conn.cursor()
    cursor.execute(
      "CREATE TABLE IF NOT EXISTS quotes ("
      "id INTEGER PRIMARY KEY AUTOINCREMENT, "
      "text TEXT, "
      "author TEXT, "
      "keyword TEXT)"
    )
    cursor.execute("PRAGMA table_info(quotes)")
    columns = for row in cursor.fetchall()]
    if "keyword" not in columns:
      cursor.execute("ALTER TABLE quotes ADD COLUMN keyword TEXT")
    conn.commit()
    conn.close()


def save_to_db(data_list, keyword=""):
    """将爬取的数据批量存入数据库"""
    if not data_list:
      return 0
    conn = sqlite3.connect(DB_NAME)
    cursor = conn.cursor()
    cursor.executemany(
      "INSERT INTO quotes (text, author, keyword) VALUES (?, ?, ?)",
      [(item["text"], item["author"], keyword) for item in data_list]
    )
    conn.commit()
    conn.close()
    return len(data_list)


def get_from_db():
    """从数据库读取所有数据"""
    conn = sqlite3.connect(DB_NAME)
    cursor = conn.cursor()
    cursor.execute("SELECT text, author, keyword FROM quotes")
    rows = cursor.fetchall()
    conn.close()
    return rows


# ==================== 统一请求头(模拟真实浏览器) ====================
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Accept-Encoding": "gzip, deflate, br",
    "Connection": "keep-alive",
    "Upgrade-Insecure-Requests": "1",
}


# ==================== 源1:古诗文网(API接口,返回JSON) ====================
def crawl_from_gushiwen_api(keyword):
    """
    古诗文网 - 通过API接口获取名句(JSON格式,无需HTML解析)
    优势:结构稳定、中文无乱码、返回数据干净
    """
    results = []
    api_url = "https://api.gushiwen.cn/v3/gshiwang/shiju/search"
   
    params = {
      "key": keyword,
      "page": 1,
      "size": 20,
    }
   
    try:
      resp = requests.get(api_url, headers=HEADERS, params=params, timeout=15)
      # 使用 resp.text(自动编码检测),不手动 decode
      resp.encoding = resp.apparent_encoding or 'utf-8'
      data = resp.json()
      
      if data and "data" in data and data["data"]:
            for item in data["data"]:
                text = item.get("title", "") or item.get("content", "")
                author = item.get("author", "") or item.get("dynasty", "") + item.get("author", "")
                if text and len(text.strip()) > 2:
                  results.append({
                        "text": text.strip(),
                        "author": author.strip() if author else "佚名"
                  })
    except Exception as e:
      raise Exception(f"古诗文网API请求失败: {str(e)}")
   
    return results


# ==================== 源2:古诗文网(HTML页面备用) ====================
def crawl_from_gushiwen_html(keyword):
    """
    古诗文网 - 通过HTML页面搜索备用
    使用多个CSS选择器策略提高兼容性
    """
    results = []
    search_url = "https://so.gushiwen.cn/search/"
   
    params = {
      "value": keyword,
    }
   
    try:
      resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
      # 关键修复:使用 resp.text 自动处理编码,避免中文乱码
      resp.encoding = resp.apparent_encoding or 'utf-8'
      html = resp.text
      
      soup = BeautifulSoup(html, "html.parser")
      
      # 策略1:尝试新的选择器 div.content > div.left > div.song > div.son > p
      items = soup.select("div.content div.left div.song div.son p a")
      if len(items) >= 2:
            for i in range(0, len(items) - 1, 2):
                text = items.get_text(strip=True)
                author = items.get_text(strip=True) if i + 1 < len(items) else ""
                if text and len(text.strip()) > 2:
                  results.append({"text": text.strip(), "author": author.strip()})
      
      # 策略2:尝试另一种选择器
      if not results:
            items = soup.select("div.sons p a")
            if len(items) >= 2:
                for i in range(0, len(items) - 1, 2):
                  text = items.get_text(strip=True)
                  author = items.get_text(strip=True) if i + 1 < len(items) else ""
                  if text and len(text.strip()) > 2:
                        results.append({"text": text.strip(), "author": author.strip()})
      
      # 策略3:尝试 div.left > div.son
      if not results:
            items = soup.select("div.left div.son p a")
            if len(items) >= 2:
                for i in range(0, len(items) - 1, 2):
                  text = items.get_text(strip=True)
                  author = items.get_text(strip=True) if i + 1 < len(items) else ""
                  if text and len(text.strip()) > 2:
                        results.append({"text": text.strip(), "author": author.strip()})
                        
    except Exception as e:
      raise Exception(f"古诗文网HTML请求失败: {str(e)}")
   
    return results


# ==================== 源3:百度汉语(成语/词语释义) ====================
def crawl_from_baidu_hanyu(keyword):
    """
    百度汉语 - 搜索成语/词语释义
    作为古诗文网的备用源
    """
    results = []
    search_url = "https://dict.baidu.com/s"
   
    params = {
      "wd": keyword,
    }
   
    try:
      resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
      resp.encoding = resp.apparent_encoding or 'utf-8'
      html = resp.text
      
      soup = BeautifulSoup(html, "html.parser")
      
      # 尝试获取百度汉语的释义内容
      # 策略:查找包含释义的div
      definitions = soup.select("div.vc-paragraph-inner, div.vc-content, div.result")
      for div in definitions:
            text = div.get_text(strip=True)
            if text and len(text) > 5 and keyword in text:
                results.append({"text": text[:200], "author": "百度汉语"})
      
      # 如果上面的选择器没匹配到,尝试更宽泛的搜索
      if not results:
            all_texts = soup.find_all(string=True)
            for t in all_texts:
                t_stripped = t.strip()
                if keyword in t_stripped and len(t_stripped) > 10:
                  results.append({"text": t_stripped[:200], "author": "百度汉语"})
                  if len(results) >= 5:
                        break
                        
    except Exception as e:
      raise Exception(f"百度汉语请求失败: {str(e)}")
   
    return results


# ==================== 源4:一言API(随机名句) ====================
def crawl_from_hitokoto():
    """
    一言API - 获取随机名句(不依赖关键词,作为兜底源)
    """
    results = []
    urls = [
      "https://v1.hitokoto.cn/?c=a",# 诗词古典
      "https://v1.hitokoto.cn/?c=f",# 哲学思想
      "https://v1.hitokoto.cn/?c=d",# 原创
    ]
   
    for url in urls:
      try:
            resp = requests.get(url, headers=HEADERS, timeout=10)
            resp.encoding = resp.apparent_encoding or 'utf-8'
            data = resp.json()
            if "hitokoto" in data and data["hitokoto"]:
                results.append({
                  "text": data["hitokoto"],
                  "author": data.get("from", "") or data.get("creator", "佚名")
                })
      except:
            continue
   
    return results


# ==================== 主爬虫函数(多源fallback + 关键词验证) ====================
def crawl_quotes(keyword=""):
    """
    多源爬虫主函数:
    1. 有关键词:优先古诗文网API -> HTML备用 -> 百度汉语
    2. 无关键词:返回一言API随机名句
    3. 自动编码检测,避免中文乱码
    4. 关键词相关性验证,确保返回内容与搜索词相关
    5. 去重、异常捕获
    """
    keyword = keyword.strip()
    results = []
    seen = set()
   
    def add_result(item):
      """添加结果并去重"""
      text = item.get("text", "").strip()
      author = item.get("author", "").strip()
      if not text or len(text) < 3:
            return
      key = (text[:50], author[:20])
      if key not in seen:
            seen.add(key)
            results.append({"text": text, "author": author})
   
    def filter_by_keyword(items, kw):
      """过滤与关键词相关的结果"""
      if not kw:
            return items
      filtered = []
      kw_lower = kw.lower()
      for item in items:
            text = item.get("text", "")
            author = item.get("author", "")
            # 只要文本或作者中包含关键词就算相关
            if kw_lower in text.lower() or kw_lower in author.lower():
                filtered.append(item)
            elif len(filtered) < 3:
                # 允许部分匹配的结果(防止严格过滤导致无结果)
                filtered.append(item)
      return filtered if filtered else items# 兜底:至少返回一些结果
   
    try:
      if keyword:
            # === 有关键词:多源搜索 ===
            
            # 源1:古诗文网API
            try:
                api_results = crawl_from_gushiwen_api(keyword)
                for item in api_results:
                  add_result(item)
            except Exception:
                pass
            
            # 源2:古诗文网HTML备用
            if len(results) < 3:
                try:
                  html_results = crawl_from_gushiwen_html(keyword)
                  for item in html_results:
                        add_result(item)
                except Exception:
                  pass
            
            # 源3:百度汉语
            if len(results) < 3:
                try:
                  baidu_results = crawl_from_baidu_hanyu(keyword)
                  for item in baidu_results:
                        add_result(item)
                except Exception:
                  pass
            
            # 关键词相关性过滤
            results = filter_by_keyword(results, keyword)
            
      else:
            # === 无关键词:返回随机名句 ===
            results = crawl_from_hitokoto()
            
            # 如果一言API也失败,尝试古诗文网默认名句
            if not results:
                try:
                  api_results = crawl_from_gushiwen_api("")
                  for item in api_results:
                        add_result(item)
                  results = api_results
                except Exception:
                  pass
                  
    except Exception as e:
      raise Exception(f"爬虫主流程异常: {str(e)}")
   
    # 最终去重
    final_results = []
    final_seen = set()
    for item in results:
      key = item["text"][:50]
      if key not in final_seen:
            final_seen.add(key)
            final_results.append(item)
   
    return final_results


# ==================== GUI界面 ====================
init_db()
root = tk.Tk()
root.title("简易AI 8.0【多源爬虫 · 中文关键词搜索 · 修复版】")
root.geometry("750x620")

# ---------- 功能一:用户输入获取区 ----------
frame_input = tk.LabelFrame(root, text="用户输入获取(支持中文,如:理想、青春、奋斗、爱情)", padx=10, pady=10)
frame_input.pack(padx=10, pady=5, fill="x")
entry = tk.Entry(frame_input, width=45)
entry.pack(side="left", padx=5)


def get_input():
    content = entry.get()
    if content:
      messagebox.showinfo("输入结果", f"你输入了:{content}")
    else:
      messagebox.showwarning("提示", "输入框为空!")


btn_get = tk.Button(frame_input, text="获取输入", command=get_input)
btn_get.pack(side="left", padx=5)

# ---------- 功能二:爬虫爬取与存储区 ----------
frame_crawl = tk.LabelFrame(root, text="爬虫爬取与存储(多源fallback机制)", padx=10, pady=10)
frame_crawl.pack(padx=10, pady=5, fill="both", expand=True)
btn_frame = tk.Frame(frame_crawl)
btn_frame.pack(pady=5)
btn_crawl = tk.Button(btn_frame, text="爬取并存储", width=14)
btn_crawl.pack(side="left", padx=5)
btn_show = tk.Button(btn_frame, text="查看数据库", width=14)
btn_show.pack(side="left", padx=5)

tk.Label(frame_crawl, text="爬取结果:", font=("微软雅黑", 10, "bold")).pack(anchor="w", padx=5)

# 使用 scrolledtext 替代 Text,支持自动滚动
text_area = scrolledtext.ScrolledText(frame_crawl, width=85, height=20, font=("微软雅黑", 10))
text_area.pack(padx=5, pady=5, fill="both", expand=True)

# ---------- 功能三:退出按钮 ----------
frame_exit = tk.Frame(root)
frame_exit.pack(pady=10)


def on_exit():
    root.destroy()


btn_exit = tk.Button(frame_exit, text="退出", command=on_exit)
btn_exit.pack()


# ==================== 按钮回调函数 ====================
def on_crawl():
    user_input = entry.get().strip()
    btn_crawl.config(state="disabled", text="爬取中...")
    root.update()
   
    # 清空结果区
    text_area.delete("1.0", tk.END)
   
    # 显示搜索信息
    if user_input:
      text_area.insert(tk.END, f"===== 搜索关键词:{user_input} =====\n")
      text_area.insert(tk.END, f"===== 多源搜索策略:古诗文网API → 古诗文网HTML → 百度汉语 =====\n\n")
    else:
      text_area.insert(tk.END, "===== 未输入关键词,抓取随机名句 =====\n")
      text_area.insert(tk.END, f"===== 源:一言API + 古诗文网 =====\n\n")
   
    # 执行爬取(带超时和异常处理)
    try:
      data = crawl_quotes(keyword=user_input)
    except Exception as e:
      text_area.insert(tk.END, f"【错误】爬虫执行失败:{str(e)}\n")
      text_area.insert(tk.END, "请检查网络连接后重试。\n")
      messagebox.showerror("爬取出错", f"爬虫执行失败:{str(e)}\n请检查网络连接!")
      btn_crawl.config(state="normal", text="爬取并存储")
      return
   
    # 处理结果
    if data:
      count = save_to_db(data, keyword=user_input)
      text_area.insert(tk.END, f"===== 共爬取并存储 {count} 条数据到数据库 =====\n\n")
      for i, item in enumerate(data, 1):
            text_area.insert(tk.END, f"【{i}】{item['text']}\n")
            text_area.insert(tk.END, f"    —— {item['author']}\n\n")
      
      # 根据结果数量给出不同反馈
      if len(data) >= 5:
            messagebox.showinfo("成功", f"爬取完成!共获取 {count} 条数据(多源合并),已存储到数据库。")
      elif len(data) >= 1:
            messagebox.showinfo("部分成功", f"爬取完成!共获取 {count} 条数据,已存储到数据库。\n如结果不够精确,请尝试更换关键词。")
      else:
            messagebox.showwarning("结果有限", "未获取到足够数据,请检查网络或更换关键词!")
    else:
      # 无结果时的智能提示
      if user_input:
            text_area.insert(tk.END, f"未找到与「{user_input}」直接匹配的内容。\n")
            text_area.insert(tk.END, "可能原因:\n")
            text_area.insert(tk.END, "1. 网络连接不稳定,部分数据源未返回结果\n")
            text_area.insert(tk.END, "2. 关键词较为生僻,尝试更通用的词语\n")
            text_area.insert(tk.END, "3. 建议尝试:理想、青春、奋斗、思念、爱情、人生、月亮、花等常见主题\n")
            messagebox.showwarning("无结果",
                f"未找到与「{user_input}」匹配的内容。\n"
                "建议:尝试更通用的关键词,如「理想」「青春」「奋斗」「思念」等。\n"
                "请检查网络连接后重试。")
      else:
            text_area.insert(tk.END, "未能抓取到名句,请检查网络连接后重试。\n")
            messagebox.showwarning("无结果", "未抓取到任何数据,请检查网络连接!")
   
    btn_crawl.config(state="normal", text="爬取并存储")


def on_show():
    rows = get_from_db()
    if not rows:
      messagebox.showwarning("提示", "数据库为空,请先爬取数据!")
      return
    text_area.delete("1.0", tk.END)
    text_area.insert(tk.END, f"===== 数据库中共有 {len(rows)} 条数据 =====\n\n")
    for i, (text, author, keyword) in enumerate(rows, 1):
      tag_info = f"(搜索关键词:{keyword})" if keyword else ""
      text_area.insert(tk.END, f"【{i}】{text}\n")
      text_area.insert(tk.END, f"    —— {author} {tag_info}\n\n")


btn_crawl.config(command=on_crawl)
btn_show.config(command=on_show)

root.mainloop()

FishC 发表于 2026-9-26 16:47:55

雨澄润 发表于 2026-09-26 16:47
FishC 发表于 2026-9-26 14:40
没有代码只能先判断方 ...

问题出在几处“结果对象”和“字段取值”上,关键词根本没被正确使用,所以爬回来的内容跟输入无关。下面直接列出核心 bug,并给出修正后的完整代码。

一、核心问题


[*] 古诗文网 API 地址 api.gushiwen.cn/v3/... 基本不存在,resp.json() 会抛异常被 except 吞掉,结果永远落到后面的“兜底源”。
[*] 兜底源里的“一言 API”是随机名句,跟关键词毫无关系,一兜底就变成“内容与输入无关”。
[*] save_to_db 里写的是 [(item, item, keyword) for item in data_list],把整个字典当文本存了,应该取 item["text"] 和 item["author"]。
[*] 界面显示时 f"【{i}】{item}" 打印的是整个字典,应打印 item["text"]。
[*] filter_by_keyword 里有一段 elif len(filtered) < 3: filtered.append(item),会把不相关的结果强行塞进去,等于白过滤。
[*] 最终去重 key = item 用字典当 key,字典不可哈希,会报 TypeError。
[*] 请求头里 br 编码没装 brotli 时容易乱码,建议去掉。


二、修正后的完整代码


import tkinter as tk
from tkinter import messagebox, scrolledtext
import requests
from bs4 import BeautifulSoup
import sqlite3

DB_NAME = "quotes.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
                  "AppleWebKit/537.36 (KHTML, like Gecko) "
                  "Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


# ==================== 数据库 ====================
def init_db():
    conn = sqlite3.connect(DB_NAME)
    cur = conn.cursor()
    cur.execute(
      "CREATE TABLE IF NOT EXISTS quotes ("
      "id INTEGER PRIMARY KEY AUTOINCREMENT, "
      "text TEXT, author TEXT, keyword TEXT)"
    )
    conn.commit()
    conn.close()


def save_to_db(data_list, keyword=""):
    if not data_list:
      return 0
    conn = sqlite3.connect(DB_NAME)
    cur = conn.cursor()
    cur.executemany(
      "INSERT INTO quotes (text, author, keyword) VALUES (?, ?, ?)",
      [(it["text"], it["author"], keyword) for it in data_list]
    )
    conn.commit()
    conn.close()
    return len(data_list)


def get_from_db():
    conn = sqlite3.connect(DB_NAME)
    cur = conn.cursor()
    cur.execute("SELECT text, author, keyword FROM quotes")
    rows = cur.fetchall()
    conn.close()
    return rows


# ==================== 源1:古诗文网搜索页(按关键词) ====================
def crawl_gushiwen(keyword):
    """古诗文网搜索页,返回与关键词相关的诗句"""
    results = []
    url = "https://so.gushiwen.cn/search.aspx"
    params = {"value": keyword, "valuej": ""}

    resp = requests.get(url, headers=HEADERS, params=params, timeout=15)
    resp.encoding = resp.apparent_encoding or "utf-8"
    soup = BeautifulSoup(resp.text, "html.parser")

    # 搜索结果结构:div.sons > div.cont > p(诗句) + p(出处/作者)
    for cont in soup.select("div.sons div.cont"):
      ps = cont.find_all("p")
      if not ps:
            continue
      text = ps.get_text(strip=True)
      author = ps.get_text(strip=True) if len(ps) > 1 else "佚名"
      if len(text) > 2:
            results.append({"text": text, "author": author})

    return results


# ==================== 源2:一言(仅无关键词时使用) ====================
def crawl_hitokoto():
    """一言随机名句,只在用户没输入关键词时用"""
    url = "https://v1.hitokoto.cn/"
    resp = requests.get(url, headers=HEADERS, timeout=10)
    data = resp.json()
    if data.get("hitokoto"):
      return [{
            "text": data["hitokoto"],
            "author": data.get("from") or data.get("creator") or "佚名",
      }]
    return []


# ==================== 主爬虫 ====================
def crawl_quotes(keyword=""):
    keyword = keyword.strip()
    results = []
    seen = set()

    def add(item):
      text = (item.get("text") or "").strip()
      author = (item.get("author") or "").strip() or "佚名"
      if len(text) < 3:
            return
      key = (text, author)
      if key not in seen:
            seen.add(key)
            results.append({"text": text, "author": author})

    if keyword:
      # 有输入:只走古诗文网搜索页,保证结果与关键词相关
      for it in crawl_gushiwen(keyword):
            add(it)
    else:
      # 无输入:返回随机名句
      for it in crawl_hitokoto():
            add(it)

    return results


# ==================== GUI ====================
init_db()
root = tk.Tk()
root.title("简易AI 8.0【多源爬虫 · 中文关键词搜索 · 修复版】")
root.geometry("750x620")

frame_input = tk.LabelFrame(root, text="用户输入获取(支持中文,如:理想、青春、奋斗、爱情)", padx=10, pady=10)
frame_input.pack(padx=10, pady=5, fill="x")
entry = tk.Entry(frame_input, width=45)
entry.pack(side="left", padx=5)


def get_input():
    content = entry.get()
    if content:
      messagebox.showinfo("输入结果", f"你输入了:{content}")
    else:
      messagebox.showwarning("提示", "输入框为空!")


btn_get = tk.Button(frame_input, text="获取输入", command=get_input)
btn_get.pack(side="left", padx=5)

frame_crawl = tk.LabelFrame(root, text="爬虫爬取与存储", padx=10, pady=10)
frame_crawl.pack(padx=10, pady=5, fill="both", expand=True)
btn_frame = tk.Frame(frame_crawl)
btn_frame.pack(pady=5)
btn_crawl = tk.Button(btn_frame, text="爬取并存储", width=14)
btn_crawl.pack(side="left", padx=5)
btn_show = tk.Button(btn_frame, text="查看数据库", width=14)
btn_show.pack(side="left", padx=5)

tk.Label(frame_crawl, text="爬取结果:", font=("微软雅黑", 10, "bold")).pack(anchor="w", padx=5)
text_area = scrolledtext.ScrolledText(frame_crawl, width=85, height=20, font=("微软雅黑", 10))
text_area.pack(padx=5, pady=5, fill="both", expand=True)

frame_exit = tk.Frame(root)
frame_exit.pack(pady=10)


def on_exit():
    root.destroy()


btn_exit = tk.Button(frame_exit, text="退出", command=on_exit)
btn_exit.pack()


def on_crawl():
    user_input = entry.get().strip()
    btn_crawl.config(state="disabled", text="爬取中...")
    root.update()

    text_area.delete("1.0", tk.END)

    if user_input:
      text_area.insert(tk.END, f"===== 搜索关键词:{user_input} =====\n\n")
    else:
      text_area.insert(tk.END, "===== 未输入关键词,抓取随机名句 =====\n\n")

    try:
      data = crawl_quotes(keyword=user_input)
    except Exception as e:
      text_area.insert(tk.END, f"【错误】爬虫执行失败:{e}\n")
      messagebox.showerror("爬取出错", f"爬虫执行失败:{e}\n请检查网络连接!")
      btn_crawl.config(state="normal", text="爬取并存储")
      return

    if data:
      count = save_to_db(data, keyword=user_input)
      text_area.insert(tk.END, f"===== 共爬取并存储 {count} 条数据 =====\n\n")
      for i, item in enumerate(data, 1):
            text_area.insert(tk.END, f"【{i}】{item['text']}\n")
            text_area.insert(tk.END, f"    —— {item['author']}\n\n")
      messagebox.showinfo("成功", f"爬取完成!共获取 {count} 条数据,已存储到数据库。")
    else:
      text_area.insert(tk.END, f"未找到与「{user_input}」相关的内容。\n")
      messagebox.showwarning("无结果", f"未找到与「{user_input}」相关的内容。")

    btn_crawl.config(state="normal", text="爬取并存储")


def on_show():
    rows = get_from_db()
    if not rows:
      messagebox.showwarning("提示", "数据库为空,请先爬取数据!")
      return
    text_area.delete("1.0", tk.END)
    text_area.insert(tk.END, f"===== 数据库中共有 {len(rows)} 条数据 =====\n\n")
    for i, (text, author, keyword) in enumerate(rows, 1):
      tag = f"(关键词:{keyword})" if keyword else ""
      text_area.insert(tk.END, f"【{i}】{text}\n")
      text_area.insert(tk.END, f"    —— {author} {tag}\n\n")


btn_crawl.config(command=on_crawl)
btn_show.config(command=on_show)

root.mainloop()


三、关键改动说明


[*] 删掉了假的古诗文网 API 和百度汉语源,只保留真正能按关键词搜索的 so.gushiwen.cn/search.aspx。
[*] 一言 API 只在没有输入关键词时才调用,避免污染结果。
[*] save_to_db 正确取 item["text"] / item["author"]。
[*] 界面显示改成取字段,不再打印整个字典。
[*] 去重用 (text, author) 元组,避免不可哈希错误。
[*] 请求头去掉 br,防止编码乱码。


如果运行后古诗文网返回 0 条,一般是它触发了反爬或页面结构改了,可以先用浏览器打开搜索页 F12 看一下 div.sons div.cont 是否还在,把真实选择器发出来我再帮你调。

以上回复来自 FishC AI 小助理,如未能正确解答您的问题,请继续追问。

isdkz 发表于 前天 16:58

我去试了一下,没有问题呀,哪里不准确,请提供一下你的输入案例
页: [1]
查看完整版本: 修正爬虫爬取内容不准确