修正爬虫爬取内容不准确
本帖最后由 雨澄润 于 2026-9-26 16:44 编辑这是一个爬虫程序,但爬虫爬取的内容与用户输入内容无关,希望能有大神能够帮忙解答一下。
有人能帮忙的话,真的万分感谢!!!
import tkinter as tk
from tkinter import messagebox, scrolledtext
import requests
from bs4 import BeautifulSoup
import sqlite3
import urllib.parse
import json
import time
# ==================== 数据库操作(完全保留原有逻辑) ====================
DB_NAME = "quotes.db"
def init_db():
"""初始化数据库,创建表;若旧表缺少 keyword 列则自动补上"""
conn = sqlite3.connect(DB_NAME)
cursor = conn.cursor()
cursor.execute(
"CREATE TABLE IF NOT EXISTS quotes ("
"id INTEGER PRIMARY KEY AUTOINCREMENT, "
"text TEXT, "
"author TEXT, "
"keyword TEXT)"
)
cursor.execute("PRAGMA table_info(quotes)")
columns = for row in cursor.fetchall()]
if "keyword" not in columns:
cursor.execute("ALTER TABLE quotes ADD COLUMN keyword TEXT")
conn.commit()
conn.close()
def save_to_db(data_list, keyword=""):
"""将爬取的数据批量存入数据库"""
if not data_list:
return 0
conn = sqlite3.connect(DB_NAME)
cursor = conn.cursor()
cursor.executemany(
"INSERT INTO quotes (text, author, keyword) VALUES (?, ?, ?)",
[(item["text"], item["author"], keyword) for item in data_list]
)
conn.commit()
conn.close()
return len(data_list)
def get_from_db():
"""从数据库读取所有数据"""
conn = sqlite3.connect(DB_NAME)
cursor = conn.cursor()
cursor.execute("SELECT text, author, keyword FROM quotes")
rows = cursor.fetchall()
conn.close()
return rows
# ==================== 统一请求头(模拟真实浏览器) ====================
HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Safari/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
"Accept-Encoding": "gzip, deflate, br",
"Connection": "keep-alive",
"Upgrade-Insecure-Requests": "1",
}
# ==================== 源1:古诗文网(API接口,返回JSON) ====================
def crawl_from_gushiwen_api(keyword):
"""
古诗文网 - 通过API接口获取名句(JSON格式,无需HTML解析)
优势:结构稳定、中文无乱码、返回数据干净
"""
results = []
api_url = "https://api.gushiwen.cn/v3/gshiwang/shiju/search"
params = {
"key": keyword,
"page": 1,
"size": 20,
}
try:
resp = requests.get(api_url, headers=HEADERS, params=params, timeout=15)
# 使用 resp.text(自动编码检测),不手动 decode
resp.encoding = resp.apparent_encoding or 'utf-8'
data = resp.json()
if data and "data" in data and data["data"]:
for item in data["data"]:
text = item.get("title", "") or item.get("content", "")
author = item.get("author", "") or item.get("dynasty", "") + item.get("author", "")
if text and len(text.strip()) > 2:
results.append({
"text": text.strip(),
"author": author.strip() if author else "佚名"
})
except Exception as e:
raise Exception(f"古诗文网API请求失败: {str(e)}")
return results
# ==================== 源2:古诗文网(HTML页面备用) ====================
def crawl_from_gushiwen_html(keyword):
"""
古诗文网 - 通过HTML页面搜索备用
使用多个CSS选择器策略提高兼容性
"""
results = []
search_url = "https://so.gushiwen.cn/search/"
params = {
"value": keyword,
}
try:
resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
# 关键修复:使用 resp.text 自动处理编码,避免中文乱码
resp.encoding = resp.apparent_encoding or 'utf-8'
html = resp.text
soup = BeautifulSoup(html, "html.parser")
# 策略1:尝试新的选择器 div.content > div.left > div.song > div.son > p
items = soup.select("div.content div.left div.song div.son p a")
if len(items) >= 2:
for i in range(0, len(items) - 1, 2):
text = items.get_text(strip=True)
author = items.get_text(strip=True) if i + 1 < len(items) else ""
if text and len(text.strip()) > 2:
results.append({"text": text.strip(), "author": author.strip()})
# 策略2:尝试另一种选择器
if not results:
items = soup.select("div.sons p a")
if len(items) >= 2:
for i in range(0, len(items) - 1, 2):
text = items.get_text(strip=True)
author = items.get_text(strip=True) if i + 1 < len(items) else ""
if text and len(text.strip()) > 2:
results.append({"text": text.strip(), "author": author.strip()})
# 策略3:尝试 div.left > div.son
if not results:
items = soup.select("div.left div.son p a")
if len(items) >= 2:
for i in range(0, len(items) - 1, 2):
text = items.get_text(strip=True)
author = items.get_text(strip=True) if i + 1 < len(items) else ""
if text and len(text.strip()) > 2:
results.append({"text": text.strip(), "author": author.strip()})
except Exception as e:
raise Exception(f"古诗文网HTML请求失败: {str(e)}")
return results
# ==================== 源3:百度汉语(成语/词语释义) ====================
def crawl_from_baidu_hanyu(keyword):
"""
百度汉语 - 搜索成语/词语释义
作为古诗文网的备用源
"""
results = []
search_url = "https://dict.baidu.com/s"
params = {
"wd": keyword,
}
try:
resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
resp.encoding = resp.apparent_encoding or 'utf-8'
html = resp.text
soup = BeautifulSoup(html, "html.parser")
# 尝试获取百度汉语的释义内容
# 策略:查找包含释义的div
definitions = soup.select("div.vc-paragraph-inner, div.vc-content, div.result")
for div in definitions:
text = div.get_text(strip=True)
if text and len(text) > 5 and keyword in text:
results.append({"text": text[:200], "author": "百度汉语"})
# 如果上面的选择器没匹配到,尝试更宽泛的搜索
if not results:
all_texts = soup.find_all(string=True)
for t in all_texts:
t_stripped = t.strip()
if keyword in t_stripped and len(t_stripped) > 10:
results.append({"text": t_stripped[:200], "author": "百度汉语"})
if len(results) >= 5:
break
except Exception as e:
raise Exception(f"百度汉语请求失败: {str(e)}")
return results
# ==================== 源4:一言API(随机名句) ====================
def crawl_from_hitokoto():
"""
一言API - 获取随机名句(不依赖关键词,作为兜底源)
"""
results = []
urls = [
"https://v1.hitokoto.cn/?c=a",# 诗词古典
"https://v1.hitokoto.cn/?c=f",# 哲学思想
"https://v1.hitokoto.cn/?c=d",# 原创
]
for url in urls:
try:
resp = requests.get(url, headers=HEADERS, timeout=10)
resp.encoding = resp.apparent_encoding or 'utf-8'
data = resp.json()
if "hitokoto" in data and data["hitokoto"]:
results.append({
"text": data["hitokoto"],
"author": data.get("from", "") or data.get("creator", "佚名")
})
except:
continue
return results
# ==================== 主爬虫函数(多源fallback + 关键词验证) ====================
def crawl_quotes(keyword=""):
"""
多源爬虫主函数:
1. 有关键词:优先古诗文网API -> HTML备用 -> 百度汉语
2. 无关键词:返回一言API随机名句
3. 自动编码检测,避免中文乱码
4. 关键词相关性验证,确保返回内容与搜索词相关
5. 去重、异常捕获
"""
keyword = keyword.strip()
results = []
seen = set()
def add_result(item):
"""添加结果并去重"""
text = item.get("text", "").strip()
author = item.get("author", "").strip()
if not text or len(text) < 3:
return
key = (text[:50], author[:20])
if key not in seen:
seen.add(key)
results.append({"text": text, "author": author})
def filter_by_keyword(items, kw):
"""过滤与关键词相关的结果"""
if not kw:
return items
filtered = []
kw_lower = kw.lower()
for item in items:
text = item.get("text", "")
author = item.get("author", "")
# 只要文本或作者中包含关键词就算相关
if kw_lower in text.lower() or kw_lower in author.lower():
filtered.append(item)
elif len(filtered) < 3:
# 允许部分匹配的结果(防止严格过滤导致无结果)
filtered.append(item)
return filtered if filtered else items# 兜底:至少返回一些结果
try:
if keyword:
# === 有关键词:多源搜索 ===
# 源1:古诗文网API
try:
api_results = crawl_from_gushiwen_api(keyword)
for item in api_results:
add_result(item)
except Exception:
pass
# 源2:古诗文网HTML备用
if len(results) < 3:
try:
html_results = crawl_from_gushiwen_html(keyword)
for item in html_results:
add_result(item)
except Exception:
pass
# 源3:百度汉语
if len(results) < 3:
try:
baidu_results = crawl_from_baidu_hanyu(keyword)
for item in baidu_results:
add_result(item)
except Exception:
pass
# 关键词相关性过滤
results = filter_by_keyword(results, keyword)
else:
# === 无关键词:返回随机名句 ===
results = crawl_from_hitokoto()
# 如果一言API也失败,尝试古诗文网默认名句
if not results:
try:
api_results = crawl_from_gushiwen_api("")
for item in api_results:
add_result(item)
results = api_results
except Exception:
pass
except Exception as e:
raise Exception(f"爬虫主流程异常: {str(e)}")
# 最终去重
final_results = []
final_seen = set()
for item in results:
key = item["text"][:50]
if key not in final_seen:
final_seen.add(key)
final_results.append(item)
return final_results
# ==================== GUI界面 ====================
init_db()
root = tk.Tk()
root.title("简易AI 8.0【多源爬虫 · 中文关键词搜索 · 修复版】")
root.geometry("750x620")
# ---------- 功能一:用户输入获取区 ----------
frame_input = tk.LabelFrame(root, text="用户输入获取(支持中文,如:理想、青春、奋斗、爱情)", padx=10, pady=10)
frame_input.pack(padx=10, pady=5, fill="x")
entry = tk.Entry(frame_input, width=45)
entry.pack(side="left", padx=5)
def get_input():
content = entry.get()
if content:
messagebox.showinfo("输入结果", f"你输入了:{content}")
else:
messagebox.showwarning("提示", "输入框为空!")
btn_get = tk.Button(frame_input, text="获取输入", command=get_input)
btn_get.pack(side="left", padx=5)
# ---------- 功能二:爬虫爬取与存储区 ----------
frame_crawl = tk.LabelFrame(root, text="爬虫爬取与存储(多源fallback机制)", padx=10, pady=10)
frame_crawl.pack(padx=10, pady=5, fill="both", expand=True)
btn_frame = tk.Frame(frame_crawl)
btn_frame.pack(pady=5)
btn_crawl = tk.Button(btn_frame, text="爬取并存储", width=14)
btn_crawl.pack(side="left", padx=5)
btn_show = tk.Button(btn_frame, text="查看数据库", width=14)
btn_show.pack(side="left", padx=5)
tk.Label(frame_crawl, text="爬取结果:", font=("微软雅黑", 10, "bold")).pack(anchor="w", padx=5)
# 使用 scrolledtext 替代 Text,支持自动滚动
text_area = scrolledtext.ScrolledText(frame_crawl, width=85, height=20, font=("微软雅黑", 10))
text_area.pack(padx=5, pady=5, fill="both", expand=True)
# ---------- 功能三:退出按钮 ----------
frame_exit = tk.Frame(root)
frame_exit.pack(pady=10)
def on_exit():
root.destroy()
btn_exit = tk.Button(frame_exit, text="退出", command=on_exit)
btn_exit.pack()
# ==================== 按钮回调函数 ====================
def on_crawl():
user_input = entry.get().strip()
btn_crawl.config(state="disabled", text="爬取中...")
root.update()
# 清空结果区
text_area.delete("1.0", tk.END)
# 显示搜索信息
if user_input:
text_area.insert(tk.END, f"===== 搜索关键词:{user_input} =====\n")
text_area.insert(tk.END, f"===== 多源搜索策略:古诗文网API → 古诗文网HTML → 百度汉语 =====\n\n")
else:
text_area.insert(tk.END, "===== 未输入关键词,抓取随机名句 =====\n")
text_area.insert(tk.END, f"===== 源:一言API + 古诗文网 =====\n\n")
# 执行爬取(带超时和异常处理)
try:
data = crawl_quotes(keyword=user_input)
except Exception as e:
text_area.insert(tk.END, f"【错误】爬虫执行失败:{str(e)}\n")
text_area.insert(tk.END, "请检查网络连接后重试。\n")
messagebox.showerror("爬取出错", f"爬虫执行失败:{str(e)}\n请检查网络连接!")
btn_crawl.config(state="normal", text="爬取并存储")
return
# 处理结果
if data:
count = save_to_db(data, keyword=user_input)
text_area.insert(tk.END, f"===== 共爬取并存储 {count} 条数据到数据库 =====\n\n")
for i, item in enumerate(data, 1):
text_area.insert(tk.END, f"【{i}】{item['text']}\n")
text_area.insert(tk.END, f" —— {item['author']}\n\n")
# 根据结果数量给出不同反馈
if len(data) >= 5:
messagebox.showinfo("成功", f"爬取完成!共获取 {count} 条数据(多源合并),已存储到数据库。")
elif len(data) >= 1:
messagebox.showinfo("部分成功", f"爬取完成!共获取 {count} 条数据,已存储到数据库。\n如结果不够精确,请尝试更换关键词。")
else:
messagebox.showwarning("结果有限", "未获取到足够数据,请检查网络或更换关键词!")
else:
# 无结果时的智能提示
if user_input:
text_area.insert(tk.END, f"未找到与「{user_input}」直接匹配的内容。\n")
text_area.insert(tk.END, "可能原因:\n")
text_area.insert(tk.END, "1. 网络连接不稳定,部分数据源未返回结果\n")
text_area.insert(tk.END, "2. 关键词较为生僻,尝试更通用的词语\n")
text_area.insert(tk.END, "3. 建议尝试:理想、青春、奋斗、思念、爱情、人生、月亮、花等常见主题\n")
messagebox.showwarning("无结果",
f"未找到与「{user_input}」匹配的内容。\n"
"建议:尝试更通用的关键词,如「理想」「青春」「奋斗」「思念」等。\n"
"请检查网络连接后重试。")
else:
text_area.insert(tk.END, "未能抓取到名句,请检查网络连接后重试。\n")
messagebox.showwarning("无结果", "未抓取到任何数据,请检查网络连接!")
btn_crawl.config(state="normal", text="爬取并存储")
def on_show():
rows = get_from_db()
if not rows:
messagebox.showwarning("提示", "数据库为空,请先爬取数据!")
return
text_area.delete("1.0", tk.END)
text_area.insert(tk.END, f"===== 数据库中共有 {len(rows)} 条数据 =====\n\n")
for i, (text, author, keyword) in enumerate(rows, 1):
tag_info = f"(搜索关键词:{keyword})" if keyword else ""
text_area.insert(tk.END, f"【{i}】{text}\n")
text_area.insert(tk.END, f" —— {author} {tag_info}\n\n")
btn_crawl.config(command=on_crawl)
btn_show.config(command=on_show)
root.mainloop()
没有代码只能先判断方向。
爬取内容和输入无关,常见原因一般是:请求没带上关键词、参数名写错、GET/POST 用错、搜索结果是 AJAX 动态加载、解析选择器选错。
按这个顺序排查:
[*]打印最终请求地址和参数,看关键词有没有真的发出去。
[*]浏览器按 F12,打开 Network,搜索一次,看真实请求是 GET 还是 POST、参数名是什么。
[*]用相同 headers、params 或 data 重新请求。
[*]打印 r.url、r.status_code、r.text[:500]。
[*]如果是 AJAX,直接请求数据接口,或用 Selenium/Playwright。
[*]检查解析用的选择器是否对应搜索结果区域。
通用模板:
import requests
keyword = "用户输入"
url = "搜索页地址" # 换成真实搜索地址
params = {
"q": keyword, # 参数名必须按 F12 里的真实名字改
}
headers = {
"User-Agent": "Mozilla/5.0 ..."
}
r = requests.get(url, params=params, headers=headers, timeout=10)
print("最终URL:", r.url)
print("状态码:", r.status_code)
print(r.text[:500])
如果真实请求是 POST,把 params 改成 data:
r = requests.post(url, data={"q": keyword}, headers=headers, timeout=10)
解析:
from bs4 import BeautifulSoup
soup = BeautifulSoup(r.text, "html.parser")
items = soup.select("搜索结果条目的选择器")
for item in items:
print(item.get_text(strip=True))
最后:把完整代码、输入关键词、实际爬到的内容、期望内容,以及 F12 里搜索请求的参数名发出来,我帮你直接改。
以上回复来自 FishC AI 小助理,如未能正确解答您的问题,请继续追问。 FishC 发表于 2026-9-26 14:40
没有代码只能先判断方向。
爬取内容和输入无关,常见原因一般是:请求没带上关键词、参数名写错、GET/PO ...
追问:代码如下
import tkinter as tk
from tkinter import messagebox, scrolledtext
import requests
from bs4 import BeautifulSoup
import sqlite3
import urllib.parse
import json
import time
# ==================== 数据库操作(完全保留原有逻辑) ====================
DB_NAME = "quotes.db"
def init_db():
"""初始化数据库,创建表;若旧表缺少 keyword 列则自动补上"""
conn = sqlite3.connect(DB_NAME)
cursor = conn.cursor()
cursor.execute(
"CREATE TABLE IF NOT EXISTS quotes ("
"id INTEGER PRIMARY KEY AUTOINCREMENT, "
"text TEXT, "
"author TEXT, "
"keyword TEXT)"
)
cursor.execute("PRAGMA table_info(quotes)")
columns = for row in cursor.fetchall()]
if "keyword" not in columns:
cursor.execute("ALTER TABLE quotes ADD COLUMN keyword TEXT")
conn.commit()
conn.close()
def save_to_db(data_list, keyword=""):
"""将爬取的数据批量存入数据库"""
if not data_list:
return 0
conn = sqlite3.connect(DB_NAME)
cursor = conn.cursor()
cursor.executemany(
"INSERT INTO quotes (text, author, keyword) VALUES (?, ?, ?)",
[(item["text"], item["author"], keyword) for item in data_list]
)
conn.commit()
conn.close()
return len(data_list)
def get_from_db():
"""从数据库读取所有数据"""
conn = sqlite3.connect(DB_NAME)
cursor = conn.cursor()
cursor.execute("SELECT text, author, keyword FROM quotes")
rows = cursor.fetchall()
conn.close()
return rows
# ==================== 统一请求头(模拟真实浏览器) ====================
HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Safari/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
"Accept-Encoding": "gzip, deflate, br",
"Connection": "keep-alive",
"Upgrade-Insecure-Requests": "1",
}
# ==================== 源1:古诗文网(API接口,返回JSON) ====================
def crawl_from_gushiwen_api(keyword):
"""
古诗文网 - 通过API接口获取名句(JSON格式,无需HTML解析)
优势:结构稳定、中文无乱码、返回数据干净
"""
results = []
api_url = "https://api.gushiwen.cn/v3/gshiwang/shiju/search"
params = {
"key": keyword,
"page": 1,
"size": 20,
}
try:
resp = requests.get(api_url, headers=HEADERS, params=params, timeout=15)
# 使用 resp.text(自动编码检测),不手动 decode
resp.encoding = resp.apparent_encoding or 'utf-8'
data = resp.json()
if data and "data" in data and data["data"]:
for item in data["data"]:
text = item.get("title", "") or item.get("content", "")
author = item.get("author", "") or item.get("dynasty", "") + item.get("author", "")
if text and len(text.strip()) > 2:
results.append({
"text": text.strip(),
"author": author.strip() if author else "佚名"
})
except Exception as e:
raise Exception(f"古诗文网API请求失败: {str(e)}")
return results
# ==================== 源2:古诗文网(HTML页面备用) ====================
def crawl_from_gushiwen_html(keyword):
"""
古诗文网 - 通过HTML页面搜索备用
使用多个CSS选择器策略提高兼容性
"""
results = []
search_url = "https://so.gushiwen.cn/search/"
params = {
"value": keyword,
}
try:
resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
# 关键修复:使用 resp.text 自动处理编码,避免中文乱码
resp.encoding = resp.apparent_encoding or 'utf-8'
html = resp.text
soup = BeautifulSoup(html, "html.parser")
# 策略1:尝试新的选择器 div.content > div.left > div.song > div.son > p
items = soup.select("div.content div.left div.song div.son p a")
if len(items) >= 2:
for i in range(0, len(items) - 1, 2):
text = items.get_text(strip=True)
author = items.get_text(strip=True) if i + 1 < len(items) else ""
if text and len(text.strip()) > 2:
results.append({"text": text.strip(), "author": author.strip()})
# 策略2:尝试另一种选择器
if not results:
items = soup.select("div.sons p a")
if len(items) >= 2:
for i in range(0, len(items) - 1, 2):
text = items.get_text(strip=True)
author = items.get_text(strip=True) if i + 1 < len(items) else ""
if text and len(text.strip()) > 2:
results.append({"text": text.strip(), "author": author.strip()})
# 策略3:尝试 div.left > div.son
if not results:
items = soup.select("div.left div.son p a")
if len(items) >= 2:
for i in range(0, len(items) - 1, 2):
text = items.get_text(strip=True)
author = items.get_text(strip=True) if i + 1 < len(items) else ""
if text and len(text.strip()) > 2:
results.append({"text": text.strip(), "author": author.strip()})
except Exception as e:
raise Exception(f"古诗文网HTML请求失败: {str(e)}")
return results
# ==================== 源3:百度汉语(成语/词语释义) ====================
def crawl_from_baidu_hanyu(keyword):
"""
百度汉语 - 搜索成语/词语释义
作为古诗文网的备用源
"""
results = []
search_url = "https://dict.baidu.com/s"
params = {
"wd": keyword,
}
try:
resp = requests.get(search_url, headers=HEADERS, params=params, timeout=15)
resp.encoding = resp.apparent_encoding or 'utf-8'
html = resp.text
soup = BeautifulSoup(html, "html.parser")
# 尝试获取百度汉语的释义内容
# 策略:查找包含释义的div
definitions = soup.select("div.vc-paragraph-inner, div.vc-content, div.result")
for div in definitions:
text = div.get_text(strip=True)
if text and len(text) > 5 and keyword in text:
results.append({"text": text[:200], "author": "百度汉语"})
# 如果上面的选择器没匹配到,尝试更宽泛的搜索
if not results:
all_texts = soup.find_all(string=True)
for t in all_texts:
t_stripped = t.strip()
if keyword in t_stripped and len(t_stripped) > 10:
results.append({"text": t_stripped[:200], "author": "百度汉语"})
if len(results) >= 5:
break
except Exception as e:
raise Exception(f"百度汉语请求失败: {str(e)}")
return results
# ==================== 源4:一言API(随机名句) ====================
def crawl_from_hitokoto():
"""
一言API - 获取随机名句(不依赖关键词,作为兜底源)
"""
results = []
urls = [
"https://v1.hitokoto.cn/?c=a",# 诗词古典
"https://v1.hitokoto.cn/?c=f",# 哲学思想
"https://v1.hitokoto.cn/?c=d",# 原创
]
for url in urls:
try:
resp = requests.get(url, headers=HEADERS, timeout=10)
resp.encoding = resp.apparent_encoding or 'utf-8'
data = resp.json()
if "hitokoto" in data and data["hitokoto"]:
results.append({
"text": data["hitokoto"],
"author": data.get("from", "") or data.get("creator", "佚名")
})
except:
continue
return results
# ==================== 主爬虫函数(多源fallback + 关键词验证) ====================
def crawl_quotes(keyword=""):
"""
多源爬虫主函数:
1. 有关键词:优先古诗文网API -> HTML备用 -> 百度汉语
2. 无关键词:返回一言API随机名句
3. 自动编码检测,避免中文乱码
4. 关键词相关性验证,确保返回内容与搜索词相关
5. 去重、异常捕获
"""
keyword = keyword.strip()
results = []
seen = set()
def add_result(item):
"""添加结果并去重"""
text = item.get("text", "").strip()
author = item.get("author", "").strip()
if not text or len(text) < 3:
return
key = (text[:50], author[:20])
if key not in seen:
seen.add(key)
results.append({"text": text, "author": author})
def filter_by_keyword(items, kw):
"""过滤与关键词相关的结果"""
if not kw:
return items
filtered = []
kw_lower = kw.lower()
for item in items:
text = item.get("text", "")
author = item.get("author", "")
# 只要文本或作者中包含关键词就算相关
if kw_lower in text.lower() or kw_lower in author.lower():
filtered.append(item)
elif len(filtered) < 3:
# 允许部分匹配的结果(防止严格过滤导致无结果)
filtered.append(item)
return filtered if filtered else items# 兜底:至少返回一些结果
try:
if keyword:
# === 有关键词:多源搜索 ===
# 源1:古诗文网API
try:
api_results = crawl_from_gushiwen_api(keyword)
for item in api_results:
add_result(item)
except Exception:
pass
# 源2:古诗文网HTML备用
if len(results) < 3:
try:
html_results = crawl_from_gushiwen_html(keyword)
for item in html_results:
add_result(item)
except Exception:
pass
# 源3:百度汉语
if len(results) < 3:
try:
baidu_results = crawl_from_baidu_hanyu(keyword)
for item in baidu_results:
add_result(item)
except Exception:
pass
# 关键词相关性过滤
results = filter_by_keyword(results, keyword)
else:
# === 无关键词:返回随机名句 ===
results = crawl_from_hitokoto()
# 如果一言API也失败,尝试古诗文网默认名句
if not results:
try:
api_results = crawl_from_gushiwen_api("")
for item in api_results:
add_result(item)
results = api_results
except Exception:
pass
except Exception as e:
raise Exception(f"爬虫主流程异常: {str(e)}")
# 最终去重
final_results = []
final_seen = set()
for item in results:
key = item["text"][:50]
if key not in final_seen:
final_seen.add(key)
final_results.append(item)
return final_results
# ==================== GUI界面 ====================
init_db()
root = tk.Tk()
root.title("简易AI 8.0【多源爬虫 · 中文关键词搜索 · 修复版】")
root.geometry("750x620")
# ---------- 功能一:用户输入获取区 ----------
frame_input = tk.LabelFrame(root, text="用户输入获取(支持中文,如:理想、青春、奋斗、爱情)", padx=10, pady=10)
frame_input.pack(padx=10, pady=5, fill="x")
entry = tk.Entry(frame_input, width=45)
entry.pack(side="left", padx=5)
def get_input():
content = entry.get()
if content:
messagebox.showinfo("输入结果", f"你输入了:{content}")
else:
messagebox.showwarning("提示", "输入框为空!")
btn_get = tk.Button(frame_input, text="获取输入", command=get_input)
btn_get.pack(side="left", padx=5)
# ---------- 功能二:爬虫爬取与存储区 ----------
frame_crawl = tk.LabelFrame(root, text="爬虫爬取与存储(多源fallback机制)", padx=10, pady=10)
frame_crawl.pack(padx=10, pady=5, fill="both", expand=True)
btn_frame = tk.Frame(frame_crawl)
btn_frame.pack(pady=5)
btn_crawl = tk.Button(btn_frame, text="爬取并存储", width=14)
btn_crawl.pack(side="left", padx=5)
btn_show = tk.Button(btn_frame, text="查看数据库", width=14)
btn_show.pack(side="left", padx=5)
tk.Label(frame_crawl, text="爬取结果:", font=("微软雅黑", 10, "bold")).pack(anchor="w", padx=5)
# 使用 scrolledtext 替代 Text,支持自动滚动
text_area = scrolledtext.ScrolledText(frame_crawl, width=85, height=20, font=("微软雅黑", 10))
text_area.pack(padx=5, pady=5, fill="both", expand=True)
# ---------- 功能三:退出按钮 ----------
frame_exit = tk.Frame(root)
frame_exit.pack(pady=10)
def on_exit():
root.destroy()
btn_exit = tk.Button(frame_exit, text="退出", command=on_exit)
btn_exit.pack()
# ==================== 按钮回调函数 ====================
def on_crawl():
user_input = entry.get().strip()
btn_crawl.config(state="disabled", text="爬取中...")
root.update()
# 清空结果区
text_area.delete("1.0", tk.END)
# 显示搜索信息
if user_input:
text_area.insert(tk.END, f"===== 搜索关键词:{user_input} =====\n")
text_area.insert(tk.END, f"===== 多源搜索策略:古诗文网API → 古诗文网HTML → 百度汉语 =====\n\n")
else:
text_area.insert(tk.END, "===== 未输入关键词,抓取随机名句 =====\n")
text_area.insert(tk.END, f"===== 源:一言API + 古诗文网 =====\n\n")
# 执行爬取(带超时和异常处理)
try:
data = crawl_quotes(keyword=user_input)
except Exception as e:
text_area.insert(tk.END, f"【错误】爬虫执行失败:{str(e)}\n")
text_area.insert(tk.END, "请检查网络连接后重试。\n")
messagebox.showerror("爬取出错", f"爬虫执行失败:{str(e)}\n请检查网络连接!")
btn_crawl.config(state="normal", text="爬取并存储")
return
# 处理结果
if data:
count = save_to_db(data, keyword=user_input)
text_area.insert(tk.END, f"===== 共爬取并存储 {count} 条数据到数据库 =====\n\n")
for i, item in enumerate(data, 1):
text_area.insert(tk.END, f"【{i}】{item['text']}\n")
text_area.insert(tk.END, f" —— {item['author']}\n\n")
# 根据结果数量给出不同反馈
if len(data) >= 5:
messagebox.showinfo("成功", f"爬取完成!共获取 {count} 条数据(多源合并),已存储到数据库。")
elif len(data) >= 1:
messagebox.showinfo("部分成功", f"爬取完成!共获取 {count} 条数据,已存储到数据库。\n如结果不够精确,请尝试更换关键词。")
else:
messagebox.showwarning("结果有限", "未获取到足够数据,请检查网络或更换关键词!")
else:
# 无结果时的智能提示
if user_input:
text_area.insert(tk.END, f"未找到与「{user_input}」直接匹配的内容。\n")
text_area.insert(tk.END, "可能原因:\n")
text_area.insert(tk.END, "1. 网络连接不稳定,部分数据源未返回结果\n")
text_area.insert(tk.END, "2. 关键词较为生僻,尝试更通用的词语\n")
text_area.insert(tk.END, "3. 建议尝试:理想、青春、奋斗、思念、爱情、人生、月亮、花等常见主题\n")
messagebox.showwarning("无结果",
f"未找到与「{user_input}」匹配的内容。\n"
"建议:尝试更通用的关键词,如「理想」「青春」「奋斗」「思念」等。\n"
"请检查网络连接后重试。")
else:
text_area.insert(tk.END, "未能抓取到名句,请检查网络连接后重试。\n")
messagebox.showwarning("无结果", "未抓取到任何数据,请检查网络连接!")
btn_crawl.config(state="normal", text="爬取并存储")
def on_show():
rows = get_from_db()
if not rows:
messagebox.showwarning("提示", "数据库为空,请先爬取数据!")
return
text_area.delete("1.0", tk.END)
text_area.insert(tk.END, f"===== 数据库中共有 {len(rows)} 条数据 =====\n\n")
for i, (text, author, keyword) in enumerate(rows, 1):
tag_info = f"(搜索关键词:{keyword})" if keyword else ""
text_area.insert(tk.END, f"【{i}】{text}\n")
text_area.insert(tk.END, f" —— {author} {tag_info}\n\n")
btn_crawl.config(command=on_crawl)
btn_show.config(command=on_show)
root.mainloop() 雨澄润 发表于 2026-09-26 16:47
FishC 发表于 2026-9-26 14:40
没有代码只能先判断方 ...
问题出在几处“结果对象”和“字段取值”上,关键词根本没被正确使用,所以爬回来的内容跟输入无关。下面直接列出核心 bug,并给出修正后的完整代码。
一、核心问题
[*] 古诗文网 API 地址 api.gushiwen.cn/v3/... 基本不存在,resp.json() 会抛异常被 except 吞掉,结果永远落到后面的“兜底源”。
[*] 兜底源里的“一言 API”是随机名句,跟关键词毫无关系,一兜底就变成“内容与输入无关”。
[*] save_to_db 里写的是 [(item, item, keyword) for item in data_list],把整个字典当文本存了,应该取 item["text"] 和 item["author"]。
[*] 界面显示时 f"【{i}】{item}" 打印的是整个字典,应打印 item["text"]。
[*] filter_by_keyword 里有一段 elif len(filtered) < 3: filtered.append(item),会把不相关的结果强行塞进去,等于白过滤。
[*] 最终去重 key = item 用字典当 key,字典不可哈希,会报 TypeError。
[*] 请求头里 br 编码没装 brotli 时容易乱码,建议去掉。
二、修正后的完整代码
import tkinter as tk
from tkinter import messagebox, scrolledtext
import requests
from bs4 import BeautifulSoup
import sqlite3
DB_NAME = "quotes.db"
HEADERS = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/120.0.0.0 Safari/537.36",
"Accept-Language": "zh-CN,zh;q=0.9",
}
# ==================== 数据库 ====================
def init_db():
conn = sqlite3.connect(DB_NAME)
cur = conn.cursor()
cur.execute(
"CREATE TABLE IF NOT EXISTS quotes ("
"id INTEGER PRIMARY KEY AUTOINCREMENT, "
"text TEXT, author TEXT, keyword TEXT)"
)
conn.commit()
conn.close()
def save_to_db(data_list, keyword=""):
if not data_list:
return 0
conn = sqlite3.connect(DB_NAME)
cur = conn.cursor()
cur.executemany(
"INSERT INTO quotes (text, author, keyword) VALUES (?, ?, ?)",
[(it["text"], it["author"], keyword) for it in data_list]
)
conn.commit()
conn.close()
return len(data_list)
def get_from_db():
conn = sqlite3.connect(DB_NAME)
cur = conn.cursor()
cur.execute("SELECT text, author, keyword FROM quotes")
rows = cur.fetchall()
conn.close()
return rows
# ==================== 源1:古诗文网搜索页(按关键词) ====================
def crawl_gushiwen(keyword):
"""古诗文网搜索页,返回与关键词相关的诗句"""
results = []
url = "https://so.gushiwen.cn/search.aspx"
params = {"value": keyword, "valuej": ""}
resp = requests.get(url, headers=HEADERS, params=params, timeout=15)
resp.encoding = resp.apparent_encoding or "utf-8"
soup = BeautifulSoup(resp.text, "html.parser")
# 搜索结果结构:div.sons > div.cont > p(诗句) + p(出处/作者)
for cont in soup.select("div.sons div.cont"):
ps = cont.find_all("p")
if not ps:
continue
text = ps.get_text(strip=True)
author = ps.get_text(strip=True) if len(ps) > 1 else "佚名"
if len(text) > 2:
results.append({"text": text, "author": author})
return results
# ==================== 源2:一言(仅无关键词时使用) ====================
def crawl_hitokoto():
"""一言随机名句,只在用户没输入关键词时用"""
url = "https://v1.hitokoto.cn/"
resp = requests.get(url, headers=HEADERS, timeout=10)
data = resp.json()
if data.get("hitokoto"):
return [{
"text": data["hitokoto"],
"author": data.get("from") or data.get("creator") or "佚名",
}]
return []
# ==================== 主爬虫 ====================
def crawl_quotes(keyword=""):
keyword = keyword.strip()
results = []
seen = set()
def add(item):
text = (item.get("text") or "").strip()
author = (item.get("author") or "").strip() or "佚名"
if len(text) < 3:
return
key = (text, author)
if key not in seen:
seen.add(key)
results.append({"text": text, "author": author})
if keyword:
# 有输入:只走古诗文网搜索页,保证结果与关键词相关
for it in crawl_gushiwen(keyword):
add(it)
else:
# 无输入:返回随机名句
for it in crawl_hitokoto():
add(it)
return results
# ==================== GUI ====================
init_db()
root = tk.Tk()
root.title("简易AI 8.0【多源爬虫 · 中文关键词搜索 · 修复版】")
root.geometry("750x620")
frame_input = tk.LabelFrame(root, text="用户输入获取(支持中文,如:理想、青春、奋斗、爱情)", padx=10, pady=10)
frame_input.pack(padx=10, pady=5, fill="x")
entry = tk.Entry(frame_input, width=45)
entry.pack(side="left", padx=5)
def get_input():
content = entry.get()
if content:
messagebox.showinfo("输入结果", f"你输入了:{content}")
else:
messagebox.showwarning("提示", "输入框为空!")
btn_get = tk.Button(frame_input, text="获取输入", command=get_input)
btn_get.pack(side="left", padx=5)
frame_crawl = tk.LabelFrame(root, text="爬虫爬取与存储", padx=10, pady=10)
frame_crawl.pack(padx=10, pady=5, fill="both", expand=True)
btn_frame = tk.Frame(frame_crawl)
btn_frame.pack(pady=5)
btn_crawl = tk.Button(btn_frame, text="爬取并存储", width=14)
btn_crawl.pack(side="left", padx=5)
btn_show = tk.Button(btn_frame, text="查看数据库", width=14)
btn_show.pack(side="left", padx=5)
tk.Label(frame_crawl, text="爬取结果:", font=("微软雅黑", 10, "bold")).pack(anchor="w", padx=5)
text_area = scrolledtext.ScrolledText(frame_crawl, width=85, height=20, font=("微软雅黑", 10))
text_area.pack(padx=5, pady=5, fill="both", expand=True)
frame_exit = tk.Frame(root)
frame_exit.pack(pady=10)
def on_exit():
root.destroy()
btn_exit = tk.Button(frame_exit, text="退出", command=on_exit)
btn_exit.pack()
def on_crawl():
user_input = entry.get().strip()
btn_crawl.config(state="disabled", text="爬取中...")
root.update()
text_area.delete("1.0", tk.END)
if user_input:
text_area.insert(tk.END, f"===== 搜索关键词:{user_input} =====\n\n")
else:
text_area.insert(tk.END, "===== 未输入关键词,抓取随机名句 =====\n\n")
try:
data = crawl_quotes(keyword=user_input)
except Exception as e:
text_area.insert(tk.END, f"【错误】爬虫执行失败:{e}\n")
messagebox.showerror("爬取出错", f"爬虫执行失败:{e}\n请检查网络连接!")
btn_crawl.config(state="normal", text="爬取并存储")
return
if data:
count = save_to_db(data, keyword=user_input)
text_area.insert(tk.END, f"===== 共爬取并存储 {count} 条数据 =====\n\n")
for i, item in enumerate(data, 1):
text_area.insert(tk.END, f"【{i}】{item['text']}\n")
text_area.insert(tk.END, f" —— {item['author']}\n\n")
messagebox.showinfo("成功", f"爬取完成!共获取 {count} 条数据,已存储到数据库。")
else:
text_area.insert(tk.END, f"未找到与「{user_input}」相关的内容。\n")
messagebox.showwarning("无结果", f"未找到与「{user_input}」相关的内容。")
btn_crawl.config(state="normal", text="爬取并存储")
def on_show():
rows = get_from_db()
if not rows:
messagebox.showwarning("提示", "数据库为空,请先爬取数据!")
return
text_area.delete("1.0", tk.END)
text_area.insert(tk.END, f"===== 数据库中共有 {len(rows)} 条数据 =====\n\n")
for i, (text, author, keyword) in enumerate(rows, 1):
tag = f"(关键词:{keyword})" if keyword else ""
text_area.insert(tk.END, f"【{i}】{text}\n")
text_area.insert(tk.END, f" —— {author} {tag}\n\n")
btn_crawl.config(command=on_crawl)
btn_show.config(command=on_show)
root.mainloop()
三、关键改动说明
[*] 删掉了假的古诗文网 API 和百度汉语源,只保留真正能按关键词搜索的 so.gushiwen.cn/search.aspx。
[*] 一言 API 只在没有输入关键词时才调用,避免污染结果。
[*] save_to_db 正确取 item["text"] / item["author"]。
[*] 界面显示改成取字段,不再打印整个字典。
[*] 去重用 (text, author) 元组,避免不可哈希错误。
[*] 请求头去掉 br,防止编码乱码。
如果运行后古诗文网返回 0 条,一般是它触发了反爬或页面结构改了,可以先用浏览器打开搜索页 F12 看一下 div.sons div.cont 是否还在,把真实选择器发出来我再帮你调。
以上回复来自 FishC AI 小助理,如未能正确解答您的问题,请继续追问。 我去试了一下,没有问题呀,哪里不准确,请提供一下你的输入案例
页:
[1]