修复搜索抓取相关问题
1. 放宽搜索结果过滤条件,返回更多结果 2. 抓取返回详细错误信息(如超时、网络错误等) 3. 错误信息记录到失败URL库中 4. 统一返回格式包含success字段
This commit is contained in:
@@ -151,14 +151,14 @@ class SearchService:
|
||||
continue
|
||||
|
||||
# 匹配标题链接:link "标题文字" [ref=eXX]
|
||||
# 需要过滤域名链接(如 "zhihu.com")和短链接
|
||||
# 过滤域名链接(如 "zhihu.com")和短链接
|
||||
if 'link "' in line and '[ref=' in line:
|
||||
match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line)
|
||||
if match:
|
||||
title = match.group(1)
|
||||
ref = match.group(2)
|
||||
# 过滤短标题(域名链接如 "zhihu.com")
|
||||
if len(title) > 20 and '.' not in title[:10]: # 不是域名格式
|
||||
# 放宽过滤条件:只要不是纯域名格式就保留
|
||||
if not (title.endswith('.com') or title.endswith('.cn') or title.endswith('.net')):
|
||||
refs.append((title, ref))
|
||||
|
||||
# 获取每个结果的 URL
|
||||
@@ -225,13 +225,18 @@ class SearchService:
|
||||
|
||||
def fetch_url_content(self, url):
|
||||
"""抓取网页内容(使用 agent-browser 浏览器方式,绕过反爬虫)"""
|
||||
error_message = None
|
||||
try:
|
||||
# 使用浏览器方式抓取,增加超时时间到60秒
|
||||
stdout, stderr, code = self._run_browser('open', url, '--timeout', '60000')
|
||||
if code != 0:
|
||||
error_message = stderr.strip() if stderr else '浏览器打开页面失败'
|
||||
print(f"打开页面失败: {stderr}")
|
||||
# 浏览器失败,尝试使用 requests 备用方案
|
||||
return self._fetch_with_requests(url)
|
||||
result = self._fetch_with_requests(url)
|
||||
if result:
|
||||
return result
|
||||
return {'success': False, 'error': error_message}
|
||||
|
||||
# 等待页面加载(增加到10秒)
|
||||
self._run_browser('wait', '10000')
|
||||
@@ -263,6 +268,7 @@ class SearchService:
|
||||
description = text[:200].strip() if text else ''
|
||||
|
||||
return {
|
||||
'success': True,
|
||||
'title': title,
|
||||
'description': description,
|
||||
'content': text,
|
||||
@@ -270,13 +276,14 @@ class SearchService:
|
||||
'fetch_date': datetime.now().isoformat()
|
||||
}
|
||||
except Exception as e:
|
||||
print(f"抓取URL失败: {url}, 错误: {str(e)}")
|
||||
error_message = str(e)
|
||||
print(f"抓取URL失败: {url}, 错误: {error_message}")
|
||||
# 尝试关闭浏览器
|
||||
try:
|
||||
self._run_browser('close')
|
||||
except:
|
||||
pass
|
||||
return None
|
||||
return {'success': False, 'error': error_message}
|
||||
|
||||
def _extract_text_from_snapshot(self, snapshot):
|
||||
"""从 accessibility tree snapshot 中提取文本内容"""
|
||||
@@ -340,6 +347,7 @@ class SearchService:
|
||||
description = text[:200].strip() if text else ''
|
||||
|
||||
return {
|
||||
'success': True,
|
||||
'title': title,
|
||||
'description': description,
|
||||
'content': text,
|
||||
|
||||
Reference in New Issue
Block a user