diff --git a/routes/articles.py b/routes/articles.py index 1e27d5e..a166480 100644 --- a/routes/articles.py +++ b/routes/articles.py @@ -114,7 +114,7 @@ def fetch_article(): result = search_service.fetch_url_content(url) - if result: + if result and result.get('success'): # 自动保存到内容库 article_id = search_service.save_to_articles( product_names=data.get('product_names', [result['title']]), @@ -132,7 +132,8 @@ def fetch_article(): 'data': result }) else: - return jsonify({'error': '抓取失败'}), 500 + error_msg = result.get('error', '抓取失败') if result else '抓取失败' + return jsonify({'success': False, 'error': error_msg}), 500 @bp.route('/internet-search', methods=['POST']) def internet_search(): diff --git a/services/search_service.py b/services/search_service.py index be5a40c..6635cea 100644 --- a/services/search_service.py +++ b/services/search_service.py @@ -151,14 +151,14 @@ class SearchService: continue # 匹配标题链接:link "标题文字" [ref=eXX] - # 需要过滤域名链接(如 "zhihu.com")和短链接 + # 过滤域名链接(如 "zhihu.com")和短链接 if 'link "' in line and '[ref=' in line: match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line) if match: title = match.group(1) ref = match.group(2) - # 过滤短标题(域名链接如 "zhihu.com") - if len(title) > 20 and '.' not in title[:10]: # 不是域名格式 + # 放宽过滤条件:只要不是纯域名格式就保留 + if not (title.endswith('.com') or title.endswith('.cn') or title.endswith('.net')): refs.append((title, ref)) # 获取每个结果的 URL @@ -225,13 +225,18 @@ class SearchService: def fetch_url_content(self, url): """抓取网页内容(使用 agent-browser 浏览器方式,绕过反爬虫)""" + error_message = None try: # 使用浏览器方式抓取,增加超时时间到60秒 stdout, stderr, code = self._run_browser('open', url, '--timeout', '60000') if code != 0: + error_message = stderr.strip() if stderr else '浏览器打开页面失败' print(f"打开页面失败: {stderr}") # 浏览器失败,尝试使用 requests 备用方案 - return self._fetch_with_requests(url) + result = self._fetch_with_requests(url) + if result: + return result + return {'success': False, 'error': error_message} # 等待页面加载(增加到10秒) self._run_browser('wait', '10000') @@ -263,6 +268,7 @@ class SearchService: description = text[:200].strip() if text else '' return { + 'success': True, 'title': title, 'description': description, 'content': text, @@ -270,13 +276,14 @@ class SearchService: 'fetch_date': datetime.now().isoformat() } except Exception as e: - print(f"抓取URL失败: {url}, 错误: {str(e)}") + error_message = str(e) + print(f"抓取URL失败: {url}, 错误: {error_message}") # 尝试关闭浏览器 try: self._run_browser('close') except: pass - return None + return {'success': False, 'error': error_message} def _extract_text_from_snapshot(self, snapshot): """从 accessibility tree snapshot 中提取文本内容""" @@ -340,6 +347,7 @@ class SearchService: description = text[:200].strip() if text else '' return { + 'success': True, 'title': title, 'description': description, 'content': text,