diff --git a/services/search_service.py b/services/search_service.py index a310bd0..b5ed215 100644 --- a/services/search_service.py +++ b/services/search_service.py @@ -4,6 +4,10 @@ import requests from bs4 import BeautifulSoup import json +import subprocess +import os +import re +import urllib.parse from datetime import datetime from config import Config from models.database import db @@ -13,19 +17,143 @@ class SearchService: self.timeout = Config.SEARCH_TIMEOUT self.max_results = Config.SEARCH_MAX_RESULTS + def _run_browser(self, *args, timeout=30000): + """运行 agent-browser 命令""" + env = os.environ.copy() + env['XDG_RUNTIME_DIR'] = '/tmp/agent-browser-runtime' + os.makedirs(env['XDG_RUNTIME_DIR'], exist_ok=True) + + cmd = ['agent-browser'] + list(args) + result = subprocess.run( + cmd, + capture_output=True, + text=True, + env=env, + timeout=timeout // 1000 + 5 + ) + return result.stdout, result.stderr, result.returncode + def search_internet(self, keyword, max_results=None): """ - 从互联网搜索(使用搜索API或爬虫) - 这里暂时使用简单的搜索模拟 + 从互联网搜索(使用 agent-browser 浏览器自动化) """ max_results = max_results or self.max_results - - # TODO: 接入真实的搜索API(如Google Custom Search、Bing等) - # 这里先返回空列表,等待后续接入真实API results = [] + try: + # 1. 打开 Bing 搜索 + encoded_keyword = urllib.parse.quote(keyword) + search_url = f"https://www.bing.com/search?q={encoded_keyword}" + + stdout, stderr, code = self._run_browser('open', search_url, '--timeout', '20000') + if code != 0: + print(f"打开搜索页面失败: {stderr}") + return results + + # 等待页面加载 + stdout, stderr, code = self._run_browser('wait', '5000') + + # 2. 获取搜索结果页面结构 (JSON 格式) + stdout, stderr, code = self._run_browser('snapshot', '--json', '--timeout', '30000') + if code != 0: + print(f"获取页面结构失败: {stderr}") + return results + + # 3. 解析 JSON 提取搜索结果 + try: + data = json.loads(stdout) + except json.JSONDecodeError: + print(f"解析 JSON 失败: {stdout[:500]}") + return results + + # 4. 从 accessibility tree 中提取搜索结果 + # Bing 搜索结果在 main[aria-label="搜索结果"] 区域内 + results = self._parse_bing_results(data, max_results) + + # 5. 关闭浏览器 + self._run_browser('close') + + except subprocess.TimeoutExpired: + print(f"搜索超时: {keyword}") + except Exception as e: + print(f"搜索出错: {str(e)}") + # 尝试关闭浏览器 + try: + self._run_browser('close') + except: + pass + return results + def _parse_bing_results(self, snapshot_data, max_results=10): + """ + 从 Bing 搜索结果的 snapshot 中解析出标题和链接 + + snapshot_data 是 agent-browser snapshot --json 的输出 + 结构: {success, data: {snapshot: "文本格式的 accessibility tree"}, error} + """ + results = [] + + # 获取 snapshot 文本 + snapshot = snapshot_data.get('data', {}).get('snapshot', '') + if not snapshot: + return results + + # 解析 accessibility tree 文本 + in_results = False + refs = [] # 存储 (title, ref) 元组 + lines = snapshot.split('\n') + + for i, line in enumerate(lines): + line = line.strip() + + # 进入搜索结果区域 + if 'main "搜索结果"' in line: + in_results = True + continue + + # 离开搜索结果区域 + if in_results and line.startswith('- ') and 'main' in line and '搜索结果' not in line: + break + + if not in_results: + continue + + # 匹配标题链接:link "标题文字" [ref=eXX] + # 需要过滤域名链接(如 "zhihu.com")和短链接 + if 'link "' in line and '[ref=' in line: + match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line) + if match: + title = match.group(1) + ref = match.group(2) + # 过滤短标题(域名链接如 "zhihu.com") + if len(title) > 20 and '.' not in title[:10]: # 不是域名格式 + refs.append((title, ref)) + + # 获取每个结果的 URL + for title, ref in refs[:max_results]: + url = self._get_link_url(ref) + if url and 'bing.com/search' not in url: # 过滤搜索结果页本身的链接 + results.append({ + 'title': title, + 'url': url, + 'snippet': '', + 'source': 'bing' + }) + + return results + + def _get_link_url(self, ref): + """通过 agent-browser 获取链接的 URL""" + try: + stdout, stderr, code = self._run_browser('get', 'attr', f'@{ref}', 'href', '--json', '--timeout', '5000') + if code == 0 and stdout: + data = json.loads(stdout) + return data.get('data', {}).get('value', '') + except Exception as e: + print(f"获取 URL 失败 (ref={ref}): {e}") + return None + def fetch_url_content(self, url): """抓取网页内容""" try: