- 使用 agent-browser 浏览器自动化进行 Bing 搜索 - 解析 accessibility tree 提取搜索结果 - 自动获取每个结果的 URL - 支持设置最大结果数量
227 lines
7.9 KiB
Python
227 lines
7.9 KiB
Python
"""
|
||
搜索服务 - 从内容库和互联网搜索数据
|
||
"""
|
||
import requests
|
||
from bs4 import BeautifulSoup
|
||
import json
|
||
import subprocess
|
||
import os
|
||
import re
|
||
import urllib.parse
|
||
from datetime import datetime
|
||
from config import Config
|
||
from models.database import db
|
||
|
||
class SearchService:
|
||
def __init__(self):
|
||
self.timeout = Config.SEARCH_TIMEOUT
|
||
self.max_results = Config.SEARCH_MAX_RESULTS
|
||
|
||
def _run_browser(self, *args, timeout=30000):
|
||
"""运行 agent-browser 命令"""
|
||
env = os.environ.copy()
|
||
env['XDG_RUNTIME_DIR'] = '/tmp/agent-browser-runtime'
|
||
os.makedirs(env['XDG_RUNTIME_DIR'], exist_ok=True)
|
||
|
||
cmd = ['agent-browser'] + list(args)
|
||
result = subprocess.run(
|
||
cmd,
|
||
capture_output=True,
|
||
text=True,
|
||
env=env,
|
||
timeout=timeout // 1000 + 5
|
||
)
|
||
return result.stdout, result.stderr, result.returncode
|
||
|
||
def search_internet(self, keyword, max_results=None):
|
||
"""
|
||
从互联网搜索(使用 agent-browser 浏览器自动化)
|
||
"""
|
||
max_results = max_results or self.max_results
|
||
results = []
|
||
|
||
try:
|
||
# 1. 打开 Bing 搜索
|
||
encoded_keyword = urllib.parse.quote(keyword)
|
||
search_url = f"https://www.bing.com/search?q={encoded_keyword}"
|
||
|
||
stdout, stderr, code = self._run_browser('open', search_url, '--timeout', '20000')
|
||
if code != 0:
|
||
print(f"打开搜索页面失败: {stderr}")
|
||
return results
|
||
|
||
# 等待页面加载
|
||
stdout, stderr, code = self._run_browser('wait', '5000')
|
||
|
||
# 2. 获取搜索结果页面结构 (JSON 格式)
|
||
stdout, stderr, code = self._run_browser('snapshot', '--json', '--timeout', '30000')
|
||
if code != 0:
|
||
print(f"获取页面结构失败: {stderr}")
|
||
return results
|
||
|
||
# 3. 解析 JSON 提取搜索结果
|
||
try:
|
||
data = json.loads(stdout)
|
||
except json.JSONDecodeError:
|
||
print(f"解析 JSON 失败: {stdout[:500]}")
|
||
return results
|
||
|
||
# 4. 从 accessibility tree 中提取搜索结果
|
||
# Bing 搜索结果在 main[aria-label="搜索结果"] 区域内
|
||
results = self._parse_bing_results(data, max_results)
|
||
|
||
# 5. 关闭浏览器
|
||
self._run_browser('close')
|
||
|
||
except subprocess.TimeoutExpired:
|
||
print(f"搜索超时: {keyword}")
|
||
except Exception as e:
|
||
print(f"搜索出错: {str(e)}")
|
||
# 尝试关闭浏览器
|
||
try:
|
||
self._run_browser('close')
|
||
except:
|
||
pass
|
||
|
||
return results
|
||
|
||
def _parse_bing_results(self, snapshot_data, max_results=10):
|
||
"""
|
||
从 Bing 搜索结果的 snapshot 中解析出标题和链接
|
||
|
||
snapshot_data 是 agent-browser snapshot --json 的输出
|
||
结构: {success, data: {snapshot: "文本格式的 accessibility tree"}, error}
|
||
"""
|
||
results = []
|
||
|
||
# 获取 snapshot 文本
|
||
snapshot = snapshot_data.get('data', {}).get('snapshot', '')
|
||
if not snapshot:
|
||
return results
|
||
|
||
# 解析 accessibility tree 文本
|
||
in_results = False
|
||
refs = [] # 存储 (title, ref) 元组
|
||
lines = snapshot.split('\n')
|
||
|
||
for i, line in enumerate(lines):
|
||
line = line.strip()
|
||
|
||
# 进入搜索结果区域
|
||
if 'main "搜索结果"' in line:
|
||
in_results = True
|
||
continue
|
||
|
||
# 离开搜索结果区域
|
||
if in_results and line.startswith('- ') and 'main' in line and '搜索结果' not in line:
|
||
break
|
||
|
||
if not in_results:
|
||
continue
|
||
|
||
# 匹配标题链接:link "标题文字" [ref=eXX]
|
||
# 需要过滤域名链接(如 "zhihu.com")和短链接
|
||
if 'link "' in line and '[ref=' in line:
|
||
match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line)
|
||
if match:
|
||
title = match.group(1)
|
||
ref = match.group(2)
|
||
# 过滤短标题(域名链接如 "zhihu.com")
|
||
if len(title) > 20 and '.' not in title[:10]: # 不是域名格式
|
||
refs.append((title, ref))
|
||
|
||
# 获取每个结果的 URL
|
||
for title, ref in refs[:max_results]:
|
||
url = self._get_link_url(ref)
|
||
if url and 'bing.com/search' not in url: # 过滤搜索结果页本身的链接
|
||
results.append({
|
||
'title': title,
|
||
'url': url,
|
||
'snippet': '',
|
||
'source': 'bing'
|
||
})
|
||
|
||
return results
|
||
|
||
def _get_link_url(self, ref):
|
||
"""通过 agent-browser 获取链接的 URL"""
|
||
try:
|
||
stdout, stderr, code = self._run_browser('get', 'attr', f'@{ref}', 'href', '--json', '--timeout', '5000')
|
||
if code == 0 and stdout:
|
||
data = json.loads(stdout)
|
||
return data.get('data', {}).get('value', '')
|
||
except Exception as e:
|
||
print(f"获取 URL 失败 (ref={ref}): {e}")
|
||
return None
|
||
|
||
def fetch_url_content(self, url):
|
||
"""抓取网页内容"""
|
||
try:
|
||
headers = {
|
||
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
|
||
}
|
||
response = requests.get(url, headers=headers, timeout=self.timeout)
|
||
response.raise_for_status()
|
||
|
||
soup = BeautifulSoup(response.text, 'lxml')
|
||
|
||
# 提取标题
|
||
title = soup.find('title')
|
||
title = title.text.strip() if title else ''
|
||
|
||
# 提取正文(简单提取,可优化)
|
||
# 移除脚本和样式
|
||
for script in soup(['script', 'style']):
|
||
script.decompose()
|
||
|
||
# 提取文本
|
||
text = soup.get_text(separator='\n', strip=True)
|
||
|
||
# 提取元数据
|
||
meta_desc = soup.find('meta', attrs={'name': 'description'})
|
||
description = meta_desc['content'] if meta_desc else ''
|
||
|
||
return {
|
||
'title': title,
|
||
'description': description,
|
||
'content': text,
|
||
'url': url,
|
||
'fetch_date': datetime.now().isoformat()
|
||
}
|
||
except Exception as e:
|
||
print(f"抓取URL失败: {url}, 错误: {str(e)}")
|
||
return None
|
||
|
||
def search_articles(self, keyword, category=None):
|
||
"""从内容库搜索"""
|
||
return db.search_articles(keyword, category)
|
||
|
||
def search_all(self, keyword, category=None, include_internet=True):
|
||
"""
|
||
综合搜索:内容库 + 互联网
|
||
"""
|
||
results = {
|
||
'articles': [],
|
||
'internet': [],
|
||
'total': 0
|
||
}
|
||
|
||
# 1. 从内容库搜索
|
||
articles = self.search_articles(keyword, category)
|
||
results['articles'] = articles
|
||
|
||
# 2. 从互联网搜索(如果启用)
|
||
if include_internet:
|
||
internet_results = self.search_internet(keyword)
|
||
results['internet'] = internet_results
|
||
|
||
results['total'] = len(articles) + len(results['internet'])
|
||
|
||
return results
|
||
|
||
def save_to_articles(self, product_names, category, keywords, summary, content, source, url=None):
|
||
"""保存搜索结果到内容库"""
|
||
return db.add_article(product_names, category, keywords, summary, content, source, url)
|
||
|
||
# 全局搜索服务实例
|
||
search_service = SearchService() |