Files
param-auto-manager/services/search_service.py
T
hz4th_coder c960d959ad 新增搜索结果缓存功能
- 新增 search_cache 数据库表存储搜索缓存
- 搜索前优先查询缓存,命中则直接返回
- 缓存默认有效期7天,可配置
- 前端支持'使用缓存'开关选项
- 显示缓存状态提示
- 支持自动清理过期缓存
2026-07-13 22:55:36 +08:00

344 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
搜索服务 - 从内容库和互联网搜索数据
"""
import requests
from bs4 import BeautifulSoup
import json
import subprocess
import os
import re
import urllib.parse
from datetime import datetime
from config import Config
from models.database import db
class SearchService:
def __init__(self):
self.timeout = Config.SEARCH_TIMEOUT
self.max_results = Config.SEARCH_MAX_RESULTS
def _run_browser(self, *args, timeout=30000):
"""运行 agent-browser 命令"""
env = os.environ.copy()
env['XDG_RUNTIME_DIR'] = '/tmp/agent-browser-runtime'
os.makedirs(env['XDG_RUNTIME_DIR'], exist_ok=True)
cmd = ['agent-browser'] + list(args)
result = subprocess.run(
cmd,
capture_output=True,
text=True,
env=env,
timeout=timeout // 1000 + 5
)
return result.stdout, result.stderr, result.returncode
def search_internet(self, keyword, max_results=None, engine='bing_cn', use_cache=True, cache_days=7):
"""
从互联网搜索(使用 agent-browser 浏览器自动化)
支持的搜索引擎:
- bing_cn: Bing 中国(默认)
- bing_global: Bing 国际版
- google: Google
- baidu: 百度
参数:
- use_cache: 是否使用缓存(默认True
- cache_days: 缓存有效天数(默认7天)
"""
max_results = max_results or self.max_results
results = []
# 优先查询缓存
if use_cache:
cached = db.get_search_cache(keyword, engine)
if cached:
print(f"使用缓存结果: {keyword} ({engine})")
return cached['results']
# 根据搜索引擎选择 URL
encoded_keyword = urllib.parse.quote(keyword)
search_urls = {
'bing_cn': f"https://cn.bing.com/search?q={encoded_keyword}",
'bing_global': f"https://www.bing.com/search?q={encoded_keyword}",
'google': f"https://www.google.com/search?q={encoded_keyword}",
'baidu': f"https://www.baidu.com/s?wd={encoded_keyword}"
}
search_url = search_urls.get(engine, search_urls['bing_cn'])
try:
# 1. 打开搜索引擎
stdout, stderr, code = self._run_browser('open', search_url, '--timeout', '20000')
if code != 0:
print(f"打开搜索页面失败: {stderr}")
return results
# 等待页面加载
stdout, stderr, code = self._run_browser('wait', '5000')
# 2. 获取搜索结果页面结构 (JSON 格式)
stdout, stderr, code = self._run_browser('snapshot', '--json', '--timeout', '30000')
if code != 0:
print(f"获取页面结构失败: {stderr}")
return results
# 3. 解析 JSON 提取搜索结果
try {
data = json.loads(stdout)
except json.JSONDecodeError:
print(f"解析 JSON 失败: {stdout[:500]}")
return results
# 4. 根据搜索引擎选择解析方法
if engine == 'baidu':
results = self._parse_baidu_results(data, max_results)
else:
results = self._parse_bing_results(data, max_results)
# 5. 关闭浏览器
self._run_browser('close')
# 6. 保存到缓存
if results and use_cache:
db.save_search_cache(keyword, engine, results, cache_days)
except subprocess.TimeoutExpired:
print(f"搜索超时: {keyword}")
except Exception as e:
print(f"搜索出错: {str(e)}")
# 尝试关闭浏览器
try:
self._run_browser('close')
except:
pass
return results
def _parse_bing_results(self, snapshot_data, max_results=10):
"""
从 Bing 搜索结果的 snapshot 中解析出标题和链接
snapshot_data 是 agent-browser snapshot --json 的输出
结构: {success, data: {snapshot: "文本格式的 accessibility tree"}, error}
"""
results = []
# 获取 snapshot 文本
snapshot = snapshot_data.get('data', {}).get('snapshot', '')
if not snapshot:
return results
# 解析 accessibility tree 文本
in_results = False
refs = [] # 存储 (title, ref) 元组
lines = snapshot.split('\n')
for i, line in enumerate(lines):
line = line.strip()
# 进入搜索结果区域
if 'main "搜索结果"' in line:
in_results = True
continue
# 离开搜索结果区域
if in_results and line.startswith('- ') and 'main' in line and '搜索结果' not in line:
break
if not in_results:
continue
# 匹配标题链接:link "标题文字" [ref=eXX]
# 需要过滤域名链接(如 "zhihu.com")和短链接
if 'link "' in line and '[ref=' in line:
match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line)
if match:
title = match.group(1)
ref = match.group(2)
# 过滤短标题(域名链接如 "zhihu.com"
if len(title) > 20 and '.' not in title[:10]: # 不是域名格式
refs.append((title, ref))
# 获取每个结果的 URL
for title, ref in refs[:max_results]:
url = self._get_link_url(ref)
if url and 'bing.com/search' not in url: # 过滤搜索结果页本身的链接
results.append({
'title': title,
'url': url,
'snippet': '',
'source': 'bing'
})
return results
def _parse_baidu_results(self, snapshot_data, max_results=10):
"""从百度搜索结果中解析标题和链接"""
results = []
snapshot = snapshot_data.get('data', {}).get('snapshot', '')
if not snapshot:
return results
# 百度搜索结果解析
refs = []
lines = snapshot.split('\n')
for line in lines:
line = line.strip()
# 百度结果通常在 link 标签中
if 'link "' in line and '[ref=' in line:
match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line)
if match:
title = match.group(1)
ref = match.group(2)
# 过滤百度内部链接和广告
if len(title) > 10 and '百度' not in title[:6]:
refs.append((title, ref))
# 获取每个结果的 URL
for title, ref in refs[:max_results]:
url = self._get_link_url(ref)
if url and 'baidu.com' not in url:
results.append({
'title': title,
'url': url,
'snippet': '',
'source': 'baidu'
})
return results
def _get_link_url(self, ref):
"""通过 agent-browser 获取链接的 URL"""
try:
stdout, stderr, code = self._run_browser('get', 'attr', f'@{ref}', 'href', '--json', '--timeout', '5000')
if code == 0 and stdout:
data = json.loads(stdout)
return data.get('data', {}).get('value', '')
except Exception as e:
print(f"获取 URL 失败 (ref={ref}): {e}")
return None
def fetch_url_content(self, url):
"""抓取网页内容(使用 agent-browser 浏览器方式,绑过反爬虫)"""
try:
# 使用浏览器方式抓取
stdout, stderr, code = self._run_browser('open', url, '--timeout', '20000')
if code != 0:
print(f"打开页面失败: {stderr}")
return None
# 等待页面加载
self._run_browser('wait', '5000')
# 获取页面标题
stdout, stderr, code = self._run_browser('get', 'title', '--timeout', '5000')
title = stdout.strip().replace('[agent-browser] ', '').strip() if code == 0 else ''
# 获取页面内容(通过 snapshot 获取 accessibility tree
stdout, stderr, code = self._run_browser('snapshot', '--json', '--timeout', '15000')
text = ''
if code == 0 and stdout:
try:
data = json.loads(stdout)
snapshot = data.get('data', {}).get('snapshot', '')
# 从 snapshot 中提取所有 StaticText
text = self._extract_text_from_snapshot(snapshot)
except:
pass
# 获取 URL(可能被重定向)
stdout, stderr, code = self._run_browser('get', 'url', '--timeout', '5000')
actual_url = stdout.strip() if code == 0 else url
# 关闭浏览器
self._run_browser('close')
# 提取描述(从页面内容的前200字符)
description = text[:200].strip() if text else ''
return {
'title': title,
'description': description,
'content': text,
'url': actual_url,
'fetch_date': datetime.now().isoformat()
}
except Exception as e:
print(f"抓取URL失败: {url}, 错误: {str(e)}")
# 尝试关闭浏览器
try:
self._run_browser('close')
except:
pass
return None
def _extract_text_from_snapshot(self, snapshot):
"""从 accessibility tree snapshot 中提取文本内容"""
texts = []
for line in snapshot.split('\n'):
line = line.strip()
if 'StaticText' in line and 'checkbox' not in line:
# 找到 StaticText 后的内容
idx = line.find('StaticText')
after = line[idx + 10:].strip() # 跳过 'StaticText'
# 去掉开头的引号
if after.startswith('"'):
after = after[1:]
# 如果以 JSON 开头(错误信息),跳过
if after.startswith('{'):
continue
# 提取文本内容
text = after.rstrip('"').strip()
if text and len(text) > 1:
texts.append(text)
result = '\n'.join(texts)
# 检测是否是反爬错误页面
if '请求存在异常' in result or '暂时限制本次访问' in result:
return '[该网站触发了反爬机制,无法抓取内容]'
return result
def search_articles(self, keyword, category=None):
"""从内容库搜索"""
return db.search_articles(keyword, category)
def search_all(self, keyword, category=None, include_internet=True):
"""
综合搜索:内容库 + 互联网
"""
results = {
'articles': [],
'internet': [],
'total': 0
}
# 1. 从内容库搜索
articles = self.search_articles(keyword, category)
results['articles'] = articles
# 2. 从互联网搜索(如果启用)
if include_internet:
internet_results = self.search_internet(keyword)
results['internet'] = internet_results
results['total'] = len(articles) + len(results['internet'])
return results
def save_to_articles(self, product_names, category, keywords, summary, content, source, url=None):
"""保存搜索结果到内容库"""
return db.add_article(product_names, category, keywords, summary, content, source, url)
# 全局搜索服务实例
search_service = SearchService()