Compare commits

..
2 Commits
Author SHA1 Message Date
hz4th_coder 4f02a6a0ea 实现浏览器方式互联网搜索功能
- 使用 agent-browser 浏览器自动化进行 Bing 搜索
- 解析 accessibility tree 提取搜索结果
- 自动获取每个结果的 URL
- 支持设置最大结果数量
2026-07-13 11:55:33 +08:00
hz4th_coder 187d5037b8 添加Git推送说明文档 2026-07-12 01:24:46 +08:00
2 changed files with 221 additions and 5 deletions
+88
View File
@@ -0,0 +1,88 @@
# Git推送说明
## 📌 当前状态
代码已准备推送,但Git服务器返回403错误:
```
remote: Push to create is not enabled for users.
```
这表示Git服务器不允许用户直接推送创建仓库。
---
## ✅ 解决方案
### 方案1: 管理员预先创建仓库
请在Git服务器上手动创建以下仓库:
**仓库信息:**
- **地址**: http://121.40.164.32:12007/hz4th_coder/param-auto-manager.git
- **账号**: hz4th_coder
- **组织**: hz4th_coder
**创建后推送代码:**
```bash
cd /home/openclaw/.openclaw/workspace-hz4th_coder/works/param-auto-manager
# 添加远程仓库(如果还没有)
git remote add origin http://hz4th_coder:262e7dbfce09c8cc21fbacff2b450cc3f1c3e265@121.40.164.32:12007/hz4th_coder/param-auto-manager.git
# 推送代码和标签
git push -u origin master
git push origin v1.0.0
git push origin v1.1.0
git push origin v1.2.0
```
---
### 方案2: 开通推送权限
联系Git服务器管理员,为 `hz4th_coder` 用户开通"Push to create"权限。
---
## 📊 当前Git状态
```bash
cd works/param-auto-manager
git log --oneline -5
```
输出:
```
47729a5 添加部署文档和说明
8ad0246 添加前端界面和操作页面
e3bc883 添加.gitignore文件,排除缓存和临时文件
b7b0925 初始化参数数据自动化管理系统
```
标签:
```
v1.0.0 - 初始化版本
v1.1.0 - 添加前端界面
v1.2.0 - 添加部署文档
```
---
## 🔐 Git认证信息
- **账号**: hz4th_coder
- **邮箱**: hz4th_coder@tphai.com
- **Token**: 262e7dbfce09c8cc21fbacff2b450cc3f1c3e265
---
## 📝 待推送文件统计
- 总文件数: 26个源文件
- 总代码行数: 约5000行
- 版本标签: 3个
---
**建议**: 请在Git服务器创建仓库后,执行推送命令即可完成部署。
+133 -5
View File
@@ -4,6 +4,10 @@
import requests import requests
from bs4 import BeautifulSoup from bs4 import BeautifulSoup
import json import json
import subprocess
import os
import re
import urllib.parse
from datetime import datetime from datetime import datetime
from config import Config from config import Config
from models.database import db from models.database import db
@@ -13,19 +17,143 @@ class SearchService:
self.timeout = Config.SEARCH_TIMEOUT self.timeout = Config.SEARCH_TIMEOUT
self.max_results = Config.SEARCH_MAX_RESULTS self.max_results = Config.SEARCH_MAX_RESULTS
def _run_browser(self, *args, timeout=30000):
"""运行 agent-browser 命令"""
env = os.environ.copy()
env['XDG_RUNTIME_DIR'] = '/tmp/agent-browser-runtime'
os.makedirs(env['XDG_RUNTIME_DIR'], exist_ok=True)
cmd = ['agent-browser'] + list(args)
result = subprocess.run(
cmd,
capture_output=True,
text=True,
env=env,
timeout=timeout // 1000 + 5
)
return result.stdout, result.stderr, result.returncode
def search_internet(self, keyword, max_results=None): def search_internet(self, keyword, max_results=None):
""" """
从互联网搜索(使用搜索API或爬虫 从互联网搜索(使用 agent-browser 浏览器自动化
这里暂时使用简单的搜索模拟
""" """
max_results = max_results or self.max_results max_results = max_results or self.max_results
# TODO: 接入真实的搜索API(如Google Custom Search、Bing等)
# 这里先返回空列表,等待后续接入真实API
results = [] results = []
try:
# 1. 打开 Bing 搜索
encoded_keyword = urllib.parse.quote(keyword)
search_url = f"https://www.bing.com/search?q={encoded_keyword}"
stdout, stderr, code = self._run_browser('open', search_url, '--timeout', '20000')
if code != 0:
print(f"打开搜索页面失败: {stderr}")
return results
# 等待页面加载
stdout, stderr, code = self._run_browser('wait', '5000')
# 2. 获取搜索结果页面结构 (JSON 格式)
stdout, stderr, code = self._run_browser('snapshot', '--json', '--timeout', '30000')
if code != 0:
print(f"获取页面结构失败: {stderr}")
return results
# 3. 解析 JSON 提取搜索结果
try:
data = json.loads(stdout)
except json.JSONDecodeError:
print(f"解析 JSON 失败: {stdout[:500]}")
return results
# 4. 从 accessibility tree 中提取搜索结果
# Bing 搜索结果在 main[aria-label="搜索结果"] 区域内
results = self._parse_bing_results(data, max_results)
# 5. 关闭浏览器
self._run_browser('close')
except subprocess.TimeoutExpired:
print(f"搜索超时: {keyword}")
except Exception as e:
print(f"搜索出错: {str(e)}")
# 尝试关闭浏览器
try:
self._run_browser('close')
except:
pass
return results return results
def _parse_bing_results(self, snapshot_data, max_results=10):
"""
从 Bing 搜索结果的 snapshot 中解析出标题和链接
snapshot_data 是 agent-browser snapshot --json 的输出
结构: {success, data: {snapshot: "文本格式的 accessibility tree"}, error}
"""
results = []
# 获取 snapshot 文本
snapshot = snapshot_data.get('data', {}).get('snapshot', '')
if not snapshot:
return results
# 解析 accessibility tree 文本
in_results = False
refs = [] # 存储 (title, ref) 元组
lines = snapshot.split('\n')
for i, line in enumerate(lines):
line = line.strip()
# 进入搜索结果区域
if 'main "搜索结果"' in line:
in_results = True
continue
# 离开搜索结果区域
if in_results and line.startswith('- ') and 'main' in line and '搜索结果' not in line:
break
if not in_results:
continue
# 匹配标题链接:link "标题文字" [ref=eXX]
# 需要过滤域名链接(如 "zhihu.com")和短链接
if 'link "' in line and '[ref=' in line:
match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line)
if match:
title = match.group(1)
ref = match.group(2)
# 过滤短标题(域名链接如 "zhihu.com"
if len(title) > 20 and '.' not in title[:10]: # 不是域名格式
refs.append((title, ref))
# 获取每个结果的 URL
for title, ref in refs[:max_results]:
url = self._get_link_url(ref)
if url and 'bing.com/search' not in url: # 过滤搜索结果页本身的链接
results.append({
'title': title,
'url': url,
'snippet': '',
'source': 'bing'
})
return results
def _get_link_url(self, ref):
"""通过 agent-browser 获取链接的 URL"""
try:
stdout, stderr, code = self._run_browser('get', 'attr', f'@{ref}', 'href', '--json', '--timeout', '5000')
if code == 0 and stdout:
data = json.loads(stdout)
return data.get('data', {}).get('value', '')
except Exception as e:
print(f"获取 URL 失败 (ref={ref}): {e}")
return None
def fetch_url_content(self, url): def fetch_url_content(self, url):
"""抓取网页内容""" """抓取网页内容"""
try: try: