修改默认端口为16026
This commit is contained in:
+3
-3
@@ -49,10 +49,10 @@ COPY . .
|
|||||||
# 创建临时目录
|
# 创建临时目录
|
||||||
RUN mkdir -p /tmp/web_captures
|
RUN mkdir -p /tmp/web_captures
|
||||||
|
|
||||||
EXPOSE 16025
|
EXPOSE 16026
|
||||||
|
|
||||||
# 健康检查
|
# 健康检查
|
||||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=5s --retries=3 \
|
HEALTHCHECK --interval=30s --timeout=10s --start-period=5s --retries=3 \
|
||||||
CMD curl -f http://localhost:16025/health || exit 1
|
CMD curl -f http://localhost:16026/health || exit 1
|
||||||
|
|
||||||
CMD ["gunicorn", "-w", "4", "-b", "0.0.0.0:16025", "--timeout", "120", "app:app"]
|
CMD ["gunicorn", "-w", "4", "-b", "0.0.0.0:16026", "--timeout", "120", "app:app"]
|
||||||
+3
-3
@@ -6,7 +6,7 @@
|
|||||||
- **当前版本**: v1.0.2
|
- **当前版本**: v1.0.2
|
||||||
- **提交历史**:
|
- **提交历史**:
|
||||||
- v1.0.0: 初始化项目
|
- v1.0.0: 初始化项目
|
||||||
- v1.0.1: 修改端口为 16025
|
- v1.0.1: 修改端口为 16026
|
||||||
- v1.0.2: 修复 agent-browser socket 权限问题
|
- v1.0.2: 修复 agent-browser socket 权限问题
|
||||||
|
|
||||||
## 操作步骤
|
## 操作步骤
|
||||||
@@ -39,5 +39,5 @@ git push -u origin master --tags
|
|||||||
|
|
||||||
## 服务地址
|
## 服务地址
|
||||||
|
|
||||||
- Web 界面: http://192.168.0.101:16025
|
- Web 界面: http://192.168.0.101:16026
|
||||||
- API 接口: http://192.168.0.101:16025/api/capture
|
- API 接口: http://192.168.0.101:16026/api/capture
|
||||||
@@ -104,18 +104,18 @@ API 信息接口
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
# 基础截图
|
# 基础截图
|
||||||
curl -X POST http://localhost:16025/api/capture \
|
curl -X POST http://localhost:16026/api/capture \
|
||||||
-H "Content-Type: application/json" \
|
-H "Content-Type: application/json" \
|
||||||
-d '{"url": "https://example.com", "action": "screenshot"}' \
|
-d '{"url": "https://example.com", "action": "screenshot"}' \
|
||||||
--output screenshot.png
|
--output screenshot.png
|
||||||
|
|
||||||
# 提取HTML
|
# 提取HTML
|
||||||
curl -X POST http://localhost:16025/api/capture \
|
curl -X POST http://localhost:16026/api/capture \
|
||||||
-H "Content-Type: application/json" \
|
-H "Content-Type: application/json" \
|
||||||
-d '{"url": "https://example.com", "action": "html"}'
|
-d '{"url": "https://example.com", "action": "html"}'
|
||||||
|
|
||||||
# 滚动加载后全页截图
|
# 滚动加载后全页截图
|
||||||
curl -X POST http://localhost:16025/api/capture \
|
curl -X POST http://localhost:16026/api/capture \
|
||||||
-H "Content-Type: application/json" \
|
-H "Content-Type: application/json" \
|
||||||
-d '{
|
-d '{
|
||||||
"url": "https://news.ycombinator.com",
|
"url": "https://news.ycombinator.com",
|
||||||
@@ -127,7 +127,7 @@ curl -X POST http://localhost:16025/api/capture \
|
|||||||
--output full_page.png
|
--output full_page.png
|
||||||
|
|
||||||
# 自定义视口
|
# 自定义视口
|
||||||
curl -X POST http://localhost:16025/api/capture \
|
curl -X POST http://localhost:16026/api/capture \
|
||||||
-H "Content-Type: application/json" \
|
-H "Content-Type: application/json" \
|
||||||
-d '{
|
-d '{
|
||||||
"url": "https://example.com",
|
"url": "https://example.com",
|
||||||
@@ -144,7 +144,7 @@ import requests
|
|||||||
|
|
||||||
# 截图
|
# 截图
|
||||||
response = requests.post(
|
response = requests.post(
|
||||||
'http://localhost:16025/api/capture',
|
'http://localhost:16026/api/capture',
|
||||||
json={
|
json={
|
||||||
'url': 'https://example.com',
|
'url': 'https://example.com',
|
||||||
'action': 'screenshot'
|
'action': 'screenshot'
|
||||||
@@ -155,7 +155,7 @@ with open('screenshot.png', 'wb') as f:
|
|||||||
|
|
||||||
# 提取HTML
|
# 提取HTML
|
||||||
response = requests.post(
|
response = requests.post(
|
||||||
'http://localhost:16025/api/capture',
|
'http://localhost:16026/api/capture',
|
||||||
json={
|
json={
|
||||||
'url': 'https://example.com',
|
'url': 'https://example.com',
|
||||||
'action': 'html'
|
'action': 'html'
|
||||||
@@ -169,7 +169,7 @@ print(html)
|
|||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// 截图
|
// 截图
|
||||||
const response = await fetch('http://localhost:16025/api/capture', {
|
const response = await fetch('http://localhost:16026/api/capture', {
|
||||||
method: 'POST',
|
method: 'POST',
|
||||||
headers: {'Content-Type': 'application/json'},
|
headers: {'Content-Type': 'application/json'},
|
||||||
body: JSON.stringify({
|
body: JSON.stringify({
|
||||||
@@ -182,7 +182,7 @@ const response = await fetch('http://localhost:16025/api/capture', {
|
|||||||
const blob = await response.blob();
|
const blob = await response.blob();
|
||||||
|
|
||||||
// 提取HTML
|
// 提取HTML
|
||||||
const response = await fetch('http://localhost:16025/api/capture', {
|
const response = await fetch('http://localhost:16026/api/capture', {
|
||||||
method: 'POST',
|
method: 'POST',
|
||||||
headers: {'Content-Type': 'application/json'},
|
headers: {'Content-Type': 'application/json'},
|
||||||
body: JSON.stringify({
|
body: JSON.stringify({
|
||||||
@@ -199,7 +199,7 @@ console.log(data.html);
|
|||||||
### 使用 Gunicorn
|
### 使用 Gunicorn
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
gunicorn -w 4 -b 0.0.0.0:16025 app:app
|
gunicorn -w 4 -b 0.0.0.0:16026 app:app
|
||||||
```
|
```
|
||||||
|
|
||||||
### 使用 Systemd 服务
|
### 使用 Systemd 服务
|
||||||
@@ -230,12 +230,12 @@ RUN playwright install chromium --with-deps
|
|||||||
|
|
||||||
COPY . .
|
COPY . .
|
||||||
EXPOSE 5000
|
EXPOSE 5000
|
||||||
CMD ["gunicorn", "-w", "4", "-b", "0.0.0.0:16025", "app:app"]
|
CMD ["gunicorn", "-w", "4", "-b", "0.0.0.0:16026", "app:app"]
|
||||||
```
|
```
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
docker build -t web-capture-api .
|
docker build -t web-capture-api .
|
||||||
docker run -p 16025:16025 web-capture-api
|
docker run -p 16026:16026 web-capture-api
|
||||||
```
|
```
|
||||||
|
|
||||||
## 性能优化
|
## 性能优化
|
||||||
|
|||||||
@@ -364,6 +364,59 @@ async def capture_with_playwright(
|
|||||||
return {"success": False, "error": str(e)}
|
return {"success": False, "error": str(e)}
|
||||||
|
|
||||||
|
|
||||||
|
async def capture_with_cdp(
|
||||||
|
url_hint: str = "",
|
||||||
|
action: str = "screenshot",
|
||||||
|
cdp_port: int = 9222
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
使用 CDP 连接到已打开的 Chrome 浏览器
|
||||||
|
用于方案三:手动验证后自动截图
|
||||||
|
"""
|
||||||
|
if not PLAYWRIGHT_AVAILABLE:
|
||||||
|
return {"success": False, "error": "Playwright not installed"}
|
||||||
|
|
||||||
|
try:
|
||||||
|
async with async_playwright() as p:
|
||||||
|
# 连接到已运行的 Chrome
|
||||||
|
browser = await p.chromium.connect_over_cdp(f"http://localhost:{cdp_port}")
|
||||||
|
|
||||||
|
# 获取所有页面
|
||||||
|
contexts = browser.contexts
|
||||||
|
result_data = []
|
||||||
|
|
||||||
|
for context in contexts:
|
||||||
|
for page in context.pages:
|
||||||
|
current_url = page.url
|
||||||
|
title = await page.title()
|
||||||
|
|
||||||
|
# 如果 URL 包含提示词或者是唯一页面
|
||||||
|
if url_hint in current_url or not url_hint:
|
||||||
|
if action == "screenshot":
|
||||||
|
temp_file = CAPTURE_DIR / f"cdp_capture_{uuid.uuid4()[:8]}.png"
|
||||||
|
await page.screenshot(path=str(temp_file), full_page=True)
|
||||||
|
result_data.append({
|
||||||
|
"url": current_url,
|
||||||
|
"title": title,
|
||||||
|
"file_path": str(temp_file)
|
||||||
|
})
|
||||||
|
elif action == "html":
|
||||||
|
html = await page.content()
|
||||||
|
result_data.append({
|
||||||
|
"url": current_url,
|
||||||
|
"title": title,
|
||||||
|
"html": html
|
||||||
|
})
|
||||||
|
|
||||||
|
if result_data:
|
||||||
|
return {"success": True, "pages": result_data}
|
||||||
|
else:
|
||||||
|
return {"success": False, "error": "No matching page found"}
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
return {"success": False, "error": f"CDP connection failed: {str(e)}"}
|
||||||
|
|
||||||
|
|
||||||
def capture_webpage(
|
def capture_webpage(
|
||||||
url: str,
|
url: str,
|
||||||
action: str = "screenshot",
|
action: str = "screenshot",
|
||||||
@@ -467,7 +520,8 @@ def capture():
|
|||||||
"full_page": false,
|
"full_page": false,
|
||||||
"viewport": {"width": 1920, "height": 1080},
|
"viewport": {"width": 1920, "height": 1080},
|
||||||
"wait_time": 2000,
|
"wait_time": 2000,
|
||||||
"backend": "auto" | "agent-browser" | "playwright"
|
"backend": "auto" | "agent-browser" | "playwright" | "chrome-cdp",
|
||||||
|
"cdp_port": 9222 // 用于 chrome-cdp 后端,连接到已打开的 Chrome
|
||||||
}
|
}
|
||||||
"""
|
"""
|
||||||
data = request.get_json()
|
data = request.get_json()
|
||||||
@@ -490,6 +544,8 @@ def capture():
|
|||||||
viewport = data.get("viewport")
|
viewport = data.get("viewport")
|
||||||
wait_time = int(data.get("wait_time", 2000))
|
wait_time = int(data.get("wait_time", 2000))
|
||||||
backend = data.get("backend", "auto")
|
backend = data.get("backend", "auto")
|
||||||
|
cdp_port = int(data.get("cdp_port", 9222))
|
||||||
|
url_hint = data.get("url_hint", "")
|
||||||
|
|
||||||
result = capture_webpage(
|
result = capture_webpage(
|
||||||
url=url,
|
url=url,
|
||||||
@@ -499,7 +555,9 @@ def capture():
|
|||||||
full_page=full_page,
|
full_page=full_page,
|
||||||
viewport=viewport,
|
viewport=viewport,
|
||||||
wait_time=wait_time,
|
wait_time=wait_time,
|
||||||
backend=backend
|
backend=backend,
|
||||||
|
|
||||||
|
|
||||||
)
|
)
|
||||||
|
|
||||||
if not result["success"]:
|
if not result["success"]:
|
||||||
@@ -525,5 +583,5 @@ if __name__ == '__main__':
|
|||||||
print(" agent-browser: npm install -g agent-browser && agent-browser install")
|
print(" agent-browser: npm install -g agent-browser && agent-browser install")
|
||||||
print(" playwright: pip install playwright && playwright install chromium")
|
print(" playwright: pip install playwright && playwright install chromium")
|
||||||
|
|
||||||
print("\n🚀 Server running on http://0.0.0.0:16025")
|
print("\n🚀 Server running on http://0.0.0.0:16026")
|
||||||
app.run(host='0.0.0.0', port=16025, debug=True)
|
app.run(host='0.0.0.0', port=16026, debug=True)
|
||||||
@@ -23,4 +23,4 @@ echo "运行方式:"
|
|||||||
echo " 开发模式: python3.12 app.py"
|
echo " 开发模式: python3.12 app.py"
|
||||||
echo " 生产模式: gunicorn -w 4 -b 0.0.0.0:5000 app:app"
|
echo " 生产模式: gunicorn -w 4 -b 0.0.0.0:5000 app:app"
|
||||||
echo ""
|
echo ""
|
||||||
echo "访问地址: http://localhost:16025"
|
echo "访问地址: http://localhost:16026"
|
||||||
+2
-2
@@ -6,14 +6,14 @@ services:
|
|||||||
image: web-capture-api:latest
|
image: web-capture-api:latest
|
||||||
container_name: web-capture-api
|
container_name: web-capture-api
|
||||||
ports:
|
ports:
|
||||||
- "16025:16025"
|
- "16026:16026"
|
||||||
environment:
|
environment:
|
||||||
- TZ=Asia/Shanghai
|
- TZ=Asia/Shanghai
|
||||||
volumes:
|
volumes:
|
||||||
- ./captures:/tmp/web_captures
|
- ./captures:/tmp/web_captures
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
healthcheck:
|
healthcheck:
|
||||||
test: ["CMD", "curl", "-f", "http://localhost:16025/health"]
|
test: ["CMD", "curl", "-f", "http://localhost:16026/health"]
|
||||||
interval: 30s
|
interval: 30s
|
||||||
timeout: 10s
|
timeout: 10s
|
||||||
retries: 3
|
retries: 3
|
||||||
|
|||||||
+2
-2
@@ -110,7 +110,7 @@ echo " cd $(pwd)"
|
|||||||
echo " $PYTHON app.py"
|
echo " $PYTHON app.py"
|
||||||
echo ""
|
echo ""
|
||||||
echo "📖 访问地址:"
|
echo "📖 访问地址:"
|
||||||
echo " http://localhost:16025"
|
echo " http://localhost:16026"
|
||||||
echo ""
|
echo ""
|
||||||
echo "📚 API 文档:"
|
echo "📚 API 文档:"
|
||||||
echo " http://localhost:16025/api"
|
echo " http://localhost:16026/api"
|
||||||
@@ -0,0 +1,69 @@
|
|||||||
|
# 手动 Chrome 捕获指南
|
||||||
|
|
||||||
|
## 方案三完整流程
|
||||||
|
|
||||||
|
### 1️⃣ 启动真实 Chrome(带调试端口)
|
||||||
|
|
||||||
|
在您的本地机器或服务器上运行:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Linux
|
||||||
|
google-chrome --remote-debugging-port=9222 https://www.techpowerup.com/gpu-specs/rtx-pro-6000-blackwell.c4272
|
||||||
|
|
||||||
|
# 或者使用已有的 Chrome 进程(需要先关闭所有 Chrome)
|
||||||
|
google-chrome-stable --remote-debugging-port=9222 --user-data-dir=/tmp/chrome-debug
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2️⃣ 手动验证(如 Cloudflare)
|
||||||
|
|
||||||
|
- 在浏览器中手动完成验证过程
|
||||||
|
- 等待页面完全加载
|
||||||
|
- 确认可以看到正常内容
|
||||||
|
|
||||||
|
### 3️⃣ 连接到 Chrome 并截图
|
||||||
|
|
||||||
|
验证完成后,运行脚本:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 连接到 Chrome 并截图
|
||||||
|
python3.12 /tmp/connect_chrome.py techpowerup
|
||||||
|
|
||||||
|
# 或者捕获当前所有页面
|
||||||
|
python3.12 /tmp/connect_chrome.py
|
||||||
|
```
|
||||||
|
|
||||||
|
### 4️⃣ 查看结果
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 查看截图
|
||||||
|
ls -lh /tmp/chrome_capture_*.png
|
||||||
|
|
||||||
|
# 查看HTML
|
||||||
|
head -100 /tmp/chrome_html_0.html
|
||||||
|
```
|
||||||
|
|
||||||
|
## API 集成方案
|
||||||
|
|
||||||
|
如果需要通过 API 获取内容,可以:
|
||||||
|
|
||||||
|
1. 在服务代码中添加 CDP 连接选项
|
||||||
|
2. 用户先手动验证
|
||||||
|
3. API 连接到已验证的浏览器实例
|
||||||
|
|
||||||
|
## 注意事项
|
||||||
|
|
||||||
|
⚠️ **重要**:
|
||||||
|
- Chrome 必须带 `--remote-debugging-port=9222` 启动
|
||||||
|
- 所有 Chrome 进程必须先关闭(或使用新的 user-data-dir)
|
||||||
|
- 验证完成后才能运行脚本
|
||||||
|
- 截图后浏览器保持打开,可以继续使用
|
||||||
|
|
||||||
|
## 一键脚本
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 启动 Chrome 并等待用户验证
|
||||||
|
google-chrome --remote-debugging-port=9222 "https://www.techpowerup.com/gpu-specs/rtx-pro-6000-blackwell.c4272" &
|
||||||
|
echo "请在浏览器中完成验证,然后按 Enter 继续截图..."
|
||||||
|
read
|
||||||
|
python3.12 /tmp/connect_chrome.py techpowerup
|
||||||
|
```
|
||||||
+1
-1
@@ -7,7 +7,7 @@ import requests
|
|||||||
import json
|
import json
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
API_URL = "http://localhost:16025"
|
API_URL = "http://localhost:16026"
|
||||||
|
|
||||||
|
|
||||||
def test_health():
|
def test_health():
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ After=network.target
|
|||||||
Type=simple
|
Type=simple
|
||||||
User=openclaw
|
User=openclaw
|
||||||
WorkingDirectory=/home/openclaw/.openclaw/workspace-hz4th_coder/works/web-capture-api
|
WorkingDirectory=/home/openclaw/.openclaw/workspace-hz4th_coder/works/web-capture-api
|
||||||
ExecStart=/home/openclaw/.openclaw/workspace-hz4th_coder/works/web-capture-api/venv/bin/gunicorn -w 4 -b 0.0.0.0:16025 app:app
|
ExecStart=/home/openclaw/.openclaw/workspace-hz4th_coder/works/web-capture-api/venv/bin/gunicorn -w 4 -b 0.0.0.0:16026 app:app
|
||||||
Restart=always
|
Restart=always
|
||||||
RestartSec=10
|
RestartSec=10
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user