v1.0.14: 自动爬取支持深度/页数无限制(默认0/1000), 待爬链接持久化缓存队列(停止后续爬), 起始网址不去重, 清空缓存API

This commit is contained in:
2026-08-11 21:07:31 +08:00
parent 3ad3d00ca8
commit 66ba1a7cfc
5 changed files with 91 additions and 25 deletions
+42 -18
View File
@@ -21,6 +21,7 @@ from playwright_stealth import Stealth
import notify
import store
import db
HERE = os.path.dirname(os.path.abspath(__file__))
DATA_DIR = os.path.join(HERE, "data")
@@ -567,34 +568,48 @@ class CrawlJob:
return included
def _crawl_auto(self):
"""自动爬取: 持续递归 爬取->发现链接->爬取... 直到无新链接可爬或达到安全上限"""
"""自动爬取: 持续递归 爬取->发现链接->爬取... 直到无新链接可爬或达到上限
- max_depth: 0=无限制, N=只爬 N 层
- max_pages: 0=无限制, N=安全上限
- 提取但未爬取的链接持久化到 auto.pending, 已爬集合持久化到 auto.visited
- 停止后再次运行从缓存队列继续爬; 起始网址每次运行都重新爬(不去重)
"""
run = self.run
auto = self.task.get("auto", {})
seed = auto.get("seed_url", "")
max_pages = int(auto.get("max_pages", 200) or 200) # 安全上限, 防失控
run["progress"]["total"] = max_pages
max_pages = int(auto.get("max_pages", 1000) or 0) # 0 = 无限制
max_depth = int(auto.get("max_depth", 0) or 0) # 0 = 无限制
out_dir = self._resolve_out_dir()
run["out_dir"] = out_dir
os.makedirs(out_dir, exist_ok=True)
# 恢复持久化状态: 已爬集合 + 上次未爬完的缓存队列
visited = set(auto.get("visited", []) or [])
pending = auto.get("pending", []) or []
# 起始网址每次运行都爬(不做去重), 缓存队列继续消费
queue = [(seed, 0, "")]
if pending:
queue.extend((p["url"], p.get("depth", 0), p.get("source", "")) for p in pending)
queued = set(visited)
for u, _d, _s in queue:
queued.add(normalize_url(u))
run["progress"]["total"] = len(queue)
self._persist()
p, browser, ctx, page, cookie_file = self._open_browser()
queue = [(seed, 0, "")] # (url, depth, 来源链接)
visited = set() # 规范化 URL 去重
queued = set([normalize_url(seed)])
idx = 0
try:
while queue and not self._stop.is_set():
self._wait_if_paused()
url, depth, src = queue.pop(0)
key = normalize_url(url)
if key in visited:
continue
if len(visited) >= max_pages:
if max_pages > 0 and len(visited) >= max_pages:
self._log("info", f"达到最大页数上限 {max_pages}, 停止")
break
url, depth, src = queue.pop(0)
key = normalize_url(url)
if key in visited and url != seed: # 起始网址不去重, 其余已爬跳过
continue
visited.add(key)
idx += 1
idx = len(visited)
run["progress"]["current_url"] = url
run["progress"]["done"] = len(visited)
self._persist()
@@ -603,8 +618,8 @@ class CrawlJob:
run["results"].append(entry)
self._bump_stats(entry)
self._persist()
# 无深度限制: 只要页面爬取成功就继续发现链接, 递归直到队列为空
if entry["status"] == "OK":
# 无深度限制或未达深度限制时持续发现链接
if entry["status"] == "OK" and (max_depth == 0 or depth < max_depth):
for link in self._discover_links(page):
lk = normalize_url(link)
if lk not in visited and lk not in queued:
@@ -612,9 +627,18 @@ class CrawlJob:
queue.append((link, depth + 1, url))
if entry["status"] == "OK":
self._delay()
run["progress"]["total"] = len(visited)
if not self._stop.is_set() and len(queue) == 0:
self._log("info", f"无新链接可爬, 任务结束 (共 {len(visited)} 页)")
finally:
self._close_browser(p, browser, ctx, cookie_file)
run["progress"]["total"] = len(visited)
if not self._stop.is_set() and len(visited) < max_pages:
self._log("info", f"无新链接可爬, 任务结束 (共 {len(visited)} 页)")
# 无论完成/停止/异常, 都保存缓存队列与已爬集合, 便于下次继续
try:
auto["pending"] = [
{"url": u, "depth": d, "source": s} for u, d, s in queue]
auto["visited"] = list(visited)
store.upsert_task(self.task)
db.upsert_task(self.task)
except Exception:
pass
self._persist()