diff --git a/db.py b/db.py index 39a7721..8afe4d1 100644 --- a/db.py +++ b/db.py @@ -67,9 +67,7 @@ DDL = [ source_url VARCHAR(2000), depth INT, attempts INT DEFAULT 1, - html_file VARCHAR(500), - txt_file VARCHAR(500), - meta_file VARCHAR(500), + base_file VARCHAR(500), image_count INT DEFAULT 0, image_files TEXT, UNIQUE KEY uk_run_url (run_id, url(500)) @@ -78,6 +76,17 @@ DDL = [ ] +def _migrate(conn): + """存量表结构迁移: 合并 html_file/txt_file/meta_file 为 base_file""" + with conn.cursor() as cur: + cur.execute("SHOW COLUMNS FROM crawl_results LIKE 'html_file'") + if cur.fetchone(): + cur.execute("ALTER TABLE crawl_results ADD COLUMN base_file VARCHAR(500) NULL AFTER meta_file") + cur.execute("UPDATE crawl_results SET base_file = REPLACE(html_file, '.html', '') WHERE base_file IS NULL") + cur.execute("ALTER TABLE crawl_results DROP COLUMN html_file, DROP COLUMN txt_file, DROP COLUMN meta_file") + print("[db] 表结构迁移完成: html_file/txt_file/meta_file -> base_file") + + def _conn(): cfg = dict(DB_CONFIG) cfg["database"] = DB_NAME @@ -100,6 +109,7 @@ def init_db(): cur.execute(f"CREATE DATABASE IF NOT EXISTS `{DB_NAME}` DEFAULT CHARACTER SET utf8mb4") conn.close() conn = _conn() + _migrate(conn) with conn.cursor() as cur: for ddl in DDL: cur.execute(ddl) @@ -219,8 +229,21 @@ def upsert_run(run): # ---------------- 爬取结果 (每页一条, 成功失败均记录) ---------------- +def _base_of(entry): + """从结果条目提取基础文件名 (html/txt/meta 三个后缀共用同一前缀)""" + for f in (entry.get("meta_file"), entry.get("html_file"), entry.get("txt_file")): + f = f or "" + if f.endswith(".meta.json"): + return f[:-10] + if f.endswith(".html"): + return f[:-5] + if f.endswith(".txt"): + return f[:-4] + return "" + + def insert_results(run, results): - """批量插入爬取结果 (增量)""" + """批量插入爬取结果 (增量); 文件只记基础名 base_file""" if not results: return @@ -234,7 +257,7 @@ def insert_results(run, results): r.get("status", "FAIL"), r.get("error"), _dt(r.get("crawl_time")), (r.get("source_url") or "")[:2000], r.get("depth"), r.get("attempts", 1), - r.get("html_file", ""), r.get("txt_file", ""), r.get("meta_file", ""), + _base_of(r), len(r.get("images", []) or []), _j([im.get("file") for im in (r.get("images") or [])]), )) @@ -243,8 +266,8 @@ def insert_results(run, results): """INSERT IGNORE INTO crawl_results (run_id, task_id, mode, url, title, status, error, crawl_time, source_url, depth, attempts, - html_file, txt_file, meta_file, image_count, image_files) - VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)""", + base_file, image_count, image_files) + VALUES (%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s,%s)""", rows, ) conn.close()