# -*- coding: utf-8 -*- """ 批量网页爬取器 (Playwright + stealth + 系统Chrome) 用法: python crawl.py urls.txt [--out 输出目录] [--delay 2,5] [--timeout 60] urls.txt: 每行一个网址, # 开头为注释, 空行忽略 输出: out/0001_<域名>_<时间戳>.html 完整网页HTML out/0001_<域名>_<时间戳>.txt 提取的纯文本正文 out/results.csv 汇总表(序号/网址/标题/状态/文件) out/cookies.json 会话cookie(自动复用, 减少重复验证) """ import argparse, csv, json, os, random, re, sys, time, urllib.parse from datetime import datetime from playwright.sync_api import sync_playwright from playwright_stealth import Stealth HERE = os.path.dirname(os.path.abspath(__file__)) COOKIE_FILE = os.path.join(HERE, "cookies.json") def load_urls(path): urls = [] with open(path, encoding="utf-8") as f: for line in f: line = line.strip() if not line or line.startswith("#"): continue if not line.startswith("http"): line = "https://" + line urls.append(line) return urls def safe_name(url, idx): host = urllib.parse.urlparse(url).netloc.replace("www.", "").replace(".", "_") ts = datetime.now().strftime("%Y%m%d_%H%M%S") return f"{idx:04d}_{host}_{ts}" def is_challenge_page(title, html): """判断是否仍在反爬验证页 (通用启发式, 可按需调整)""" low = html.lower() if "access denied" in title.lower() or "403" in title: return True if "bot check" in low or "captcha" in low or "cf-challenge" in low: return True # 页面过小且标题是站名本身 -> 多半是验证壳 if len(html) < 8000 and "techpowerup" in low: return True return False def wait_page_settle(page, timeout_s=60): """等页面稳定: 挑战自动跳转结束 + 内容可读""" last_title, stable = "", 0 start = time.time() while time.time() - start < timeout_s: time.sleep(1) try: title = page.title() html = page.content() except Exception: continue # 正在跳转 if is_challenge_page(title, html): stable = 0 continue if title == last_title: stable += 1 if stable >= 2 and len(html) > 1000: return True, title, html else: stable = 0 last_title = title try: return True, page.title(), page.content() except Exception: return False, "", "" def crawl_one(page, url, timeout_s): page.goto(url, wait_until="domcontentloaded", timeout=timeout_s * 1000) ok, title, html = wait_page_settle(page, timeout_s) if not ok: raise RuntimeError(f"页面加载超时({timeout_s}s)") if is_challenge_page(title, html): raise RuntimeError(f"仍被反爬拦截: title={title!r} size={len(html)}") text = page.inner_text("body") return title, html, text def main(): ap = argparse.ArgumentParser() ap.add_argument("urls_file") ap.add_argument("--out", default=os.path.join(HERE, "out")) ap.add_argument("--delay", default="2,5", help="每次请求间随机延迟秒数, 如 2,5") ap.add_argument("--timeout", type=int, default=60, help="单页最长等待秒数") ap.add_argument("--retry", type=int, default=2, help="失败重试次数") args = ap.parse_args() urls = load_urls(args.urls_file) if not urls: print("urls.txt 里没有有效网址"); return os.makedirs(args.out, exist_ok=True) dmin, dmax = map(float, args.delay.split(",")) results = [] with sync_playwright() as p: browser = p.chromium.launch( headless=True, executable_path="/usr/bin/google-chrome", args=["--disable-blink-features=AutomationControlled", "--no-sandbox", "--disable-gpu"], ) ctx = browser.new_context( user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36", viewport={"width": 1920, "height": 1080}, locale="en-US", ) if os.path.exists(COOKIE_FILE): try: ctx.add_cookies(json.load(open(COOKIE_FILE))) print("[i] 已复用上次会话 cookie") except Exception as e: print(f"[i] cookie 复用失败(忽略): {e}") Stealth().apply_stealth_sync(ctx) page = ctx.new_page() for i, url in enumerate(urls, 1): print(f"[{i}/{len(urls)}] {url}", flush=True) base = safe_name(url, i) title, status = "", "" for attempt in range(args.retry + 1): try: title, html, text = crawl_one(page, url, args.timeout) status = "OK" html_path = os.path.join(args.out, base + ".html") txt_path = os.path.join(args.out, base + ".txt") with open(html_path, "w", encoding="utf-8") as f: f.write(html) with open(txt_path, "w", encoding="utf-8") as f: f.write(text) print(f" -> OK {title[:60]!r} html={len(html)//1024}KB txt={len(text)//1024}KB") break except Exception as e: status = f"FAIL: {e}" print(f" -> 第{attempt+1}次失败: {e}") time.sleep(3) results.append({"no": i, "url": url, "title": title, "status": status, "file": base + ".html"}) # 保存 cookie (验证通过后), 供下次复用 try: json.dump(ctx.cookies(), open(COOKIE_FILE, "w")) except Exception: pass time.sleep(random.uniform(dmin, dmax)) # 礼貌延迟 browser.close() # 汇总 csv_path = os.path.join(args.out, "results.csv") with open(csv_path, "w", newline="", encoding="utf-8-sig") as f: w = csv.writer(f) w.writerow(["序号", "网址", "标题", "状态", "文件"]) for r in results: w.writerow([r["no"], r["url"], r["title"], r["status"], r["file"]]) ok = sum(1 for r in results if r["status"] == "OK") print(f"\n完成: {ok}/{len(results)} 成功") print(f"HTML/文本: {args.out}/") print(f"汇总表: {csv_path}") if __name__ == "__main__": main()