Files
news-tracker/crawler.py
T

403 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""
新闻智能跟踪系统 - 真实网页采集层
职责:
1. 抓取数据源页面(真实 URL),用 readability 风格启发式清洗出「干净可读正文」
2. 从页面提取候选资讯链接(标题 + 绝对 URL),按文章相似度排序
3. 对前 N 条候选链接抓取全文,一并入库(存 articles.full_text 便于后期查看)
4. example.com 等模拟源 / 抓取失败源,回退到 simulate 模拟数据,保证全链路不空转
数据结构与 simulate 保持一致:fetch_source(source) -> list[dict]
title/url/content/summary/published_at/domain/entities/source_id/full_text
"""
import json
import re
import time
from datetime import datetime
from urllib.parse import urljoin, urlparse
import requests
from bs4 import BeautifulSoup
import config
import db
import simulate
# 直接丢弃的标签(导航/脚本/广告/评论等)
_BAD_TAGS = [
"script", "style", "noscript", "iframe", "form", "button", "select", "option",
"input", "svg", "canvas", "video", "audio", "nav", "footer", "header", "aside",
]
# 出现在 class/id 中即视为噪音块的片段
_BAD_HINTS = [
"comment", "share", "social", "advert", "ads", "menu", "sidebar", "footer",
"nav", "related", "recommend", "promo", "newsletter", "subscribe",
"cookie", "banner", "breadcrumb", "pagination", "toolbar", "login", "signup",
"tag-list", "author", "meta", "bottom", "widget", "popup", "overlay",
]
# 不是资讯链接的后缀/前缀
_BAD_HREFS = ("mailto:", "tel:", "javascript:", "#", "?login", "?signup")
_BAD_EXTS = (".css", ".js", ".ico", ".png", ".jpg", ".jpeg", ".gif", ".webp",
".pdf", ".zip", ".xml", ".rss", ".atom", ".json")
# 常见导航词(短标题/纯导航链接,不当作资讯)
_NAV_WORDS = {
"research", "business", "developers", "about", "careers", "blog", "contact",
"sign in", "sign up", "login", "register", "privacy", "terms", "legal", "trust",
"customer stories", "partners", "docs", "api log in", "skip to main content",
"home", "news", "newsroom", "press", "company", "safety", "product", "products",
"pricing", "solutions", "learn more", "read more", "view all", "see all",
"all news", "all stories", "open a new window", "instagram", "twitter", "facebook",
"linkedin", "youtube", "github", "login", "log in", "get started", "try now",
}
# 标题尾部常见的“分类 + 日期”后缀,清洗掉(如 "Company Aug 27, 2026"
_TRAIL_DATE_RE = re.compile(r"\s+\S+\s+[A-Z][a-z]{2}\s+\d{1,2},?\s+\d{4}$")
def fetch_page(url, timeout=None, retries=2):
"""抓取页面 HTML,带浏览器 UA 与语言头;瞬断自动重试"""
hdr = {
"User-Agent": config.CRAWL_DEFAULTS["user_agent"],
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}
last = None
for i in range(retries + 1):
try:
r = requests.get(url, headers=hdr,
timeout=timeout or config.CRAWL_DEFAULTS["crawl_timeout"])
r.raise_for_status()
return r.text
except Exception as e:
last = e
if i < retries:
time.sleep(1.5 * (i + 1))
raise last
# ---------------- 正文清洗(readability 风格) ----------------
def _is_noise(tag):
if tag is None or not hasattr(tag, "name") or tag.name is None:
return True
if tag.name in _BAD_TAGS:
return True
try:
tokens = []
for t in tag.get("class") or []:
# Tailwind 任意值类(如 [--header-h:...])只是 CSS 变量,语义噪音,跳过
if t.startswith("[") and t.endswith("]"):
continue
tokens.append(t.lower())
tokens.append((tag.get("id") or "").lower())
cls = " ".join(tokens)
except Exception:
cls = ""
return any(h in cls for h in _BAD_HINTS)
def _strip_noise(soup):
"""删除噪音节点,返回清洗后的 body。先收集再统一删除,避免迭代中改树。"""
noise = []
for tag in list(soup.find_all(True)):
try:
if tag is None:
continue
if _is_noise(tag):
noise.append(tag)
continue
style = (tag.get("style") or "").lower()
if "display:none" in style or "visibility:hidden" in style:
noise.append(tag)
except Exception:
continue
for tag in noise:
try:
tag.decompose()
except Exception:
pass
return soup.body or soup
def _text_density(tag):
"""正文块得分:有效文本长度 + 段落/标题数量"""
text = tag.get_text(" ", strip=True)
if not text:
return 0
paras = len(tag.find_all(["p", "h1", "h2", "h3", "h4", "li", "pre", "blockquote"]))
links = len(tag.find_all("a"))
# 链接占比例过高多半是导航/聚合页,降权
link_penalty = min(1.0, links / max(1, paras) * 0.5)
return len(text) * (1 - link_penalty) + paras * 30
def _best_content_node(root):
"""从清洗后的文档里挑选正文块(按文本密度打分)"""
if root is None:
return None
candidates = root.find_all(["article", "main", "div", "section"])
if not candidates:
return root
best, best_score = root, 0
for c in candidates:
score = _text_density(c)
if score > best_score:
best, best_score = c, score
return best
def _node_to_lines(node):
"""把正文块转成干净的按行文本(保留段落结构)"""
if node is None:
return []
lines = []
for el in node.find_all(["h1", "h2", "h3", "h4", "p", "li", "pre", "blockquote", "td", "th"]):
if el.find_parent("pre") is not None and el.name != "pre":
continue
t = el.get_text(" ", strip=True)
t = re.sub(r"\s+", " ", t)
if len(t) >= 2:
lines.append(t)
if not lines:
t = node.get_text(" ", strip=True)
t = re.sub(r"\s+", " ", t)
lines = [t] if t else []
return lines
def clean_html(html, url=""):
"""抓到的原始 HTML -> (页面标题, 干净可读正文纯文本)"""
soup = BeautifulSoup(html, "lxml")
title = (soup.title.get_text(strip=True) if soup.title else "") or url
body = _strip_noise(soup)
node = _best_content_node(body)
text = "\n".join(_node_to_lines(node))
text = re.sub(r"\n{3,}", "\n\n", text).strip()
return title, text
# ---------------- 链接提取 ----------------
def _is_plausible_article(a, base_url):
href = (a.get("href") or "").strip()
if not href or href.startswith(_BAD_HREFS):
return False
low = href.lower()
if any(low.endswith(e) for e in _BAD_EXTS):
return False
if urlparse(urljoin(base_url, href)).fragment:
return False
text = re.sub(r"\s+", " ", a.get_text(" ", strip=True)).strip()
if len(text) < 10:
return False
if text.lower().strip() in _NAV_WORDS:
return False
if re.fullmatch(r"[\d\s·|/\\\-_]+", text):
return False
return True
def extract_links(html, base_url, max_n=10):
"""从页面提取候选资讯链接 -> [{'title','url','summary'}](按文章相似度排序)"""
soup = BeautifulSoup(html, "lxml")
cands, seen = [], set()
for a in soup.find_all("a", href=True):
href = urljoin(base_url, a["href"].strip())
if href in seen or not _is_plausible_article(a, base_url):
continue
seen.add(href)
text = re.sub(r"\s+", " ", a.get_text(" ", strip=True)).strip()
text = _TRAIL_DATE_RE.sub("", text).strip()
if len(text) < 10:
continue
# 标题在标题标签/文章块内 → 更可能是资讯
in_head = 1 if a.find_parent(["h1", "h2", "h3", "h4", "article", "main"]) else 0
parent = a.find_parent(["li", "p", "h1", "h2", "h3", "h4", "article"]) or a.parent
summary = ""
if parent is not None and getattr(parent, "get_text", None):
summary = re.sub(r"\s+", " ", parent.get_text(" ", strip=True))[:200]
cands.append({"title": text, "url": href, "summary": summary,
"score": len(text) + in_head * 200})
cands.sort(key=lambda x: x["score"], reverse=True)
return [{"title": c["title"], "url": c["url"], "summary": c["summary"]}
for c in cands[:max_n]]
# ---------------- web-capture-api 集成(数据源网页抓取) ----------------
def _webcapture_cfg():
cfg = dict(config.WEBCAPTURE_DEFAULTS)
cfg.update(db.get_all_settings().get("webcapture", {}))
return cfg
def _source_capture_params(source):
"""解析数据源的抓取参数(capture_params JSON -> dict"""
raw = source.get("capture_params") or "{}"
try:
p = json.loads(raw)
except Exception:
p = {}
return p if isinstance(p, dict) else {}
def _call_webcapture(url, action, params=None):
"""调用 web-capture-api 抓取网页(html/text)。失败抛异常。
返回: {"success": True, "title": ..., "html"|"text": ...}
"""
cfg = _webcapture_cfg()
base = (cfg.get("api_url") or "").rstrip("/")
if not base:
raise RuntimeError("web-capture-api 地址未配置(设置页可修改)")
payload = {"url": url, "action": action}
p = params or {}
for k in ("wait_time", "scroll_times", "scroll_delay", "full_page", "backend", "viewport"):
if k in p and p[k] not in (None, ""):
payload[k] = p[k]
r = requests.post(f"{base}/api/capture", json=payload,
timeout=int(cfg.get("timeout", 60)))
r.raise_for_status()
data = r.json()
if not data.get("success"):
raise RuntimeError(data.get("error", "web-capture-api 返回失败"))
return data
# ---------------- 按源采集 ----------------
def _now():
return datetime.now().strftime("%Y-%m-%d %H:%M:%S")
def _fetch_list_via_direct(url):
"""直接抓取(requests + bs4):返回 (title, page_text, links)"""
per = config.CRAWL_DEFAULTS["per_source_links"]
html = fetch_page(url)
page_title, page_text = clean_html(html, url)
links = extract_links(html, url, max_n=per)
return page_title, page_text, links
def _fetch_list_via_webcapture(url, params):
"""走 web-capture-api 抓取列表页:返回 (title, page_text, links)"""
per = config.CRAWL_DEFAULTS["per_source_links"]
action = (params.get("action") or "html").strip() or "html"
data = _call_webcapture(url, action, params)
title = data.get("title") or url
if action == "text":
# 只取正文,无法提取子链接 → 整页作为一条
return title, (data.get("text") or ""), []
html = data.get("html") or ""
page_text = (data.get("text") or "").strip()
if not page_text and html:
page_text = clean_html(html, url)[1]
links = extract_links(html, url, max_n=per) if html else []
return title, page_text, links
def _fetch_full_via_direct(url):
return clean_html(fetch_page(url), url)[1]
def _fetch_full_via_webcapture(url, params):
data = _call_webcapture(url, "text", params)
return data.get("text") or ""
def fetch_source(source):
"""采集单个数据源 -> list[item]。
example.com 占位源返回模拟数据;真实源抓取失败抛异常(由调用方捕获并标记 error)。
fetch_methodauto=优先 web-capture-api、失败回退直接抓取;webcapture=仅 web-capture-api
direct=直接抓取(requests+bs4)。各源可单独配置 capture_params 抓取参数。
"""
url = (source.get("url") or "").strip()
# 模拟源(example.com 占位)→ 用仿真数据填充(补齐 source_id,保证定制监控识别正确)
if not url or "example.com" in url:
items = simulate.items_for_source(source["id"], source.get("type", ""))
for it in items:
it["source_id"] = source["id"]
return items
method = (source.get("fetch_method") or "auto").strip() or "auto"
per = config.CRAWL_DEFAULTS["per_source_links"]
full = config.CRAWL_DEFAULTS["full_fetch_links"]
params = _source_capture_params(source)
# 抓取列表页(标题 + 正文 + 候选链接)
if method == "direct":
page_title, page_text, links = _fetch_list_via_direct(url)
else:
try:
page_title, page_text, links = _fetch_list_via_webcapture(url, params)
except Exception:
if method == "webcapture":
raise
# auto:回退直接抓取
page_title, page_text, links = _fetch_list_via_direct(url)
items = []
for i, lk in enumerate(links[:per]):
full_text = ""
content = lk["summary"]
if i < full and lk["url"]:
try:
if method == "direct":
full_text = _fetch_full_via_direct(lk["url"])
else:
try:
full_text = _fetch_full_via_webcapture(lk["url"], params)
except Exception:
if method == "webcapture":
raise
full_text = _fetch_full_via_direct(lk["url"])
if not content or len(content) < len(full_text):
content = full_text
except Exception:
full_text = ""
items.append({
"title": lk["title"],
"url": lk["url"],
"author": source.get("name", ""),
"content": (content or page_text)[:4000],
"summary": (lk["summary"] or content or page_text)[:220],
"domain": source.get("type") or "",
"entities": [],
"source_id": source["id"],
"published_at": _now(),
"full_text": full_text or "",
})
if not items and page_text:
# 页面本身即正文(如单篇/无链接页/action=text)→ 整页作为一条
items.append({
"title": page_title,
"url": url,
"author": source.get("name", ""),
"content": page_text[:4000],
"summary": page_text[:220],
"domain": source.get("type") or "",
"entities": [],
"source_id": source["id"],
"published_at": _now(),
"full_text": page_text,
})
return items
def fetch_all():
"""采集全部启用数据源 -> (items, {source_id: 条数})
真实源失败:跳过并标记 sources.status=error(网页可见),不塞模拟数据。"""
sources = db.list_sources(only_enabled=True)
items, per, failed = [], {}, []
for s in sources:
try:
got = fetch_source(s)
per[s["id"]] = len(got)
items.extend(got)
except Exception:
failed.append(s["id"])
per[s["id"]] = 0
# 标记各源采集状态
for sid in failed:
db.update_source_fetch(sid, status="error", count=0)
for sid in set(per.keys()) - set(failed):
db.update_source_fetch(sid, status="ok", count=per.get(sid, 0))
return items, per