Files
2026-07-19 18:11:08 +08:00

302 lines
11 KiB
Python

#!/usr/bin/env python3
"""Incrementally collect Hongcan articles into SQLite using a real browser."""
from __future__ import annotations
import argparse
import hashlib
import json
import re
import sqlite3
import sys
import time
from contextlib import closing
from datetime import datetime, timedelta, timezone
from pathlib import Path
from urllib.parse import urljoin, urlsplit, urlunsplit
from playwright.sync_api import BrowserContext, Page, sync_playwright
DEFAULT_LIST_URL = "https://m.canyin88.com/zixun/"
ARTICLE_PATH_RE = re.compile(r"/zixun/\d{4}/\d{1,2}/\d{1,2}/\d+\.html$")
DATE_RE = re.compile(r"(20\d{2})[-/.年](\d{1,2})[-/.月](\d{1,2})(?:日)?(?:\s+(\d{1,2}):(\d{2})(?::(\d{2}))?)?")
def now_iso() -> str:
return datetime.now(timezone.utc).astimezone().isoformat(timespec="seconds")
def normalize_url(raw_url: str, base_url: str = DEFAULT_LIST_URL) -> str | None:
absolute = urljoin(base_url, raw_url.strip())
parts = urlsplit(absolute)
host = parts.netloc.lower().removeprefix("m.").removeprefix("www.")
if host != "canyin88.com" or not ARTICLE_PATH_RE.search(parts.path):
return None
return urlunsplit(("https", "www.canyin88.com", parts.path, "", ""))
def clean_text(value: str | None) -> str:
return re.sub(r"\s+", " ", value or "").strip().lstrip("\ufeff")
def parse_date(value: str | None) -> str | None:
if not value:
return None
match = DATE_RE.search(value)
if not match:
return None
year, month, day, hour, minute, second = match.groups()
dt = datetime(
int(year), int(month), int(day), int(hour or 0), int(minute or 0), int(second or 0)
)
return dt.isoformat(timespec="seconds")
def date_from_article_url(url: str) -> datetime | None:
match = re.search(r"/zixun/(20\d{2})/(\d{1,2})/(\d{1,2})/", url)
if not match:
return None
return datetime(*(int(part) for part in match.groups()))
def init_db(conn: sqlite3.Connection, schema_path: Path) -> None:
conn.executescript(schema_path.read_text(encoding="utf-8"))
def discover_urls(page: Page, list_url: str, scrolls: int, pause_ms: int) -> list[str]:
page.goto(list_url, wait_until="domcontentloaded", timeout=60_000)
page.wait_for_timeout(2_000)
found: set[str] = set()
stagnant = 0
for _ in range(scrolls + 1):
hrefs = page.locator("a[href]").evaluate_all("els => els.map(e => e.href)")
before = len(found)
for href in hrefs:
normalized = normalize_url(href, list_url)
if normalized:
found.add(normalized)
stagnant = stagnant + 1 if len(found) == before else 0
if stagnant >= 3:
break
page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
page.wait_for_timeout(pause_ms)
return sorted(found, reverse=True)
def meta(page: Page, selector: str) -> str:
locator = page.locator(selector).first
return clean_text(locator.get_attribute("content")) if locator.count() else ""
def first_text(page: Page, selectors: list[str]) -> str:
for selector in selectors:
locator = page.locator(selector).first
if locator.count():
text = clean_text(locator.inner_text(timeout=3_000))
if text:
return text
return ""
def extract_article(page: Page, url: str) -> dict[str, str | None]:
page.goto(url, wait_until="domcontentloaded", timeout=60_000)
page.wait_for_timeout(800)
document_title = clean_text(page.title())
document_title = re.sub(r"(?:[_-]红餐网.*)$", "", document_title).strip()
title = meta(page, 'meta[property="og:title"]') or document_title
if not title or title == "相关推荐":
title = first_text(page, [".title", "h1"])
summary = (
meta(page, 'meta[name="description"]')
or meta(page, 'meta[property="og:description"]')
)
author = (
meta(page, 'meta[name="author"]')
or first_text(page, [".author", ".source", "[class*=author]"])
)
published_raw = (
meta(page, 'meta[property="article:published_time"]')
or first_text(page, ["time", ".time", "[class*=time]", "[class*=date]"])
or page.locator("body").inner_text(timeout=5_000)[:1000]
)
content = first_text(
page,
[
"article",
".article-content",
".article_content",
".content",
"[class*=detail-content]",
"[class*=article] [class*=content]",
],
)
if not content:
paragraphs = page.locator("p").all_inner_texts()
content = "\n\n".join(clean_text(p) for p in paragraphs if len(clean_text(p)) >= 20)
if not title or len(content) < 80:
raise ValueError(f"article extraction incomplete: title={bool(title)}, content_length={len(content)}")
published_at = parse_date(published_raw)
url_date = date_from_article_url(url)
if url_date and (
not published_at or datetime.fromisoformat(published_at).date() != url_date.date()
):
published_at = url_date.isoformat(timespec="seconds")
content_hash = hashlib.sha256(f"{title}\n{content}".encode("utf-8")).hexdigest()
return {
"canonical_url": url,
"title": title,
"author": author or None,
"published_at": published_at,
"summary": summary or None,
"content": content,
"content_hash": content_hash,
}
def pending_urls(conn: sqlite3.Connection, discovered: list[str], max_retries: int) -> list[str]:
rows = conn.execute(
"SELECT canonical_url FROM articles WHERE crawl_status != 'success' AND retry_count < ?",
(max_retries,),
).fetchall()
return list(dict.fromkeys(discovered + [row[0] for row in rows]))
def insert_discoveries(conn: sqlite3.Connection, urls: list[str]) -> int:
timestamp = now_iso()
inserted = 0
for url in urls:
cursor = conn.execute(
"""
INSERT OR IGNORE INTO articles
(canonical_url, discovered_at, updated_at)
VALUES (?, ?, ?)
""",
(url, timestamp, timestamp),
)
inserted += cursor.rowcount
conn.commit()
return inserted
def save_success(conn: sqlite3.Connection, article: dict[str, str | None]) -> None:
timestamp = now_iso()
conn.execute(
"""
UPDATE articles SET
title = ?, author = ?, published_at = ?, summary = ?, content = ?,
content_hash = ?, crawl_status = 'success', last_error = NULL,
crawled_at = ?, updated_at = ?
WHERE canonical_url = ?
""",
(
article["title"], article["author"], article["published_at"], article["summary"],
article["content"], article["content_hash"], timestamp, timestamp,
article["canonical_url"],
),
)
conn.commit()
def save_failure(conn: sqlite3.Connection, url: str, error: Exception) -> None:
conn.execute(
"""
UPDATE articles SET
crawl_status = 'failed', retry_count = retry_count + 1,
last_error = ?, updated_at = ?
WHERE canonical_url = ?
""",
(str(error)[:1000], now_iso(), url),
)
conn.commit()
def new_context(playwright, headless: bool) -> tuple[BrowserContext, Page]:
browser = playwright.chromium.launch(headless=headless)
context = browser.new_context(
locale="zh-CN",
timezone_id="Asia/Shanghai",
viewport={"width": 1280, "height": 900},
user_agent=(
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36"
),
)
context.set_default_timeout(15_000)
return context, context.new_page()
def run(args: argparse.Namespace) -> int:
base_dir = Path(__file__).resolve().parent
db_path = Path(args.db).expanduser().resolve()
db_path.parent.mkdir(parents=True, exist_ok=True)
with closing(sqlite3.connect(db_path)) as conn:
init_db(conn, base_dir / "schema.sql")
started_at = now_iso()
run_id = conn.execute(
"INSERT INTO crawl_runs(started_at) VALUES (?)", (started_at,)
).lastrowid
conn.commit()
discovered_count = inserted_count = updated_count = failed_count = 0
try:
with sync_playwright() as playwright:
context, page = new_context(playwright, args.headed is False)
try:
urls = discover_urls(page, args.list_url, args.scrolls, args.pause_ms)
discovered_count = len(urls)
inserted_count = insert_discoveries(conn, urls)
queue = pending_urls(conn, urls, args.max_retries)
cutoff = datetime.now() - timedelta(days=args.lookback_days)
for index, url in enumerate(queue, start=1):
path_date = parse_date(url.replace("/", "-"))
if path_date and datetime.fromisoformat(path_date) < cutoff:
continue
try:
article = extract_article(page, url)
save_success(conn, article)
updated_count += 1
print(f"[{index}/{len(queue)}] OK {article['title']}")
except Exception as exc:
failed_count += 1
save_failure(conn, url, exc)
print(f"[{index}/{len(queue)}] FAIL {url}: {exc}", file=sys.stderr)
time.sleep(args.article_pause)
finally:
context.close()
status = "success" if failed_count == 0 else "partial"
message = json.dumps({"db": str(db_path)}, ensure_ascii=False)
except Exception as exc:
status, message = "failed", str(exc)[:1000]
print(f"crawl failed: {exc}", file=sys.stderr)
conn.execute(
"""
UPDATE crawl_runs SET finished_at=?, status=?, discovered_count=?,
inserted_count=?, updated_count=?, failed_count=?, message=?
WHERE id=?
""",
(now_iso(), status, discovered_count, inserted_count, updated_count, failed_count, message, run_id),
)
conn.commit()
print(json.dumps({
"status": status, "discovered": discovered_count, "inserted": inserted_count,
"updated": updated_count, "failed": failed_count, "database": str(db_path),
}, ensure_ascii=False))
return 0 if status in {"success", "partial"} else 1
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description="红餐网文章每日增量采集入库")
parser.add_argument("--db", default="hongcan.db", help="SQLite 数据库路径")
parser.add_argument("--list-url", default=DEFAULT_LIST_URL)
parser.add_argument("--lookback-days", type=int, default=7, help="回溯天数")
parser.add_argument("--scrolls", type=int, default=12, help="列表页最大下拉次数")
parser.add_argument("--pause-ms", type=int, default=1200, help="列表页下拉等待毫秒")
parser.add_argument("--article-pause", type=float, default=0.8, help="文章间隔秒数")
parser.add_argument("--max-retries", type=int, default=3)
parser.add_argument("--headed", action="store_true", help="显示浏览器,便于排查验证码")
return parser
if __name__ == "__main__":
raise SystemExit(run(build_parser().parse_args()))