302 lines
11 KiB
Python
302 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""Incrementally collect Hongcan articles into SQLite using a real browser."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import re
|
|
import sqlite3
|
|
import sys
|
|
import time
|
|
from contextlib import closing
|
|
from datetime import datetime, timedelta, timezone
|
|
from pathlib import Path
|
|
from urllib.parse import urljoin, urlsplit, urlunsplit
|
|
|
|
from playwright.sync_api import BrowserContext, Page, sync_playwright
|
|
|
|
|
|
DEFAULT_LIST_URL = "https://m.canyin88.com/zixun/"
|
|
ARTICLE_PATH_RE = re.compile(r"/zixun/\d{4}/\d{1,2}/\d{1,2}/\d+\.html$")
|
|
DATE_RE = re.compile(r"(20\d{2})[-/.年](\d{1,2})[-/.月](\d{1,2})(?:日)?(?:\s+(\d{1,2}):(\d{2})(?::(\d{2}))?)?")
|
|
|
|
|
|
def now_iso() -> str:
|
|
return datetime.now(timezone.utc).astimezone().isoformat(timespec="seconds")
|
|
|
|
|
|
def normalize_url(raw_url: str, base_url: str = DEFAULT_LIST_URL) -> str | None:
|
|
absolute = urljoin(base_url, raw_url.strip())
|
|
parts = urlsplit(absolute)
|
|
host = parts.netloc.lower().removeprefix("m.").removeprefix("www.")
|
|
if host != "canyin88.com" or not ARTICLE_PATH_RE.search(parts.path):
|
|
return None
|
|
return urlunsplit(("https", "www.canyin88.com", parts.path, "", ""))
|
|
|
|
|
|
def clean_text(value: str | None) -> str:
|
|
return re.sub(r"\s+", " ", value or "").strip().lstrip("\ufeff")
|
|
|
|
|
|
def parse_date(value: str | None) -> str | None:
|
|
if not value:
|
|
return None
|
|
match = DATE_RE.search(value)
|
|
if not match:
|
|
return None
|
|
year, month, day, hour, minute, second = match.groups()
|
|
dt = datetime(
|
|
int(year), int(month), int(day), int(hour or 0), int(minute or 0), int(second or 0)
|
|
)
|
|
return dt.isoformat(timespec="seconds")
|
|
|
|
|
|
def date_from_article_url(url: str) -> datetime | None:
|
|
match = re.search(r"/zixun/(20\d{2})/(\d{1,2})/(\d{1,2})/", url)
|
|
if not match:
|
|
return None
|
|
return datetime(*(int(part) for part in match.groups()))
|
|
|
|
|
|
def init_db(conn: sqlite3.Connection, schema_path: Path) -> None:
|
|
conn.executescript(schema_path.read_text(encoding="utf-8"))
|
|
|
|
|
|
def discover_urls(page: Page, list_url: str, scrolls: int, pause_ms: int) -> list[str]:
|
|
page.goto(list_url, wait_until="domcontentloaded", timeout=60_000)
|
|
page.wait_for_timeout(2_000)
|
|
found: set[str] = set()
|
|
stagnant = 0
|
|
for _ in range(scrolls + 1):
|
|
hrefs = page.locator("a[href]").evaluate_all("els => els.map(e => e.href)")
|
|
before = len(found)
|
|
for href in hrefs:
|
|
normalized = normalize_url(href, list_url)
|
|
if normalized:
|
|
found.add(normalized)
|
|
stagnant = stagnant + 1 if len(found) == before else 0
|
|
if stagnant >= 3:
|
|
break
|
|
page.evaluate("window.scrollTo(0, document.body.scrollHeight)")
|
|
page.wait_for_timeout(pause_ms)
|
|
return sorted(found, reverse=True)
|
|
|
|
|
|
def meta(page: Page, selector: str) -> str:
|
|
locator = page.locator(selector).first
|
|
return clean_text(locator.get_attribute("content")) if locator.count() else ""
|
|
|
|
|
|
def first_text(page: Page, selectors: list[str]) -> str:
|
|
for selector in selectors:
|
|
locator = page.locator(selector).first
|
|
if locator.count():
|
|
text = clean_text(locator.inner_text(timeout=3_000))
|
|
if text:
|
|
return text
|
|
return ""
|
|
|
|
|
|
def extract_article(page: Page, url: str) -> dict[str, str | None]:
|
|
page.goto(url, wait_until="domcontentloaded", timeout=60_000)
|
|
page.wait_for_timeout(800)
|
|
document_title = clean_text(page.title())
|
|
document_title = re.sub(r"(?:[_-]红餐网.*)$", "", document_title).strip()
|
|
title = meta(page, 'meta[property="og:title"]') or document_title
|
|
if not title or title == "相关推荐":
|
|
title = first_text(page, [".title", "h1"])
|
|
summary = (
|
|
meta(page, 'meta[name="description"]')
|
|
or meta(page, 'meta[property="og:description"]')
|
|
)
|
|
author = (
|
|
meta(page, 'meta[name="author"]')
|
|
or first_text(page, [".author", ".source", "[class*=author]"])
|
|
)
|
|
published_raw = (
|
|
meta(page, 'meta[property="article:published_time"]')
|
|
or first_text(page, ["time", ".time", "[class*=time]", "[class*=date]"])
|
|
or page.locator("body").inner_text(timeout=5_000)[:1000]
|
|
)
|
|
content = first_text(
|
|
page,
|
|
[
|
|
"article",
|
|
".article-content",
|
|
".article_content",
|
|
".content",
|
|
"[class*=detail-content]",
|
|
"[class*=article] [class*=content]",
|
|
],
|
|
)
|
|
if not content:
|
|
paragraphs = page.locator("p").all_inner_texts()
|
|
content = "\n\n".join(clean_text(p) for p in paragraphs if len(clean_text(p)) >= 20)
|
|
if not title or len(content) < 80:
|
|
raise ValueError(f"article extraction incomplete: title={bool(title)}, content_length={len(content)}")
|
|
published_at = parse_date(published_raw)
|
|
url_date = date_from_article_url(url)
|
|
if url_date and (
|
|
not published_at or datetime.fromisoformat(published_at).date() != url_date.date()
|
|
):
|
|
published_at = url_date.isoformat(timespec="seconds")
|
|
content_hash = hashlib.sha256(f"{title}\n{content}".encode("utf-8")).hexdigest()
|
|
return {
|
|
"canonical_url": url,
|
|
"title": title,
|
|
"author": author or None,
|
|
"published_at": published_at,
|
|
"summary": summary or None,
|
|
"content": content,
|
|
"content_hash": content_hash,
|
|
}
|
|
|
|
|
|
def pending_urls(conn: sqlite3.Connection, discovered: list[str], max_retries: int) -> list[str]:
|
|
rows = conn.execute(
|
|
"SELECT canonical_url FROM articles WHERE crawl_status != 'success' AND retry_count < ?",
|
|
(max_retries,),
|
|
).fetchall()
|
|
return list(dict.fromkeys(discovered + [row[0] for row in rows]))
|
|
|
|
|
|
def insert_discoveries(conn: sqlite3.Connection, urls: list[str]) -> int:
|
|
timestamp = now_iso()
|
|
inserted = 0
|
|
for url in urls:
|
|
cursor = conn.execute(
|
|
"""
|
|
INSERT OR IGNORE INTO articles
|
|
(canonical_url, discovered_at, updated_at)
|
|
VALUES (?, ?, ?)
|
|
""",
|
|
(url, timestamp, timestamp),
|
|
)
|
|
inserted += cursor.rowcount
|
|
conn.commit()
|
|
return inserted
|
|
|
|
|
|
def save_success(conn: sqlite3.Connection, article: dict[str, str | None]) -> None:
|
|
timestamp = now_iso()
|
|
conn.execute(
|
|
"""
|
|
UPDATE articles SET
|
|
title = ?, author = ?, published_at = ?, summary = ?, content = ?,
|
|
content_hash = ?, crawl_status = 'success', last_error = NULL,
|
|
crawled_at = ?, updated_at = ?
|
|
WHERE canonical_url = ?
|
|
""",
|
|
(
|
|
article["title"], article["author"], article["published_at"], article["summary"],
|
|
article["content"], article["content_hash"], timestamp, timestamp,
|
|
article["canonical_url"],
|
|
),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def save_failure(conn: sqlite3.Connection, url: str, error: Exception) -> None:
|
|
conn.execute(
|
|
"""
|
|
UPDATE articles SET
|
|
crawl_status = 'failed', retry_count = retry_count + 1,
|
|
last_error = ?, updated_at = ?
|
|
WHERE canonical_url = ?
|
|
""",
|
|
(str(error)[:1000], now_iso(), url),
|
|
)
|
|
conn.commit()
|
|
|
|
|
|
def new_context(playwright, headless: bool) -> tuple[BrowserContext, Page]:
|
|
browser = playwright.chromium.launch(headless=headless)
|
|
context = browser.new_context(
|
|
locale="zh-CN",
|
|
timezone_id="Asia/Shanghai",
|
|
viewport={"width": 1280, "height": 900},
|
|
user_agent=(
|
|
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0 Safari/537.36"
|
|
),
|
|
)
|
|
context.set_default_timeout(15_000)
|
|
return context, context.new_page()
|
|
|
|
|
|
def run(args: argparse.Namespace) -> int:
|
|
base_dir = Path(__file__).resolve().parent
|
|
db_path = Path(args.db).expanduser().resolve()
|
|
db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
with closing(sqlite3.connect(db_path)) as conn:
|
|
init_db(conn, base_dir / "schema.sql")
|
|
started_at = now_iso()
|
|
run_id = conn.execute(
|
|
"INSERT INTO crawl_runs(started_at) VALUES (?)", (started_at,)
|
|
).lastrowid
|
|
conn.commit()
|
|
discovered_count = inserted_count = updated_count = failed_count = 0
|
|
try:
|
|
with sync_playwright() as playwright:
|
|
context, page = new_context(playwright, args.headed is False)
|
|
try:
|
|
urls = discover_urls(page, args.list_url, args.scrolls, args.pause_ms)
|
|
discovered_count = len(urls)
|
|
inserted_count = insert_discoveries(conn, urls)
|
|
queue = pending_urls(conn, urls, args.max_retries)
|
|
cutoff = datetime.now() - timedelta(days=args.lookback_days)
|
|
for index, url in enumerate(queue, start=1):
|
|
path_date = parse_date(url.replace("/", "-"))
|
|
if path_date and datetime.fromisoformat(path_date) < cutoff:
|
|
continue
|
|
try:
|
|
article = extract_article(page, url)
|
|
save_success(conn, article)
|
|
updated_count += 1
|
|
print(f"[{index}/{len(queue)}] OK {article['title']}")
|
|
except Exception as exc:
|
|
failed_count += 1
|
|
save_failure(conn, url, exc)
|
|
print(f"[{index}/{len(queue)}] FAIL {url}: {exc}", file=sys.stderr)
|
|
time.sleep(args.article_pause)
|
|
finally:
|
|
context.close()
|
|
status = "success" if failed_count == 0 else "partial"
|
|
message = json.dumps({"db": str(db_path)}, ensure_ascii=False)
|
|
except Exception as exc:
|
|
status, message = "failed", str(exc)[:1000]
|
|
print(f"crawl failed: {exc}", file=sys.stderr)
|
|
conn.execute(
|
|
"""
|
|
UPDATE crawl_runs SET finished_at=?, status=?, discovered_count=?,
|
|
inserted_count=?, updated_count=?, failed_count=?, message=?
|
|
WHERE id=?
|
|
""",
|
|
(now_iso(), status, discovered_count, inserted_count, updated_count, failed_count, message, run_id),
|
|
)
|
|
conn.commit()
|
|
print(json.dumps({
|
|
"status": status, "discovered": discovered_count, "inserted": inserted_count,
|
|
"updated": updated_count, "failed": failed_count, "database": str(db_path),
|
|
}, ensure_ascii=False))
|
|
return 0 if status in {"success", "partial"} else 1
|
|
|
|
|
|
def build_parser() -> argparse.ArgumentParser:
|
|
parser = argparse.ArgumentParser(description="红餐网文章每日增量采集入库")
|
|
parser.add_argument("--db", default="hongcan.db", help="SQLite 数据库路径")
|
|
parser.add_argument("--list-url", default=DEFAULT_LIST_URL)
|
|
parser.add_argument("--lookback-days", type=int, default=7, help="回溯天数")
|
|
parser.add_argument("--scrolls", type=int, default=12, help="列表页最大下拉次数")
|
|
parser.add_argument("--pause-ms", type=int, default=1200, help="列表页下拉等待毫秒")
|
|
parser.add_argument("--article-pause", type=float, default=0.8, help="文章间隔秒数")
|
|
parser.add_argument("--max-retries", type=int, default=3)
|
|
parser.add_argument("--headed", action="store_true", help="显示浏览器,便于排查验证码")
|
|
return parser
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(run(build_parser().parse_args()))
|