628 lines
27 KiB
Python
628 lines
27 KiB
Python
#!/usr/bin/env python3
|
||
# -*- coding: utf-8 -*-
|
||
"""
|
||
SRE Weekly 抓取与文章提取脚本
|
||
|
||
模仿 blogwatcher-daily 的单文件风格:feedparser 抓取 → HTML 落盘 → 逐篇文章导出 Markdown,
|
||
并用 manifest.json 记录每期的状态(是否已存 HTML / 是否已提取文章)。
|
||
|
||
设计给 n8n 定时任务调用,通过不同参数驱动,脚本自身不发送邮件。
|
||
|
||
目录结构(默认工作目录 ~/Workspace/nexus/sreweekly/):
|
||
sreweekly/
|
||
├── manifest.json # 状态记录(每期: html 是否保存、文章是否提取、抓取统计)
|
||
├── html/<id>-<日期>.html # 每期原始 HTML(feed 的 content:encoded 原样保存)
|
||
├── articles/<id>/ # 每篇文章原始 HTML 缓存 + index.json 抓取日志
|
||
│ ├── 01-<slug>.html
|
||
│ └── index.json
|
||
└── markdown/<id>/ # 每期文章的 markdown 目录(每篇一个文件,含正文)
|
||
├── 01-<slug>.md
|
||
└── 02-<slug>.md
|
||
|
||
用法:
|
||
python3 sreweekly.py # 扫描 feed,新增期保存为 HTML 并登记 manifest
|
||
python3 sreweekly.py --check # 只检查有无新期,输出 533,532 或 NONE(供 n8n 分支)
|
||
python3 sreweekly.py --extract ID # 提取指定期(如 533)的文章 → markdown(默认抓正文)
|
||
python3 sreweekly.py --extract 433..464 # 提取期号区间(只处理 manifest 中已登记的期)
|
||
python3 sreweekly.py --extract latest # 提取最新一期
|
||
python3 sreweekly.py --extract pending # 提取所有尚未提取的期(n8n 每日流程推荐)
|
||
python3 sreweekly.py --extract all # 提取全部(含已提取的,需配合 --force 覆盖)
|
||
python3 sreweekly.py --status # 打印 manifest 状态表
|
||
python3 sreweekly.py --issues # 列出已登记的所有期
|
||
python3 sreweekly.py --backfill 400..500 # 回溯抓取指定期号区间(绕过 RSS 只有最近 ~10 期的限制)
|
||
|
||
选项:
|
||
--feed URL RSS 地址(默认 https://sreweekly.com/feed/)
|
||
--dir PATH 工作目录(默认 ~/Workspace/nexus/sreweekly/)
|
||
--force 重新提取已提取过的期(覆盖旧 markdown)
|
||
--skip-fetch 不抓文章正文,只用 feed 简介(老行为)
|
||
--force-fetch 忽略 articles/ 本地缓存,重新联网抓
|
||
--sleep 秒数 每篇文章抓取之间的间隔(默认 1.5,避免打站过快)
|
||
|
||
依赖:
|
||
pip3 install feedparser trafilatura
|
||
"""
|
||
from html.parser import HTMLParser
|
||
import argparse, html, http.client, json, os, re, sys, time, urllib.error, urllib.parse, urllib.request
|
||
|
||
try:
|
||
import feedparser
|
||
except ImportError:
|
||
sys.exit("缺少依赖: 请先执行 pip3 install feedparser")
|
||
|
||
try:
|
||
import trafilatura
|
||
except ImportError:
|
||
trafilatura = None # 抓正文时才强制要求
|
||
|
||
DEFAULT_FEED = "https://sreweekly.com/feed/"
|
||
DEFAULT_DIR = os.path.expanduser("~/Workspace/nexus/sreweekly")
|
||
DEFAULT_UA = (
|
||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0 Safari/537.36"
|
||
)
|
||
FETCH_TIMEOUT = 20
|
||
NON_HTML_EXTS = (".pdf", ".zip", ".tar", ".gz", ".mp4", ".mp3", ".png", ".jpg", ".jpeg", ".gif", ".svg")
|
||
|
||
# ---------------------------------------------------------------- manifest
|
||
|
||
def load_manifest(base_dir):
|
||
path = os.path.join(base_dir, "manifest.json")
|
||
if os.path.exists(path):
|
||
try:
|
||
with open(path, "r", encoding="utf-8") as f:
|
||
return json.load(f)
|
||
except (json.JSONDecodeError, OSError) as e:
|
||
print(f"⚠️ manifest.json 读取失败({e}),按空库处理", file=sys.stderr)
|
||
return {"feed": DEFAULT_FEED, "last_scanned": None, "issues": {}}
|
||
|
||
def save_manifest(base_dir, manifest):
|
||
path = os.path.join(base_dir, "manifest.json")
|
||
with open(path, "w", encoding="utf-8") as f:
|
||
json.dump(manifest, f, ensure_ascii=False, indent=2)
|
||
return path
|
||
|
||
def issue_id_from_link(link):
|
||
"""从文章链接提取期号: sre-weekly-issue-533 → 533;失败则用最后一段路径。"""
|
||
m = re.search(r"sre-weekly-issue-(\d+)", link or "")
|
||
if m:
|
||
return int(m.group(1))
|
||
m = re.search(r"/([^/]+?)/?$", (link or "").rstrip("/"))
|
||
return m.group(1) if m else "unknown"
|
||
|
||
def format_pubdate(entry):
|
||
"""优先用 feed 自带发布日期(YYYY-MM-DD),没有则用当天。"""
|
||
if getattr(entry, "published_parsed", None):
|
||
return time.strftime("%Y-%m-%d", entry.published_parsed)
|
||
return time.strftime("%Y-%m-%d")
|
||
|
||
def now_iso():
|
||
return time.strftime("%Y-%m-%dT%H:%M:%S")
|
||
|
||
# ---------------------------------------------------------------- feed 扫描
|
||
|
||
def fetch_feed(feed_url):
|
||
d = feedparser.parse(feed_url)
|
||
if not d.entries:
|
||
sys.exit(f"❌ feed 抓取失败或为空: {feed_url}(bozo={d.bozo})")
|
||
return d.entries
|
||
|
||
def cmd_scan(args, manifest, base_dir):
|
||
html_dir = os.path.join(base_dir, "html")
|
||
os.makedirs(html_dir, exist_ok=True)
|
||
entries = fetch_feed(args.feed)
|
||
new_ids = []
|
||
for e in entries:
|
||
iid = str(issue_id_from_link(e.link))
|
||
if iid in manifest["issues"]:
|
||
continue # 已登记,跳过(去重)
|
||
content_html = e.content[0].value if getattr(e, "content", None) else ""
|
||
pubdate = format_pubdate(e)
|
||
fname = f"{iid}-{pubdate}.html"
|
||
fpath = os.path.join(html_dir, fname)
|
||
with open(fpath, "w", encoding="utf-8") as f:
|
||
f.write(content_html)
|
||
manifest["issues"][iid] = {
|
||
"id": iid,
|
||
"title": e.title,
|
||
"url": e.link,
|
||
"pub_date": pubdate,
|
||
"html_file": f"html/{fname}",
|
||
"fetched_at": now_iso(),
|
||
"extracted": False,
|
||
"article_count": 0,
|
||
"markdown_dir": None,
|
||
}
|
||
new_ids.append(iid)
|
||
print(f"🆕 新期 #{iid} {e.title} → {fname}({len(content_html)} bytes)", file=sys.stderr)
|
||
manifest["last_scanned"] = now_iso()
|
||
save_manifest(base_dir, manifest)
|
||
print(f"📊 扫描完成:新增 {len(new_ids)} 期,共 {len(manifest['issues'])} 期", file=sys.stderr)
|
||
return new_ids
|
||
|
||
# ---------------------------------------------------------------- 回溯抓取(老期刊)
|
||
|
||
MONTHS = {m: i for i, m in enumerate(
|
||
["January","February","March","April","May","June",
|
||
"July","August","September","October","November","December"], 1)}
|
||
|
||
def parse_id_range(spec, flag_name="--backfill"):
|
||
"""'400..500' → (400, 500);'450' → (450, 450)。"""
|
||
m = re.match(r"^\s*(\d+)\s*(?:\.\.\s*(\d+))?\s*$", spec or "")
|
||
if not m:
|
||
sys.exit(f"❌ {flag_name} 参数格式错误(示例 400..500 或 450): {spec}")
|
||
lo = int(m.group(1)); hi = int(m.group(2) or m.group(1))
|
||
if lo > hi: lo, hi = hi, lo
|
||
return lo, hi
|
||
|
||
def extract_pub_date_from_page(html_text):
|
||
"""从 sreweekly 站点页面 HTML 里提取发布日期 → YYYY-MM-DD,失败返回 None。
|
||
|
||
结构: <a class="updated" href="...">June 23, 2024</a>
|
||
"""
|
||
if not html_text:
|
||
return None
|
||
m = re.search(r'class="updated"[^>]*>\s*([A-Za-z]+)\s+(\d{1,2}),\s*(\d{4})', html_text)
|
||
if not m:
|
||
return None
|
||
mon = MONTHS.get(m.group(1))
|
||
if not mon:
|
||
return None
|
||
return f"{int(m.group(3)):04d}-{mon:02d}-{int(m.group(2)):02d}"
|
||
|
||
def extract_title_from_page(html_text, iid):
|
||
m = re.search(r"<title>\s*(.+?)\s*</title>", html_text or "", re.I | re.S)
|
||
if not m:
|
||
return f"SRE Weekly Issue #{iid}"
|
||
t = html.unescape(m.group(1))
|
||
# 去掉站点后缀 " – SRE WEEKLY" / " - SRE WEEKLY"
|
||
t = re.sub(r"\s*[\u2013\u2014\-]\s*SRE WEEKLY\s*$", "", t, flags=re.I).strip()
|
||
return t or f"SRE Weekly Issue #{iid}"
|
||
|
||
def cmd_backfill(args, manifest, base_dir):
|
||
lo, hi = parse_id_range(args.backfill, "--backfill")
|
||
html_dir = os.path.join(base_dir, "html")
|
||
page_cache_dir = os.path.join(base_dir, "pages") # 存整页 HTML 作缓存
|
||
os.makedirs(html_dir, exist_ok=True)
|
||
added, skipped, missing = [], [], []
|
||
for n in range(lo, hi + 1):
|
||
iid = str(n)
|
||
if iid in manifest["issues"]:
|
||
skipped.append(iid)
|
||
continue
|
||
url = f"https://sreweekly.com/sre-weekly-issue-{n}/"
|
||
cache_path = os.path.join(page_cache_dir, f"{n}.html")
|
||
raw, err = fetch_article_html(url, cache_path, force=args.force_fetch)
|
||
if not raw:
|
||
print(f"⚠️ #{iid} 抓取失败:{err}({url})", file=sys.stderr)
|
||
missing.append(iid)
|
||
if args.sleep > 0:
|
||
time.sleep(args.sleep)
|
||
continue
|
||
# 站点页 HTML 含 sreweekly-entry 结构,SREEntryParser 兼容;直接落到 html/ 供后续 --extract 使用
|
||
pubdate = extract_pub_date_from_page(raw) or time.strftime("%Y-%m-%d")
|
||
title = extract_title_from_page(raw, iid)
|
||
fname = f"{iid}-{pubdate}.html"
|
||
with open(os.path.join(html_dir, fname), "w", encoding="utf-8") as f:
|
||
f.write(raw)
|
||
manifest["issues"][iid] = {
|
||
"id": iid,
|
||
"title": title,
|
||
"url": url,
|
||
"pub_date": pubdate,
|
||
"html_file": f"html/{fname}",
|
||
"fetched_at": now_iso(),
|
||
"source": "backfill",
|
||
"extracted": False,
|
||
"article_count": 0,
|
||
"markdown_dir": None,
|
||
}
|
||
added.append(iid)
|
||
print(f"🆕 回溯 #{iid} {title} → {fname}({len(raw)} bytes)", file=sys.stderr)
|
||
if args.sleep > 0 and n < hi:
|
||
time.sleep(args.sleep)
|
||
save_manifest(base_dir, manifest)
|
||
print(f"📊 回溯完成:新增 {len(added)} 期 / 跳过已有 {len(skipped)} / 抓取失败 {len(missing)}", file=sys.stderr)
|
||
if missing:
|
||
print(f" 失败期号: {','.join(missing)}", file=sys.stderr)
|
||
return added
|
||
|
||
def cmd_check(args, manifest):
|
||
entries = fetch_feed(args.feed)
|
||
known = set(manifest["issues"].keys())
|
||
new_ids = []
|
||
for e in entries:
|
||
iid = str(issue_id_from_link(e.link))
|
||
if iid not in known:
|
||
new_ids.append(iid)
|
||
if new_ids:
|
||
print(",".join(new_ids))
|
||
else:
|
||
print("NONE")
|
||
return 0
|
||
|
||
# ---------------------------------------------------------------- HTML → 文章
|
||
|
||
class SREEntryParser(HTMLParser):
|
||
"""解析每期的 content:encoded HTML,提取所有 sreweekly-entry 文章条目。
|
||
|
||
页面结构(见实测):
|
||
<div class="sreweekly-sponsor-message">…赞助商…</div> ← 跳过
|
||
<div class="sreweekly-entry">
|
||
<div class="sreweekly-title"><a href="文章URL">标题</a></div>
|
||
<div class="sreweekly-description">
|
||
<p>简介…</p>
|
||
<blockquote><p>引用…</p></blockquote>
|
||
<p> <small>作者</small></p>
|
||
</div>
|
||
</div>
|
||
"""
|
||
|
||
def __init__(self):
|
||
super().__init__(convert_charrefs=True)
|
||
self.div_stack = [] # 当前打开中的 div class 栈
|
||
self.tag_stack = [] # 当前打开中的标签栈(用于识别 a/blockquote/small)
|
||
self.entries = [] # 提取结果: [{title, url, body, author}, ...]
|
||
self.cur = None
|
||
self.in_title = False
|
||
self.in_desc = False
|
||
self.in_small = False
|
||
self.quote_depth = 0
|
||
|
||
def _cur_div_class(self):
|
||
return self.div_stack[-1] if self.div_stack else None
|
||
|
||
def handle_starttag(self, tag, attrs):
|
||
attrs = dict(attrs)
|
||
self.tag_stack.append(tag)
|
||
cls = attrs.get("class", "")
|
||
if tag == "div":
|
||
self.div_stack.append(cls)
|
||
if "sreweekly-entry" in cls:
|
||
self.cur = {"title": "", "url": "", "body": [], "author": None}
|
||
elif "sreweekly-title" in cls and self.cur is not None:
|
||
self.in_title = True
|
||
elif "sreweekly-description" in cls and self.cur is not None:
|
||
self.in_desc = True
|
||
return
|
||
if self.cur is None:
|
||
return
|
||
if tag == "a" and self.in_title:
|
||
self.cur["url"] = attrs.get("href", "")
|
||
elif tag == "blockquote" and self.in_desc:
|
||
self.quote_depth += 1
|
||
if self.cur["body"] and self.cur["body"][-1] not in ("\n\n", "\n> ", "\n"):
|
||
self.cur["body"].append("\n\n")
|
||
self.cur["body"].append("\n> ")
|
||
elif tag == "small" and self.in_desc:
|
||
self.in_small = True
|
||
elif tag == "p" and self.in_desc and self.cur["body"] and self.quote_depth == 0:
|
||
# 段落间空行(避免在开头及 blockquote 内部加多余空行)
|
||
self.cur["body"].append("\n\n")
|
||
|
||
def handle_endtag(self, tag):
|
||
if self.tag_stack:
|
||
pop_idx = None
|
||
for i in range(len(self.tag_stack) - 1, -1, -1):
|
||
if self.tag_stack[i] == tag:
|
||
pop_idx = i
|
||
break
|
||
if pop_idx is not None:
|
||
del self.tag_stack[pop_idx:]
|
||
if tag == "small" and self.in_small:
|
||
self.in_small = False
|
||
if tag == "blockquote" and self.quote_depth > 0:
|
||
self.quote_depth -= 1
|
||
if self.cur is not None and self.in_desc:
|
||
self.cur["body"].append("\n")
|
||
if tag == "div":
|
||
if self.div_stack:
|
||
cls = self.div_stack.pop()
|
||
if "sreweekly-title" in cls:
|
||
self.in_title = False
|
||
elif "sreweekly-description" in cls:
|
||
self.in_desc = False
|
||
elif "sreweekly-entry" in cls:
|
||
self._finalize()
|
||
|
||
def handle_data(self, data):
|
||
if self.cur is None:
|
||
return
|
||
if self.in_title:
|
||
self.cur["title"] += data
|
||
elif self.in_desc:
|
||
if self.in_small:
|
||
self.cur["author"] = (self.cur["author"] or "") + data
|
||
else:
|
||
if not data.strip():
|
||
return # 标签间的纯空白(换行/缩进)跳过,段落结构由 p/blockquote 标记生成
|
||
if self.cur["body"] and self.cur["body"][-1] in ("\n> ", "\n\n", "\n"):
|
||
data = data.lstrip("\n\t \xa0")
|
||
self.cur["body"].append(data)
|
||
|
||
def _finalize(self):
|
||
if self.cur is None:
|
||
return
|
||
title = html.unescape(self.cur["title"]).strip()
|
||
body = "".join(self.cur["body"])
|
||
body = body.replace("\xa0", " ")
|
||
# 压缩多余的连续空行
|
||
body = re.sub(r"\n{3,}", "\n\n", body)
|
||
# 每行去掉首尾空白,但保留引用行前缀 ">"
|
||
lines = []
|
||
for ln in body.split("\n"):
|
||
stripped = ln.strip()
|
||
if stripped.startswith(">"):
|
||
lines.append("> " + stripped[1:].strip())
|
||
elif stripped:
|
||
lines.append(stripped)
|
||
else:
|
||
lines.append("")
|
||
body = "\n".join(lines).strip("\n")
|
||
author = html.unescape((self.cur["author"] or "").strip()) or None
|
||
self.entries.append({"title": title, "url": self.cur["url"], "body": body, "author": author})
|
||
self.cur = None
|
||
|
||
def parse_issue_html(html_path):
|
||
with open(html_path, "r", encoding="utf-8") as f:
|
||
p = SREEntryParser()
|
||
p.feed(f.read())
|
||
return p.entries
|
||
|
||
def slugify(title, max_len=70):
|
||
"""标题 → 文件名 slug:保留字母数字与 CJK,其余转连字符。"""
|
||
s = re.sub(r"[^\w\u4e00-\u9fff]+", "-", title.lower(), flags=re.UNICODE).strip("-")
|
||
s = re.sub(r"-{2,}", "-", s)
|
||
return s[:max_len].strip("-") or "untitled"
|
||
|
||
# ---------------------------------------------------------------- 正文抓取 & 转 markdown
|
||
|
||
def _looks_like_binary(url):
|
||
u = (url or "").lower().split("?", 1)[0]
|
||
return u.endswith(NON_HTML_EXTS)
|
||
|
||
def fetch_article_html(url, cache_path, force=False):
|
||
"""抓文章原始 HTML,落到 cache_path。返回 (html_text, error)。"""
|
||
if not url:
|
||
return None, "empty url"
|
||
if _looks_like_binary(url):
|
||
return None, "non-html resource"
|
||
if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0:
|
||
with open(cache_path, "r", encoding="utf-8", errors="replace") as f:
|
||
return f.read(), None
|
||
try:
|
||
safe_url = urllib.parse.quote(url, safe=":/?#[]@!$&'()*+,;=%~")
|
||
except (UnicodeError, TypeError) as e:
|
||
return None, f"url sanitize failed: {e}"
|
||
try:
|
||
req = urllib.request.Request(safe_url, headers={
|
||
"User-Agent": DEFAULT_UA,
|
||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||
"Accept-Language": "en-US,en;q=0.9",
|
||
})
|
||
with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r:
|
||
ctype = r.headers.get("Content-Type", "")
|
||
if "html" not in ctype.lower() and "xml" not in ctype.lower():
|
||
return None, f"non-html content-type: {ctype}"
|
||
raw = r.read()
|
||
charset = None
|
||
m = re.search(r"charset=([\w-]+)", ctype, re.I)
|
||
if m:
|
||
charset = m.group(1)
|
||
text = raw.decode(charset or "utf-8", errors="replace")
|
||
except urllib.error.HTTPError as e:
|
||
return None, f"HTTP {e.code}"
|
||
except urllib.error.URLError as e:
|
||
return None, f"URLError: {e.reason}"
|
||
except (TimeoutError, OSError, http.client.HTTPException, ValueError, UnicodeError) as e:
|
||
return None, f"{type(e).__name__}: {e}"
|
||
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
|
||
with open(cache_path, "w", encoding="utf-8") as f:
|
||
f.write(text)
|
||
return text, None
|
||
|
||
def html_to_markdown(html_text, url):
|
||
"""用 trafilatura 抽正文并输出 markdown。失败返回 None。"""
|
||
if not html_text or trafilatura is None:
|
||
return None
|
||
try:
|
||
md = trafilatura.extract(
|
||
html_text,
|
||
output_format="markdown",
|
||
url=url,
|
||
include_links=True,
|
||
include_images=True,
|
||
include_tables=True,
|
||
favor_recall=True,
|
||
with_metadata=False,
|
||
)
|
||
if md:
|
||
md = md.strip()
|
||
return md or None
|
||
except Exception as e:
|
||
print(f"⚠️ trafilatura 抽取失败 {url}: {e}", file=sys.stderr)
|
||
return None
|
||
|
||
def extract_one(iid, entry, base_dir, force=False, fetch=True, force_fetch=False, sleep=1.5):
|
||
"""把某一期 HTML 解析为每篇文章一个 markdown 文件。返回 (文章数, 输出目录, changed)。"""
|
||
md_root = os.path.join(base_dir, "markdown")
|
||
os.makedirs(md_root, exist_ok=True)
|
||
html_rel = entry.get("html_file")
|
||
if not html_rel:
|
||
sys.exit(f"❌ 期 #{iid} 没有 html_file 记录,请先扫描")
|
||
html_path = os.path.join(base_dir, html_rel)
|
||
if not os.path.exists(html_path):
|
||
sys.exit(f"❌ HTML 文件不存在: {html_path}")
|
||
already_done = entry.get("extracted") and entry.get("article_count", 0) > 0
|
||
if already_done and not force:
|
||
print(f"⏭️ 期 #{iid} 已提取过({entry['article_count']} 篇),跳过(--force 可覆盖)", file=sys.stderr)
|
||
return entry.get("article_count", 0), entry.get("markdown_dir"), False
|
||
|
||
articles = parse_issue_html(html_path)
|
||
out_dir = os.path.join(md_root, str(iid))
|
||
os.makedirs(out_dir, exist_ok=True)
|
||
art_cache_dir = os.path.join(base_dir, "articles", str(iid))
|
||
|
||
fetched, failed = 0, 0
|
||
fetch_log = []
|
||
for idx, a in enumerate(articles, 1):
|
||
slug = slugify(a["title"])
|
||
body_md, err = None, None
|
||
if fetch and a.get("url"):
|
||
cache_path = os.path.join(art_cache_dir, f"{idx:02d}-{slug}.html")
|
||
need_net = force_fetch or not (os.path.exists(cache_path) and os.path.getsize(cache_path) > 0)
|
||
raw, err = fetch_article_html(a["url"], cache_path, force=force_fetch)
|
||
if raw:
|
||
body_md = html_to_markdown(raw, a["url"])
|
||
if not body_md:
|
||
err = err or "trafilatura returned empty"
|
||
if raw and not err:
|
||
fetched += 1
|
||
else:
|
||
failed += 1
|
||
fetch_log.append({"idx": idx, "url": a["url"], "ok": bool(body_md), "error": err})
|
||
if need_net and sleep > 0 and idx < len(articles):
|
||
time.sleep(sleep)
|
||
fname = f"{idx:02d}-{slug}.md"
|
||
fpath = os.path.join(out_dir, fname)
|
||
content = render_markdown(a, entry, body_md=body_md, fetch_error=err if fetch else None)
|
||
with open(fpath, "w", encoding="utf-8") as f:
|
||
f.write(content)
|
||
|
||
if fetch and fetch_log:
|
||
os.makedirs(art_cache_dir, exist_ok=True)
|
||
with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f:
|
||
json.dump(fetch_log, f, ensure_ascii=False, indent=2)
|
||
|
||
entry["extracted"] = True
|
||
entry["article_count"] = len(articles)
|
||
entry["markdown_dir"] = f"markdown/{iid}"
|
||
entry["extracted_at"] = now_iso()
|
||
if fetch:
|
||
entry["articles_fetched"] = fetched
|
||
entry["articles_failed"] = failed
|
||
print(f"📄 期 #{iid} → {len(articles)} 篇文章 → {out_dir}(抓取 {fetched} 成功 / {failed} 失败)", file=sys.stderr)
|
||
return len(articles), out_dir, True
|
||
|
||
def render_markdown(article, entry, body_md=None, fetch_error=None):
|
||
"""文章 → markdown 文件内容。"""
|
||
pub = entry.get("pub_date", "")
|
||
issue_label = entry.get("title", f"SRE Weekly Issue #{entry.get('id')}")
|
||
out = [f"# {article['title']}", ""]
|
||
out.append(f"- **期号**: {issue_label}({pub})")
|
||
out.append(f"- **作者**: {article['author'] or '—'}")
|
||
out.append(f"- **链接**: {article['url']}")
|
||
if article["body"]:
|
||
out += ["", "## 简介", "", article["body"]]
|
||
if body_md:
|
||
out += ["", "## 正文", "", body_md]
|
||
elif fetch_error:
|
||
out += ["", "## 正文", "", f"> ⚠️ 抓取失败:{fetch_error}"]
|
||
return "\n".join(out) + "\n"
|
||
|
||
def resolve_targets(target, manifest):
|
||
"""解析 --extract 的目标: ID / FROM..TO / latest / pending / all → 期号列表。"""
|
||
issues = manifest["issues"]
|
||
if target == "latest":
|
||
ids = sorted(issues.keys(), key=lambda x: (isinstance(x, str), x))
|
||
# 数字优先按数值排,latest 取最大
|
||
nums = [i for i in issues if str(i).isdigit()]
|
||
return [str(max(int(n) for n in nums))] if nums else [ids[-1]]
|
||
if target == "pending":
|
||
return [str(i) for i, v in issues.items() if not v.get("extracted") or v.get("article_count", 0) == 0]
|
||
if target == "all":
|
||
return list(issues.keys())
|
||
t = str(target)
|
||
if ".." in t:
|
||
lo, hi = parse_id_range(t, "--extract")
|
||
ids, missing = [], []
|
||
for n in range(lo, hi + 1):
|
||
iid = str(n)
|
||
(ids if iid in issues else missing).append(iid)
|
||
if missing:
|
||
preview = ",".join(missing[:10]) + ("..." if len(missing) > 10 else "")
|
||
print(f"⚠️ 范围 {lo}..{hi} 中 {len(missing)} 期不在 manifest(跳过): {preview}", file=sys.stderr)
|
||
if not ids:
|
||
sys.exit(f"❌ 范围 {lo}..{hi} 中没有已登记的期,请先 --backfill 或扫描")
|
||
return ids
|
||
if t not in issues:
|
||
sys.exit(f"❌ 期 #{t} 不在 manifest 中,请先扫描")
|
||
return [t]
|
||
|
||
def cmd_extract(args, manifest, base_dir):
|
||
ids = resolve_targets(args.extract, manifest)
|
||
if not ids:
|
||
print("✅ 无待提取的期", file=sys.stderr)
|
||
return
|
||
fetch = not args.skip_fetch
|
||
if fetch and trafilatura is None:
|
||
sys.exit("❌ 需要抓取正文但缺少依赖:pip3 install --user --break-system-packages trafilatura(或加 --skip-fetch)")
|
||
for iid in ids:
|
||
extract_one(
|
||
iid, manifest["issues"][str(iid)], base_dir,
|
||
force=args.force, fetch=fetch,
|
||
force_fetch=args.force_fetch, sleep=args.sleep,
|
||
)
|
||
save_manifest(base_dir, manifest)
|
||
|
||
# ---------------------------------------------------------------- 展示
|
||
|
||
def cmd_status(manifest):
|
||
issues = manifest["issues"]
|
||
print(f"📡 SRE Weekly 共 {len(issues)} 期(最后扫描: {manifest['last_scanned']})")
|
||
print()
|
||
header = f"{'期号':>6} {'日期':<12} {'HTML':<6} {'提取':<5} {'文章数':<6} 标题"
|
||
print(header)
|
||
print("-" * len(header))
|
||
for iid in sorted(issues.keys(), key=lambda x: int(x) if str(x).isdigit() else 0):
|
||
v = issues[iid]
|
||
html_ok = "✅" if v.get("html_file") else "—"
|
||
ext_ok = "✅" if v.get("extracted") and v.get("article_count", 0) > 0 else "—"
|
||
print(f"{iid:>6} {v.get('pub_date',''):<12} {html_ok:<6} {ext_ok:<5} {v.get('article_count',0):<6} {v.get('title','')}")
|
||
|
||
def cmd_issues(manifest):
|
||
for iid in sorted(manifest["issues"].keys(), key=lambda x: int(x) if str(x).isdigit() else 0):
|
||
v = manifest["issues"][iid]
|
||
print(f"{iid}\t{v.get('pub_date','')}\t{v.get('title','')}\t{v.get('url','')}")
|
||
|
||
# ---------------------------------------------------------------- main
|
||
|
||
def main():
|
||
ap = argparse.ArgumentParser(description="SRE Weekly 抓取 + 文章提取(供 n8n 定时调用)")
|
||
ap.add_argument("--feed", default=DEFAULT_FEED, help=f"RSS 地址(默认 {DEFAULT_FEED})")
|
||
ap.add_argument("--dir", default=DEFAULT_DIR, help=f"工作目录(默认 {DEFAULT_DIR})")
|
||
ap.add_argument("--check", action="store_true", help="只检查有无新期,输出 id,id 或 NONE")
|
||
ap.add_argument("--extract", metavar="ID|FROM..TO|latest|pending|all", help="提取文章 → markdown(支持区间如 433..464)")
|
||
ap.add_argument("--backfill", metavar="FROM..TO", help="回溯抓取指定期号区间(绕过 RSS 限制,例:--backfill 400..500)")
|
||
ap.add_argument("--status", action="store_true", help="打印 manifest 状态表")
|
||
ap.add_argument("--issues", action="store_true", help="列出已登记的所有期")
|
||
ap.add_argument("--force", action="store_true", help="强制重新提取已提取过的期")
|
||
ap.add_argument("--skip-fetch", action="store_true", help="不抓文章正文(仅使用 feed 简介)")
|
||
ap.add_argument("--force-fetch", action="store_true", help="忽略本地缓存重新抓文章 HTML")
|
||
ap.add_argument("--sleep", type=float, default=1.5, help="每篇文章抓取之间的间隔秒数(默认 1.5)")
|
||
args = ap.parse_args()
|
||
|
||
base_dir = os.path.abspath(os.path.expanduser(args.dir))
|
||
os.makedirs(base_dir, exist_ok=True)
|
||
|
||
if args.extract or args.status or args.issues or args.backfill:
|
||
manifest = load_manifest(base_dir)
|
||
if args.backfill:
|
||
cmd_backfill(args, manifest, base_dir)
|
||
if args.extract:
|
||
cmd_extract(args, manifest, base_dir)
|
||
if args.status:
|
||
cmd_status(manifest)
|
||
if args.issues:
|
||
cmd_issues(manifest)
|
||
elif args.check:
|
||
manifest = load_manifest(base_dir)
|
||
cmd_check(args, manifest)
|
||
else:
|
||
manifest = load_manifest(base_dir)
|
||
cmd_scan(args, manifest, base_dir)
|
||
|
||
return 0
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main()) |