Files
atlas/sreweekly-digest/scripts/sreweekly.py
2026-09-12 14:17:06 +08:00

628 lines
27 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
SRE Weekly 抓取与文章提取脚本
模仿 blogwatcher-daily 的单文件风格:feedparser 抓取 → HTML 落盘 → 逐篇文章导出 Markdown,
并用 manifest.json 记录每期的状态(是否已存 HTML / 是否已提取文章)。
设计给 n8n 定时任务调用,通过不同参数驱动,脚本自身不发送邮件。
目录结构(默认工作目录 ~/Workspace/nexus/sreweekly/):
sreweekly/
├── manifest.json # 状态记录(每期: html 是否保存、文章是否提取、抓取统计)
├── html/<id>-<日期>.html # 每期原始 HTML(feed 的 content:encoded 原样保存)
├── articles/<id>/ # 每篇文章原始 HTML 缓存 + index.json 抓取日志
│ ├── 01-<slug>.html
│ └── index.json
└── markdown/<id>/ # 每期文章的 markdown 目录(每篇一个文件,含正文)
├── 01-<slug>.md
└── 02-<slug>.md
用法:
python3 sreweekly.py # 扫描 feed,新增期保存为 HTML 并登记 manifest
python3 sreweekly.py --check # 只检查有无新期,输出 NEW:533,532 或 NONE(供 n8n 分支)
python3 sreweekly.py --extract ID # 提取指定期(如 533)的文章 → markdown(默认抓正文)
python3 sreweekly.py --extract 433..464 # 提取期号区间(只处理 manifest 中已登记的期)
python3 sreweekly.py --extract latest # 提取最新一期
python3 sreweekly.py --extract pending # 提取所有尚未提取的期(n8n 每日流程推荐)
python3 sreweekly.py --extract all # 提取全部(含已提取的,需配合 --force 覆盖)
python3 sreweekly.py --status # 打印 manifest 状态表
python3 sreweekly.py --issues # 列出已登记的所有期
python3 sreweekly.py --backfill 400..500 # 回溯抓取指定期号区间(绕过 RSS 只有最近 ~10 期的限制)
选项:
--feed URL RSS 地址(默认 https://sreweekly.com/feed/)
--dir PATH 工作目录(默认 ~/Workspace/nexus/sreweekly/)
--force 重新提取已提取过的期(覆盖旧 markdown)
--skip-fetch 不抓文章正文,只用 feed 简介(老行为)
--force-fetch 忽略 articles/ 本地缓存,重新联网抓
--sleep 秒数 每篇文章抓取之间的间隔(默认 1.5,避免打站过快)
依赖:
pip3 install feedparser trafilatura
"""
from html.parser import HTMLParser
import argparse, html, http.client, json, os, re, sys, time, urllib.error, urllib.parse, urllib.request
try:
import feedparser
except ImportError:
sys.exit("缺少依赖: 请先执行 pip3 install feedparser")
try:
import trafilatura
except ImportError:
trafilatura = None # 抓正文时才强制要求
DEFAULT_FEED = "https://sreweekly.com/feed/"
DEFAULT_DIR = os.path.expanduser("~/Workspace/nexus/sreweekly")
DEFAULT_UA = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0 Safari/537.36"
)
FETCH_TIMEOUT = 20
NON_HTML_EXTS = (".pdf", ".zip", ".tar", ".gz", ".mp4", ".mp3", ".png", ".jpg", ".jpeg", ".gif", ".svg")
# ---------------------------------------------------------------- manifest
def load_manifest(base_dir):
path = os.path.join(base_dir, "manifest.json")
if os.path.exists(path):
try:
with open(path, "r", encoding="utf-8") as f:
return json.load(f)
except (json.JSONDecodeError, OSError) as e:
print(f"⚠️ manifest.json 读取失败({e}),按空库处理", file=sys.stderr)
return {"feed": DEFAULT_FEED, "last_scanned": None, "issues": {}}
def save_manifest(base_dir, manifest):
path = os.path.join(base_dir, "manifest.json")
with open(path, "w", encoding="utf-8") as f:
json.dump(manifest, f, ensure_ascii=False, indent=2)
return path
def issue_id_from_link(link):
"""从文章链接提取期号: sre-weekly-issue-533 → 533;失败则用最后一段路径。"""
m = re.search(r"sre-weekly-issue-(\d+)", link or "")
if m:
return int(m.group(1))
m = re.search(r"/([^/]+?)/?$", (link or "").rstrip("/"))
return m.group(1) if m else "unknown"
def format_pubdate(entry):
"""优先用 feed 自带发布日期(YYYY-MM-DD),没有则用当天。"""
if getattr(entry, "published_parsed", None):
return time.strftime("%Y-%m-%d", entry.published_parsed)
return time.strftime("%Y-%m-%d")
def now_iso():
return time.strftime("%Y-%m-%dT%H:%M:%S")
# ---------------------------------------------------------------- feed 扫描
def fetch_feed(feed_url):
d = feedparser.parse(feed_url)
if not d.entries:
sys.exit(f"❌ feed 抓取失败或为空: {feed_url}(bozo={d.bozo})")
return d.entries
def cmd_scan(args, manifest, base_dir):
html_dir = os.path.join(base_dir, "html")
os.makedirs(html_dir, exist_ok=True)
entries = fetch_feed(args.feed)
new_ids = []
for e in entries:
iid = str(issue_id_from_link(e.link))
if iid in manifest["issues"]:
continue # 已登记,跳过(去重)
content_html = e.content[0].value if getattr(e, "content", None) else ""
pubdate = format_pubdate(e)
fname = f"{iid}-{pubdate}.html"
fpath = os.path.join(html_dir, fname)
with open(fpath, "w", encoding="utf-8") as f:
f.write(content_html)
manifest["issues"][iid] = {
"id": iid,
"title": e.title,
"url": e.link,
"pub_date": pubdate,
"html_file": f"html/{fname}",
"fetched_at": now_iso(),
"extracted": False,
"article_count": 0,
"markdown_dir": None,
}
new_ids.append(iid)
print(f"🆕 新期 #{iid} {e.title} → {fname}({len(content_html)} bytes)", file=sys.stderr)
manifest["last_scanned"] = now_iso()
save_manifest(base_dir, manifest)
print(f"📊 扫描完成:新增 {len(new_ids)} 期,共 {len(manifest['issues'])} 期", file=sys.stderr)
return new_ids
# ---------------------------------------------------------------- 回溯抓取(老期刊)
MONTHS = {m: i for i, m in enumerate(
["January","February","March","April","May","June",
"July","August","September","October","November","December"], 1)}
def parse_id_range(spec, flag_name="--backfill"):
"""'400..500' → (400, 500);'450' → (450, 450)。"""
m = re.match(r"^\s*(\d+)\s*(?:\.\.\s*(\d+))?\s*$", spec or "")
if not m:
sys.exit(f"❌ {flag_name} 参数格式错误(示例 400..500 或 450): {spec}")
lo = int(m.group(1)); hi = int(m.group(2) or m.group(1))
if lo > hi: lo, hi = hi, lo
return lo, hi
def extract_pub_date_from_page(html_text):
"""从 sreweekly 站点页面 HTML 里提取发布日期 → YYYY-MM-DD,失败返回 None。
结构: <a class="updated" href="...">June 23, 2024</a>
"""
if not html_text:
return None
m = re.search(r'class="updated"[^>]*>\s*([A-Za-z]+)\s+(\d{1,2}),\s*(\d{4})', html_text)
if not m:
return None
mon = MONTHS.get(m.group(1))
if not mon:
return None
return f"{int(m.group(3)):04d}-{mon:02d}-{int(m.group(2)):02d}"
def extract_title_from_page(html_text, iid):
m = re.search(r"<title>\s*(.+?)\s*</title>", html_text or "", re.I | re.S)
if not m:
return f"SRE Weekly Issue #{iid}"
t = html.unescape(m.group(1))
# 去掉站点后缀 " – SRE WEEKLY" / " - SRE WEEKLY"
t = re.sub(r"\s*[\u2013\u2014\-]\s*SRE WEEKLY\s*$", "", t, flags=re.I).strip()
return t or f"SRE Weekly Issue #{iid}"
def cmd_backfill(args, manifest, base_dir):
lo, hi = parse_id_range(args.backfill, "--backfill")
html_dir = os.path.join(base_dir, "html")
page_cache_dir = os.path.join(base_dir, "pages") # 存整页 HTML 作缓存
os.makedirs(html_dir, exist_ok=True)
added, skipped, missing = [], [], []
for n in range(lo, hi + 1):
iid = str(n)
if iid in manifest["issues"]:
skipped.append(iid)
continue
url = f"https://sreweekly.com/sre-weekly-issue-{n}/"
cache_path = os.path.join(page_cache_dir, f"{n}.html")
raw, err = fetch_article_html(url, cache_path, force=args.force_fetch)
if not raw:
print(f"⚠️ #{iid} 抓取失败:{err}({url})", file=sys.stderr)
missing.append(iid)
if args.sleep > 0:
time.sleep(args.sleep)
continue
# 站点页 HTML 含 sreweekly-entry 结构,SREEntryParser 兼容;直接落到 html/ 供后续 --extract 使用
pubdate = extract_pub_date_from_page(raw) or time.strftime("%Y-%m-%d")
title = extract_title_from_page(raw, iid)
fname = f"{iid}-{pubdate}.html"
with open(os.path.join(html_dir, fname), "w", encoding="utf-8") as f:
f.write(raw)
manifest["issues"][iid] = {
"id": iid,
"title": title,
"url": url,
"pub_date": pubdate,
"html_file": f"html/{fname}",
"fetched_at": now_iso(),
"source": "backfill",
"extracted": False,
"article_count": 0,
"markdown_dir": None,
}
added.append(iid)
print(f"🆕 回溯 #{iid} {title} → {fname}({len(raw)} bytes)", file=sys.stderr)
if args.sleep > 0 and n < hi:
time.sleep(args.sleep)
save_manifest(base_dir, manifest)
print(f"📊 回溯完成:新增 {len(added)} 期 / 跳过已有 {len(skipped)} / 抓取失败 {len(missing)}", file=sys.stderr)
if missing:
print(f" 失败期号: {','.join(missing)}", file=sys.stderr)
return added
def cmd_check(args, manifest):
entries = fetch_feed(args.feed)
known = set(manifest["issues"].keys())
new_ids = []
for e in entries:
iid = str(issue_id_from_link(e.link))
if iid not in known:
new_ids.append(iid)
if new_ids:
print("NEW:" + ",".join(new_ids))
else:
print("NONE")
return 0
# ---------------------------------------------------------------- HTML → 文章
class SREEntryParser(HTMLParser):
"""解析每期的 content:encoded HTML,提取所有 sreweekly-entry 文章条目。
页面结构(见实测):
<div class="sreweekly-sponsor-message">…赞助商…</div> ← 跳过
<div class="sreweekly-entry">
<div class="sreweekly-title"><a href="文章URL">标题</a></div>
<div class="sreweekly-description">
<p>简介…</p>
<blockquote><p>引用…</p></blockquote>
<p>&nbsp;&nbsp;<small>作者</small></p>
</div>
</div>
"""
def __init__(self):
super().__init__(convert_charrefs=True)
self.div_stack = [] # 当前打开中的 div class 栈
self.tag_stack = [] # 当前打开中的标签栈(用于识别 a/blockquote/small)
self.entries = [] # 提取结果: [{title, url, body, author}, ...]
self.cur = None
self.in_title = False
self.in_desc = False
self.in_small = False
self.quote_depth = 0
def _cur_div_class(self):
return self.div_stack[-1] if self.div_stack else None
def handle_starttag(self, tag, attrs):
attrs = dict(attrs)
self.tag_stack.append(tag)
cls = attrs.get("class", "")
if tag == "div":
self.div_stack.append(cls)
if "sreweekly-entry" in cls:
self.cur = {"title": "", "url": "", "body": [], "author": None}
elif "sreweekly-title" in cls and self.cur is not None:
self.in_title = True
elif "sreweekly-description" in cls and self.cur is not None:
self.in_desc = True
return
if self.cur is None:
return
if tag == "a" and self.in_title:
self.cur["url"] = attrs.get("href", "")
elif tag == "blockquote" and self.in_desc:
self.quote_depth += 1
if self.cur["body"] and self.cur["body"][-1] not in ("\n\n", "\n> ", "\n"):
self.cur["body"].append("\n\n")
self.cur["body"].append("\n> ")
elif tag == "small" and self.in_desc:
self.in_small = True
elif tag == "p" and self.in_desc and self.cur["body"] and self.quote_depth == 0:
# 段落间空行(避免在开头及 blockquote 内部加多余空行)
self.cur["body"].append("\n\n")
def handle_endtag(self, tag):
if self.tag_stack:
pop_idx = None
for i in range(len(self.tag_stack) - 1, -1, -1):
if self.tag_stack[i] == tag:
pop_idx = i
break
if pop_idx is not None:
del self.tag_stack[pop_idx:]
if tag == "small" and self.in_small:
self.in_small = False
if tag == "blockquote" and self.quote_depth > 0:
self.quote_depth -= 1
if self.cur is not None and self.in_desc:
self.cur["body"].append("\n")
if tag == "div":
if self.div_stack:
cls = self.div_stack.pop()
if "sreweekly-title" in cls:
self.in_title = False
elif "sreweekly-description" in cls:
self.in_desc = False
elif "sreweekly-entry" in cls:
self._finalize()
def handle_data(self, data):
if self.cur is None:
return
if self.in_title:
self.cur["title"] += data
elif self.in_desc:
if self.in_small:
self.cur["author"] = (self.cur["author"] or "") + data
else:
if not data.strip():
return # 标签间的纯空白(换行/缩进)跳过,段落结构由 p/blockquote 标记生成
if self.cur["body"] and self.cur["body"][-1] in ("\n> ", "\n\n", "\n"):
data = data.lstrip("\n\t \xa0")
self.cur["body"].append(data)
def _finalize(self):
if self.cur is None:
return
title = html.unescape(self.cur["title"]).strip()
body = "".join(self.cur["body"])
body = body.replace("\xa0", " ")
# 压缩多余的连续空行
body = re.sub(r"\n{3,}", "\n\n", body)
# 每行去掉首尾空白,但保留引用行前缀 ">"
lines = []
for ln in body.split("\n"):
stripped = ln.strip()
if stripped.startswith(">"):
lines.append("> " + stripped[1:].strip())
elif stripped:
lines.append(stripped)
else:
lines.append("")
body = "\n".join(lines).strip("\n")
author = html.unescape((self.cur["author"] or "").strip()) or None
self.entries.append({"title": title, "url": self.cur["url"], "body": body, "author": author})
self.cur = None
def parse_issue_html(html_path):
with open(html_path, "r", encoding="utf-8") as f:
p = SREEntryParser()
p.feed(f.read())
return p.entries
def slugify(title, max_len=70):
"""标题 → 文件名 slug:保留字母数字与 CJK,其余转连字符。"""
s = re.sub(r"[^\w\u4e00-\u9fff]+", "-", title.lower(), flags=re.UNICODE).strip("-")
s = re.sub(r"-{2,}", "-", s)
return s[:max_len].strip("-") or "untitled"
# ---------------------------------------------------------------- 正文抓取 & 转 markdown
def _looks_like_binary(url):
u = (url or "").lower().split("?", 1)[0]
return u.endswith(NON_HTML_EXTS)
def fetch_article_html(url, cache_path, force=False):
"""抓文章原始 HTML,落到 cache_path。返回 (html_text, error)。"""
if not url:
return None, "empty url"
if _looks_like_binary(url):
return None, "non-html resource"
if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0:
with open(cache_path, "r", encoding="utf-8", errors="replace") as f:
return f.read(), None
try:
safe_url = urllib.parse.quote(url, safe=":/?#[]@!$&'()*+,;=%~")
except (UnicodeError, TypeError) as e:
return None, f"url sanitize failed: {e}"
try:
req = urllib.request.Request(safe_url, headers={
"User-Agent": DEFAULT_UA,
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.9",
})
with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r:
ctype = r.headers.get("Content-Type", "")
if "html" not in ctype.lower() and "xml" not in ctype.lower():
return None, f"non-html content-type: {ctype}"
raw = r.read()
charset = None
m = re.search(r"charset=([\w-]+)", ctype, re.I)
if m:
charset = m.group(1)
text = raw.decode(charset or "utf-8", errors="replace")
except urllib.error.HTTPError as e:
return None, f"HTTP {e.code}"
except urllib.error.URLError as e:
return None, f"URLError: {e.reason}"
except (TimeoutError, OSError, http.client.HTTPException, ValueError, UnicodeError) as e:
return None, f"{type(e).__name__}: {e}"
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
with open(cache_path, "w", encoding="utf-8") as f:
f.write(text)
return text, None
def html_to_markdown(html_text, url):
"""用 trafilatura 抽正文并输出 markdown。失败返回 None。"""
if not html_text or trafilatura is None:
return None
try:
md = trafilatura.extract(
html_text,
output_format="markdown",
url=url,
include_links=True,
include_images=True,
include_tables=True,
favor_recall=True,
with_metadata=False,
)
if md:
md = md.strip()
return md or None
except Exception as e:
print(f"⚠️ trafilatura 抽取失败 {url}: {e}", file=sys.stderr)
return None
def extract_one(iid, entry, base_dir, force=False, fetch=True, force_fetch=False, sleep=1.5):
"""把某一期 HTML 解析为每篇文章一个 markdown 文件。返回 (文章数, 输出目录, changed)。"""
md_root = os.path.join(base_dir, "markdown")
os.makedirs(md_root, exist_ok=True)
html_rel = entry.get("html_file")
if not html_rel:
sys.exit(f"❌ 期 #{iid} 没有 html_file 记录,请先扫描")
html_path = os.path.join(base_dir, html_rel)
if not os.path.exists(html_path):
sys.exit(f"❌ HTML 文件不存在: {html_path}")
already_done = entry.get("extracted") and entry.get("article_count", 0) > 0
if already_done and not force:
print(f"⏭️ 期 #{iid} 已提取过({entry['article_count']} 篇),跳过(--force 可覆盖)", file=sys.stderr)
return entry.get("article_count", 0), entry.get("markdown_dir"), False
articles = parse_issue_html(html_path)
out_dir = os.path.join(md_root, str(iid))
os.makedirs(out_dir, exist_ok=True)
art_cache_dir = os.path.join(base_dir, "articles", str(iid))
fetched, failed = 0, 0
fetch_log = []
for idx, a in enumerate(articles, 1):
slug = slugify(a["title"])
body_md, err = None, None
if fetch and a.get("url"):
cache_path = os.path.join(art_cache_dir, f"{idx:02d}-{slug}.html")
need_net = force_fetch or not (os.path.exists(cache_path) and os.path.getsize(cache_path) > 0)
raw, err = fetch_article_html(a["url"], cache_path, force=force_fetch)
if raw:
body_md = html_to_markdown(raw, a["url"])
if not body_md:
err = err or "trafilatura returned empty"
if raw and not err:
fetched += 1
else:
failed += 1
fetch_log.append({"idx": idx, "url": a["url"], "ok": bool(body_md), "error": err})
if need_net and sleep > 0 and idx < len(articles):
time.sleep(sleep)
fname = f"{idx:02d}-{slug}.md"
fpath = os.path.join(out_dir, fname)
content = render_markdown(a, entry, body_md=body_md, fetch_error=err if fetch else None)
with open(fpath, "w", encoding="utf-8") as f:
f.write(content)
if fetch and fetch_log:
os.makedirs(art_cache_dir, exist_ok=True)
with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f:
json.dump(fetch_log, f, ensure_ascii=False, indent=2)
entry["extracted"] = True
entry["article_count"] = len(articles)
entry["markdown_dir"] = f"markdown/{iid}"
entry["extracted_at"] = now_iso()
if fetch:
entry["articles_fetched"] = fetched
entry["articles_failed"] = failed
print(f"📄 期 #{iid} → {len(articles)} 篇文章 → {out_dir}(抓取 {fetched} 成功 / {failed} 失败)", file=sys.stderr)
return len(articles), out_dir, True
def render_markdown(article, entry, body_md=None, fetch_error=None):
"""文章 → markdown 文件内容。"""
pub = entry.get("pub_date", "")
issue_label = entry.get("title", f"SRE Weekly Issue #{entry.get('id')}")
out = [f"# {article['title']}", ""]
out.append(f"- **期号**: {issue_label}({pub})")
out.append(f"- **作者**: {article['author'] or '—'}")
out.append(f"- **链接**: {article['url']}")
if article["body"]:
out += ["", "## 简介", "", article["body"]]
if body_md:
out += ["", "## 正文", "", body_md]
elif fetch_error:
out += ["", "## 正文", "", f"> ⚠️ 抓取失败:{fetch_error}"]
return "\n".join(out) + "\n"
def resolve_targets(target, manifest):
"""解析 --extract 的目标: ID / FROM..TO / latest / pending / all → 期号列表。"""
issues = manifest["issues"]
if target == "latest":
ids = sorted(issues.keys(), key=lambda x: (isinstance(x, str), x))
# 数字优先按数值排,latest 取最大
nums = [i for i in issues if str(i).isdigit()]
return [str(max(int(n) for n in nums))] if nums else [ids[-1]]
if target == "pending":
return [str(i) for i, v in issues.items() if not v.get("extracted") or v.get("article_count", 0) == 0]
if target == "all":
return list(issues.keys())
t = str(target)
if ".." in t:
lo, hi = parse_id_range(t, "--extract")
ids, missing = [], []
for n in range(lo, hi + 1):
iid = str(n)
(ids if iid in issues else missing).append(iid)
if missing:
preview = ",".join(missing[:10]) + ("..." if len(missing) > 10 else "")
print(f"⚠️ 范围 {lo}..{hi} 中 {len(missing)} 期不在 manifest(跳过): {preview}", file=sys.stderr)
if not ids:
sys.exit(f"❌ 范围 {lo}..{hi} 中没有已登记的期,请先 --backfill 或扫描")
return ids
if t not in issues:
sys.exit(f"❌ 期 #{t} 不在 manifest 中,请先扫描")
return [t]
def cmd_extract(args, manifest, base_dir):
ids = resolve_targets(args.extract, manifest)
if not ids:
print("✅ 无待提取的期", file=sys.stderr)
return
fetch = not args.skip_fetch
if fetch and trafilatura is None:
sys.exit("❌ 需要抓取正文但缺少依赖:pip3 install --user --break-system-packages trafilatura(或加 --skip-fetch)")
for iid in ids:
extract_one(
iid, manifest["issues"][str(iid)], base_dir,
force=args.force, fetch=fetch,
force_fetch=args.force_fetch, sleep=args.sleep,
)
save_manifest(base_dir, manifest)
# ---------------------------------------------------------------- 展示
def cmd_status(manifest):
issues = manifest["issues"]
print(f"📡 SRE Weekly 共 {len(issues)} 期(最后扫描: {manifest['last_scanned']})")
print()
header = f"{'期号':>6} {'日期':<12} {'HTML':<6} {'提取':<5} {'文章数':<6} 标题"
print(header)
print("-" * len(header))
for iid in sorted(issues.keys(), key=lambda x: int(x) if str(x).isdigit() else 0):
v = issues[iid]
html_ok = "✅" if v.get("html_file") else "—"
ext_ok = "✅" if v.get("extracted") and v.get("article_count", 0) > 0 else "—"
print(f"{iid:>6} {v.get('pub_date',''):<12} {html_ok:<6} {ext_ok:<5} {v.get('article_count',0):<6} {v.get('title','')}")
def cmd_issues(manifest):
for iid in sorted(manifest["issues"].keys(), key=lambda x: int(x) if str(x).isdigit() else 0):
v = manifest["issues"][iid]
print(f"{iid}\t{v.get('pub_date','')}\t{v.get('title','')}\t{v.get('url','')}")
# ---------------------------------------------------------------- main
def main():
ap = argparse.ArgumentParser(description="SRE Weekly 抓取 + 文章提取(供 n8n 定时调用)")
ap.add_argument("--feed", default=DEFAULT_FEED, help=f"RSS 地址(默认 {DEFAULT_FEED})")
ap.add_argument("--dir", default=DEFAULT_DIR, help=f"工作目录(默认 {DEFAULT_DIR})")
ap.add_argument("--check", action="store_true", help="只检查有无新期,输出 NEW:id,id 或 NONE")
ap.add_argument("--extract", metavar="ID|FROM..TO|latest|pending|all", help="提取文章 → markdown(支持区间如 433..464)")
ap.add_argument("--backfill", metavar="FROM..TO", help="回溯抓取指定期号区间(绕过 RSS 限制,例:--backfill 400..500)")
ap.add_argument("--status", action="store_true", help="打印 manifest 状态表")
ap.add_argument("--issues", action="store_true", help="列出已登记的所有期")
ap.add_argument("--force", action="store_true", help="强制重新提取已提取过的期")
ap.add_argument("--skip-fetch", action="store_true", help="不抓文章正文(仅使用 feed 简介)")
ap.add_argument("--force-fetch", action="store_true", help="忽略本地缓存重新抓文章 HTML")
ap.add_argument("--sleep", type=float, default=1.5, help="每篇文章抓取之间的间隔秒数(默认 1.5)")
args = ap.parse_args()
base_dir = os.path.abspath(os.path.expanduser(args.dir))
os.makedirs(base_dir, exist_ok=True)
if args.extract or args.status or args.issues or args.backfill:
manifest = load_manifest(base_dir)
if args.backfill:
cmd_backfill(args, manifest, base_dir)
if args.extract:
cmd_extract(args, manifest, base_dir)
if args.status:
cmd_status(manifest)
if args.issues:
cmd_issues(manifest)
elif args.check:
manifest = load_manifest(base_dir)
cmd_check(args, manifest)
else:
manifest = load_manifest(base_dir)
cmd_scan(args, manifest, base_dir)
return 0
if __name__ == "__main__":
sys.exit(main())