Files
atlas/sreweekly-digest/scripts/sreweekly.py

391 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
SRE Weekly 抓取与文章提取脚本
模仿 blogwatcher-daily 的单文件风格:feedparser 抓取 → HTML 落盘 → 逐篇文章导出 Markdown,
并用 manifest.json 记录每期的状态(是否已存 HTML / 是否已提取文章)。
设计给 n8n 定时任务调用,通过不同参数驱动,脚本自身不发送邮件。
目录结构(默认工作目录 ~/Workspace/nexus/sreweekly/):
sreweekly/
├── manifest.json # 状态记录(每期: html 是否保存、文章是否提取、文章数)
├── html/<id>-<日期>.html # 每期原始 HTML(feed 的 content:encoded 原样保存)
└── markdown/<id>/ # 每期文章的 markdown 目录(每篇一个文件)
├── 01-<标题slug>.md
└── 02-<标题slug>.md
用法:
python3 sreweekly.py # 扫描 feed,新增期保存为 HTML 并登记 manifest
python3 sreweekly.py --check # 只检查有无新期,输出 NEW:533,532 或 NONE(供 n8n 分支)
python3 sreweekly.py --extract ID # 提取指定期(如 533)的文章 → markdown
python3 sreweekly.py --extract latest # 提取最新一期
python3 sreweekly.py --extract pending # 提取所有尚未提取的期(n8n 每日流程推荐)
python3 sreweekly.py --extract all # 提取全部(含已提取的,需配合 --force 覆盖)
python3 sreweekly.py --status # 打印 manifest 状态表
python3 sreweekly.py --issues # 列出已登记的所有期
选项:
--feed URL RSS 地址(默认 https://sreweekly.com/feed/)
--dir PATH 工作目录(默认 ~/Workspace/nexus/sreweekly/)
--force 重新提取已提取过的期(覆盖旧文件)
"""
from html.parser import HTMLParser
import argparse, html, json, os, re, sys, time, urllib.request
try:
import feedparser
except ImportError:
sys.exit("缺少依赖: 请先执行 pip3 install feedparser")
DEFAULT_FEED = "https://sreweekly.com/feed/"
DEFAULT_DIR = os.path.expanduser("~/Workspace/nexus/sreweekly")
# ---------------------------------------------------------------- manifest
def load_manifest(base_dir):
path = os.path.join(base_dir, "manifest.json")
if os.path.exists(path):
try:
with open(path, "r", encoding="utf-8") as f:
return json.load(f)
except (json.JSONDecodeError, OSError) as e:
print(f"⚠️ manifest.json 读取失败({e}),按空库处理", file=sys.stderr)
return {"feed": DEFAULT_FEED, "last_scanned": None, "issues": {}}
def save_manifest(base_dir, manifest):
path = os.path.join(base_dir, "manifest.json")
with open(path, "w", encoding="utf-8") as f:
json.dump(manifest, f, ensure_ascii=False, indent=2)
return path
def issue_id_from_link(link):
"""从文章链接提取期号: sre-weekly-issue-533 → 533;失败则用最后一段路径。"""
m = re.search(r"sre-weekly-issue-(\d+)", link or "")
if m:
return int(m.group(1))
m = re.search(r"/([^/]+?)/?$", (link or "").rstrip("/"))
return m.group(1) if m else "unknown"
def format_pubdate(entry):
"""优先用 feed 自带发布日期(YYYY-MM-DD),没有则用当天。"""
if getattr(entry, "published_parsed", None):
return time.strftime("%Y-%m-%d", entry.published_parsed)
return time.strftime("%Y-%m-%d")
def now_iso():
return time.strftime("%Y-%m-%dT%H:%M:%S")
# ---------------------------------------------------------------- feed 扫描
def fetch_feed(feed_url):
d = feedparser.parse(feed_url)
if not d.entries:
sys.exit(f"❌ feed 抓取失败或为空: {feed_url}(bozo={d.bozo})")
return d.entries
def cmd_scan(args, manifest, base_dir):
html_dir = os.path.join(base_dir, "html")
os.makedirs(html_dir, exist_ok=True)
entries = fetch_feed(args.feed)
new_ids = []
for e in entries:
iid = str(issue_id_from_link(e.link))
if iid in manifest["issues"]:
continue # 已登记,跳过(去重)
content_html = e.content[0].value if getattr(e, "content", None) else ""
pubdate = format_pubdate(e)
fname = f"{iid}-{pubdate}.html"
fpath = os.path.join(html_dir, fname)
with open(fpath, "w", encoding="utf-8") as f:
f.write(content_html)
manifest["issues"][iid] = {
"id": iid,
"title": e.title,
"url": e.link,
"pub_date": pubdate,
"html_file": f"html/{fname}",
"fetched_at": now_iso(),
"extracted": False,
"article_count": 0,
"markdown_dir": None,
}
new_ids.append(iid)
print(f"🆕 新期 #{iid} {e.title} → {fname}({len(content_html)} bytes)", file=sys.stderr)
manifest["last_scanned"] = now_iso()
save_manifest(base_dir, manifest)
print(f"📊 扫描完成:新增 {len(new_ids)} 期,共 {len(manifest['issues'])} 期", file=sys.stderr)
return new_ids
def cmd_check(args, manifest):
entries = fetch_feed(args.feed)
known = set(manifest["issues"].keys())
new_ids = []
for e in entries:
iid = str(issue_id_from_link(e.link))
if iid not in known:
new_ids.append(iid)
if new_ids:
print("NEW:" + ",".join(new_ids))
else:
print("NONE")
return 0
# ---------------------------------------------------------------- HTML → 文章
class SREEntryParser(HTMLParser):
"""解析每期的 content:encoded HTML,提取所有 sreweekly-entry 文章条目。
页面结构(见实测):
<div class="sreweekly-sponsor-message">…赞助商…</div> ← 跳过
<div class="sreweekly-entry">
<div class="sreweekly-title"><a href="文章URL">标题</a></div>
<div class="sreweekly-description">
<p>简介…</p>
<blockquote><p>引用…</p></blockquote>
<p>&nbsp;&nbsp;<small>作者</small></p>
</div>
</div>
"""
def __init__(self):
super().__init__(convert_charrefs=True)
self.div_stack = [] # 当前打开中的 div class 栈
self.tag_stack = [] # 当前打开中的标签栈(用于识别 a/blockquote/small)
self.entries = [] # 提取结果: [{title, url, body, author}, ...]
self.cur = None
self.in_title = False
self.in_desc = False
self.in_small = False
self.quote_depth = 0
def _cur_div_class(self):
return self.div_stack[-1] if self.div_stack else None
def handle_starttag(self, tag, attrs):
attrs = dict(attrs)
self.tag_stack.append(tag)
cls = attrs.get("class", "")
if tag == "div":
self.div_stack.append(cls)
if "sreweekly-entry" in cls:
self.cur = {"title": "", "url": "", "body": [], "author": None}
elif "sreweekly-title" in cls and self.cur is not None:
self.in_title = True
elif "sreweekly-description" in cls and self.cur is not None:
self.in_desc = True
return
if self.cur is None:
return
if tag == "a" and self.in_title:
self.cur["url"] = attrs.get("href", "")
elif tag == "blockquote" and self.in_desc:
self.quote_depth += 1
if self.cur["body"] and self.cur["body"][-1] not in ("\n\n", "\n> ", "\n"):
self.cur["body"].append("\n\n")
self.cur["body"].append("\n> ")
elif tag == "small" and self.in_desc:
self.in_small = True
elif tag == "p" and self.in_desc and self.cur["body"] and self.quote_depth == 0:
# 段落间空行(避免在开头及 blockquote 内部加多余空行)
self.cur["body"].append("\n\n")
def handle_endtag(self, tag):
if self.tag_stack:
pop_idx = None
for i in range(len(self.tag_stack) - 1, -1, -1):
if self.tag_stack[i] == tag:
pop_idx = i
break
if pop_idx is not None:
del self.tag_stack[pop_idx:]
if tag == "small" and self.in_small:
self.in_small = False
if tag == "blockquote" and self.quote_depth > 0:
self.quote_depth -= 1
if self.cur is not None and self.in_desc:
self.cur["body"].append("\n")
if tag == "div":
if self.div_stack:
cls = self.div_stack.pop()
if "sreweekly-title" in cls:
self.in_title = False
elif "sreweekly-description" in cls:
self.in_desc = False
elif "sreweekly-entry" in cls:
self._finalize()
def handle_data(self, data):
if self.cur is None:
return
if self.in_title:
self.cur["title"] += data
elif self.in_desc:
if self.in_small:
self.cur["author"] = (self.cur["author"] or "") + data
else:
if not data.strip():
return # 标签间的纯空白(换行/缩进)跳过,段落结构由 p/blockquote 标记生成
if self.cur["body"] and self.cur["body"][-1] in ("\n> ", "\n\n", "\n"):
data = data.lstrip("\n\t \xa0")
self.cur["body"].append(data)
def _finalize(self):
if self.cur is None:
return
title = html.unescape(self.cur["title"]).strip()
body = "".join(self.cur["body"])
body = body.replace("\xa0", " ")
# 压缩多余的连续空行
body = re.sub(r"\n{3,}", "\n\n", body)
# 每行去掉首尾空白,但保留引用行前缀 ">"
lines = []
for ln in body.split("\n"):
stripped = ln.strip()
if stripped.startswith(">"):
lines.append("> " + stripped[1:].strip())
elif stripped:
lines.append(stripped)
else:
lines.append("")
body = "\n".join(lines).strip("\n")
author = html.unescape((self.cur["author"] or "").strip()) or None
self.entries.append({"title": title, "url": self.cur["url"], "body": body, "author": author})
self.cur = None
def parse_issue_html(html_path):
with open(html_path, "r", encoding="utf-8") as f:
p = SREEntryParser()
p.feed(f.read())
return p.entries
def slugify(title, max_len=70):
"""标题 → 文件名 slug:保留字母数字与 CJK,其余转连字符。"""
s = re.sub(r"[^\w\u4e00-\u9fff]+", "-", title.lower(), flags=re.UNICODE).strip("-")
s = re.sub(r"-{2,}", "-", s)
return s[:max_len].strip("-") or "untitled"
def extract_one(iid, entry, base_dir, force=False):
"""把某一期 HTML 解析为每篇文章一个 markdown 文件。返回 (文章数, 输出目录)。"""
md_root = os.path.join(base_dir, "markdown")
os.makedirs(md_root, exist_ok=True)
html_rel = entry.get("html_file")
if not html_rel:
sys.exit(f"❌ 期 #{iid} 没有 html_file 记录,请先扫描")
html_path = os.path.join(base_dir, html_rel)
if not os.path.exists(html_path):
sys.exit(f"❌ HTML 文件不存在: {html_path}")
if entry.get("extracted") and entry.get("article_count", 0) > 0 and not force:
print(f"⏭️ 期 #{iid} 已提取过({entry['article_count']} 篇),跳过(--force 可覆盖)", file=sys.stderr)
return entry.get("article_count", 0), entry.get("markdown_dir"), False
articles = parse_issue_html(html_path)
out_dir = os.path.join(md_root, str(iid))
os.makedirs(out_dir, exist_ok=True)
for idx, a in enumerate(articles, 1):
fname = f"{idx:02d}-{slugify(a['title'])}.md"
fpath = os.path.join(out_dir, fname)
content = render_markdown(a, entry)
with open(fpath, "w", encoding="utf-8") as f:
f.write(content)
entry["extracted"] = True
entry["article_count"] = len(articles)
entry["markdown_dir"] = f"markdown/{iid}"
entry["extracted_at"] = now_iso()
print(f"📄 期 #{iid} → {len(articles)} 篇文章 → {out_dir}", file=sys.stderr)
return len(articles), out_dir, True
def render_markdown(article, entry):
"""文章 → markdown 文件内容。"""
pub = entry.get("pub_date", "")
issue_label = entry.get("title", f"SRE Weekly Issue #{entry.get('id')}")
out = [f"# {article['title']}", ""]
out.append(f"- **期号**: {issue_label}({pub})")
out.append(f"- **作者**: {article['author'] or '—'}")
out.append(f"- **链接**: {article['url']}")
if article["body"]:
out += ["", "## 简介", "", article["body"]]
return "\n".join(out) + "\n"
def resolve_targets(target, manifest):
"""解析 --extract 的目标: ID / latest / pending / all → 期号列表。"""
issues = manifest["issues"]
if target == "latest":
ids = sorted(issues.keys(), key=lambda x: (isinstance(x, str), x))
# 数字优先按数值排,latest 取最大
nums = [i for i in issues if str(i).isdigit()]
return [str(max(int(n) for n in nums))] if nums else [ids[-1]]
if target == "pending":
return [str(i) for i, v in issues.items() if not v.get("extracted") or v.get("article_count", 0) == 0]
if target == "all":
return list(issues.keys())
t = str(target)
if t not in issues:
sys.exit(f"❌ 期 #{t} 不在 manifest 中,请先扫描")
return [t]
def cmd_extract(args, manifest, base_dir):
ids = resolve_targets(args.extract, manifest)
if not ids:
print("✅ 无待提取的期", file=sys.stderr)
return
for iid in ids:
extract_one(iid, manifest["issues"][str(iid)], base_dir, force=args.force)
save_manifest(base_dir, manifest)
# ---------------------------------------------------------------- 展示
def cmd_status(manifest):
issues = manifest["issues"]
print(f"📡 SRE Weekly 共 {len(issues)} 期(最后扫描: {manifest['last_scanned']})")
print()
header = f"{'期号':>6} {'日期':<12} {'HTML':<6} {'提取':<5} {'文章数':<6} 标题"
print(header)
print("-" * len(header))
for iid in sorted(issues.keys(), key=lambda x: int(x) if str(x).isdigit() else 0):
v = issues[iid]
html_ok = "✅" if v.get("html_file") else "—"
ext_ok = "✅" if v.get("extracted") and v.get("article_count", 0) > 0 else "—"
print(f"{iid:>6} {v.get('pub_date',''):<12} {html_ok:<6} {ext_ok:<5} {v.get('article_count',0):<6} {v.get('title','')}")
def cmd_issues(manifest):
for iid in sorted(manifest["issues"].keys(), key=lambda x: int(x) if str(x).isdigit() else 0):
v = manifest["issues"][iid]
print(f"{iid}\t{v.get('pub_date','')}\t{v.get('title','')}\t{v.get('url','')}")
# ---------------------------------------------------------------- main
def main():
ap = argparse.ArgumentParser(description="SRE Weekly 抓取 + 文章提取(供 n8n 定时调用)")
ap.add_argument("--feed", default=DEFAULT_FEED, help=f"RSS 地址(默认 {DEFAULT_FEED})")
ap.add_argument("--dir", default=DEFAULT_DIR, help=f"工作目录(默认 {DEFAULT_DIR})")
ap.add_argument("--check", action="store_true", help="只检查有无新期,输出 NEW:id,id 或 NONE")
ap.add_argument("--extract", metavar="ID|latest|pending|all", help="提取文章 → markdown")
ap.add_argument("--status", action="store_true", help="打印 manifest 状态表")
ap.add_argument("--issues", action="store_true", help="列出已登记的所有期")
ap.add_argument("--force", action="store_true", help="强制重新提取已提取过的期")
args = ap.parse_args()
base_dir = os.path.abspath(os.path.expanduser(args.dir))
os.makedirs(base_dir, exist_ok=True)
if args.extract or args.status or args.issues:
manifest = load_manifest(base_dir)
if args.extract:
cmd_extract(args, manifest, base_dir)
if args.status:
cmd_status(manifest)
if args.issues:
cmd_issues(manifest)
elif args.check:
manifest = load_manifest(base_dir)
cmd_check(args, manifest)
else:
manifest = load_manifest(base_dir)
cmd_scan(args, manifest, base_dir)
return 0
if __name__ == "__main__":
sys.exit(main())