#!/usr/bin/env python3 # -*- coding: utf-8 -*- """ SRE Weekly 抓取与文章提取脚本 模仿 blogwatcher-daily 的单文件风格:feedparser 抓取 → HTML 落盘 → 逐篇文章导出 Markdown, 并用 manifest.json 记录每期的状态(是否已存 HTML / 是否已提取文章)。 设计给 n8n 定时任务调用,通过不同参数驱动,脚本自身不发送邮件。 目录结构(默认工作目录 ~/Workspace/nexus/Hermes/xingzhi/sreweekly/): sreweekly/ ├── manifest.json # 状态记录(每期: html 是否保存、文章是否提取、文章数) ├── html/-<日期>.html # 每期原始 HTML(feed 的 content:encoded 原样保存) └── markdown// # 每期文章的 markdown 目录(每篇一个文件) ├── 01-<标题slug>.md └── 02-<标题slug>.md 用法: python3 sreweekly.py # 扫描 feed,新增期保存为 HTML 并登记 manifest python3 sreweekly.py --check # 只检查有无新期,输出 NEW:533,532 或 NONE(供 n8n 分支) python3 sreweekly.py --extract ID # 提取指定期(如 533)的文章 → markdown python3 sreweekly.py --extract latest # 提取最新一期 python3 sreweekly.py --extract pending # 提取所有尚未提取的期(n8n 每日流程推荐) python3 sreweekly.py --extract all # 提取全部(含已提取的,需配合 --force 覆盖) python3 sreweekly.py --status # 打印 manifest 状态表 python3 sreweekly.py --issues # 列出已登记的所有期 选项: --feed URL RSS 地址(默认 https://sreweekly.com/feed/) --dir PATH 工作目录(默认 ~/Workspace/nexus/Hermes/xingzhi/sreweekly/) --force 重新提取已提取过的期(覆盖旧文件) """ from html.parser import HTMLParser import argparse, html, json, os, re, sys, time, urllib.request try: import feedparser except ImportError: sys.exit("缺少依赖: 请先执行 pip3 install feedparser") DEFAULT_FEED = "https://sreweekly.com/feed/" DEFAULT_DIR = os.path.expanduser("~/Workspace/nexus/Hermes/xingzhi/sreweekly") # ---------------------------------------------------------------- manifest def load_manifest(base_dir): path = os.path.join(base_dir, "manifest.json") if os.path.exists(path): try: with open(path, "r", encoding="utf-8") as f: return json.load(f) except (json.JSONDecodeError, OSError) as e: print(f"⚠️ manifest.json 读取失败({e}),按空库处理", file=sys.stderr) return {"feed": DEFAULT_FEED, "last_scanned": None, "issues": {}} def save_manifest(base_dir, manifest): path = os.path.join(base_dir, "manifest.json") with open(path, "w", encoding="utf-8") as f: json.dump(manifest, f, ensure_ascii=False, indent=2) return path def issue_id_from_link(link): """从文章链接提取期号: sre-weekly-issue-533 → 533;失败则用最后一段路径。""" m = re.search(r"sre-weekly-issue-(\d+)", link or "") if m: return int(m.group(1)) m = re.search(r"/([^/]+?)/?$", (link or "").rstrip("/")) return m.group(1) if m else "unknown" def format_pubdate(entry): """优先用 feed 自带发布日期(YYYY-MM-DD),没有则用当天。""" if getattr(entry, "published_parsed", None): return time.strftime("%Y-%m-%d", entry.published_parsed) return time.strftime("%Y-%m-%d") def now_iso(): return time.strftime("%Y-%m-%dT%H:%M:%S") # ---------------------------------------------------------------- feed 扫描 def fetch_feed(feed_url): d = feedparser.parse(feed_url) if not d.entries: sys.exit(f"❌ feed 抓取失败或为空: {feed_url}(bozo={d.bozo})") return d.entries def cmd_scan(args, manifest, base_dir): html_dir = os.path.join(base_dir, "html") os.makedirs(html_dir, exist_ok=True) entries = fetch_feed(args.feed) new_ids = [] for e in entries: iid = str(issue_id_from_link(e.link)) if iid in manifest["issues"]: continue # 已登记,跳过(去重) content_html = e.content[0].value if getattr(e, "content", None) else "" pubdate = format_pubdate(e) fname = f"{iid}-{pubdate}.html" fpath = os.path.join(html_dir, fname) with open(fpath, "w", encoding="utf-8") as f: f.write(content_html) manifest["issues"][iid] = { "id": iid, "title": e.title, "url": e.link, "pub_date": pubdate, "html_file": f"html/{fname}", "fetched_at": now_iso(), "extracted": False, "article_count": 0, "markdown_dir": None, } new_ids.append(iid) print(f"🆕 新期 #{iid} {e.title} → {fname}({len(content_html)} bytes)", file=sys.stderr) manifest["last_scanned"] = now_iso() save_manifest(base_dir, manifest) print(f"📊 扫描完成:新增 {len(new_ids)} 期,共 {len(manifest['issues'])} 期", file=sys.stderr) return new_ids def cmd_check(args, manifest): entries = fetch_feed(args.feed) known = set(manifest["issues"].keys()) new_ids = [] for e in entries: iid = str(issue_id_from_link(e.link)) if iid not in known: new_ids.append(iid) if new_ids: print("NEW:" + ",".join(new_ids)) else: print("NONE") return 0 # ---------------------------------------------------------------- HTML → 文章 class SREEntryParser(HTMLParser): """解析每期的 content:encoded HTML,提取所有 sreweekly-entry 文章条目。 页面结构(见实测):
…赞助商…
← 跳过

简介…

引用…

  作者

""" def __init__(self): super().__init__(convert_charrefs=True) self.div_stack = [] # 当前打开中的 div class 栈 self.tag_stack = [] # 当前打开中的标签栈(用于识别 a/blockquote/small) self.entries = [] # 提取结果: [{title, url, body, author}, ...] self.cur = None self.in_title = False self.in_desc = False self.in_small = False self.quote_depth = 0 def _cur_div_class(self): return self.div_stack[-1] if self.div_stack else None def handle_starttag(self, tag, attrs): attrs = dict(attrs) self.tag_stack.append(tag) cls = attrs.get("class", "") if tag == "div": self.div_stack.append(cls) if "sreweekly-entry" in cls: self.cur = {"title": "", "url": "", "body": [], "author": None} elif "sreweekly-title" in cls and self.cur is not None: self.in_title = True elif "sreweekly-description" in cls and self.cur is not None: self.in_desc = True return if self.cur is None: return if tag == "a" and self.in_title: self.cur["url"] = attrs.get("href", "") elif tag == "blockquote" and self.in_desc: self.quote_depth += 1 if self.cur["body"] and self.cur["body"][-1] not in ("\n\n", "\n> ", "\n"): self.cur["body"].append("\n\n") self.cur["body"].append("\n> ") elif tag == "small" and self.in_desc: self.in_small = True elif tag == "p" and self.in_desc and self.cur["body"] and self.quote_depth == 0: # 段落间空行(避免在开头及 blockquote 内部加多余空行) self.cur["body"].append("\n\n") def handle_endtag(self, tag): if self.tag_stack: pop_idx = None for i in range(len(self.tag_stack) - 1, -1, -1): if self.tag_stack[i] == tag: pop_idx = i break if pop_idx is not None: del self.tag_stack[pop_idx:] if tag == "small" and self.in_small: self.in_small = False if tag == "blockquote" and self.quote_depth > 0: self.quote_depth -= 1 if self.cur is not None and self.in_desc: self.cur["body"].append("\n") if tag == "div": if self.div_stack: cls = self.div_stack.pop() if "sreweekly-title" in cls: self.in_title = False elif "sreweekly-description" in cls: self.in_desc = False elif "sreweekly-entry" in cls: self._finalize() def handle_data(self, data): if self.cur is None: return if self.in_title: self.cur["title"] += data elif self.in_desc: if self.in_small: self.cur["author"] = (self.cur["author"] or "") + data else: if not data.strip(): return # 标签间的纯空白(换行/缩进)跳过,段落结构由 p/blockquote 标记生成 if self.cur["body"] and self.cur["body"][-1] in ("\n> ", "\n\n", "\n"): data = data.lstrip("\n\t \xa0") self.cur["body"].append(data) def _finalize(self): if self.cur is None: return title = html.unescape(self.cur["title"]).strip() body = "".join(self.cur["body"]) body = body.replace("\xa0", " ") # 压缩多余的连续空行 body = re.sub(r"\n{3,}", "\n\n", body) # 每行去掉首尾空白,但保留引用行前缀 ">" lines = [] for ln in body.split("\n"): stripped = ln.strip() if stripped.startswith(">"): lines.append("> " + stripped[1:].strip()) elif stripped: lines.append(stripped) else: lines.append("") body = "\n".join(lines).strip("\n") author = html.unescape((self.cur["author"] or "").strip()) or None self.entries.append({"title": title, "url": self.cur["url"], "body": body, "author": author}) self.cur = None def parse_issue_html(html_path): with open(html_path, "r", encoding="utf-8") as f: p = SREEntryParser() p.feed(f.read()) return p.entries def slugify(title, max_len=70): """标题 → 文件名 slug:保留字母数字与 CJK,其余转连字符。""" s = re.sub(r"[^\w\u4e00-\u9fff]+", "-", title.lower(), flags=re.UNICODE).strip("-") s = re.sub(r"-{2,}", "-", s) return s[:max_len].strip("-") or "untitled" def extract_one(iid, entry, base_dir, force=False): """把某一期 HTML 解析为每篇文章一个 markdown 文件。返回 (文章数, 输出目录)。""" md_root = os.path.join(base_dir, "markdown") os.makedirs(md_root, exist_ok=True) html_rel = entry.get("html_file") if not html_rel: sys.exit(f"❌ 期 #{iid} 没有 html_file 记录,请先扫描") html_path = os.path.join(base_dir, html_rel) if not os.path.exists(html_path): sys.exit(f"❌ HTML 文件不存在: {html_path}") if entry.get("extracted") and entry.get("article_count", 0) > 0 and not force: print(f"⏭️ 期 #{iid} 已提取过({entry['article_count']} 篇),跳过(--force 可覆盖)", file=sys.stderr) return entry.get("article_count", 0), entry.get("markdown_dir"), False articles = parse_issue_html(html_path) out_dir = os.path.join(md_root, str(iid)) os.makedirs(out_dir, exist_ok=True) for idx, a in enumerate(articles, 1): fname = f"{idx:02d}-{slugify(a['title'])}.md" fpath = os.path.join(out_dir, fname) content = render_markdown(a, entry) with open(fpath, "w", encoding="utf-8") as f: f.write(content) entry["extracted"] = True entry["article_count"] = len(articles) entry["markdown_dir"] = f"markdown/{iid}" entry["extracted_at"] = now_iso() print(f"📄 期 #{iid} → {len(articles)} 篇文章 → {out_dir}", file=sys.stderr) return len(articles), out_dir, True def render_markdown(article, entry): """文章 → markdown 文件内容。""" pub = entry.get("pub_date", "") issue_label = entry.get("title", f"SRE Weekly Issue #{entry.get('id')}") out = [f"# {article['title']}", ""] out.append(f"- **期号**: {issue_label}({pub})") out.append(f"- **作者**: {article['author'] or '—'}") out.append(f"- **链接**: {article['url']}") if article["body"]: out += ["", "## 简介", "", article["body"]] return "\n".join(out) + "\n" def resolve_targets(target, manifest): """解析 --extract 的目标: ID / latest / pending / all → 期号列表。""" issues = manifest["issues"] if target == "latest": ids = sorted(issues.keys(), key=lambda x: (isinstance(x, str), x)) # 数字优先按数值排,latest 取最大 nums = [i for i in issues if str(i).isdigit()] return [str(max(int(n) for n in nums))] if nums else [ids[-1]] if target == "pending": return [str(i) for i, v in issues.items() if not v.get("extracted") or v.get("article_count", 0) == 0] if target == "all": return list(issues.keys()) t = str(target) if t not in issues: sys.exit(f"❌ 期 #{t} 不在 manifest 中,请先扫描") return [t] def cmd_extract(args, manifest, base_dir): ids = resolve_targets(args.extract, manifest) if not ids: print("✅ 无待提取的期", file=sys.stderr) return for iid in ids: extract_one(iid, manifest["issues"][str(iid)], base_dir, force=args.force) save_manifest(base_dir, manifest) # ---------------------------------------------------------------- 展示 def cmd_status(manifest): issues = manifest["issues"] print(f"📡 SRE Weekly 共 {len(issues)} 期(最后扫描: {manifest['last_scanned']})") print() header = f"{'期号':>6} {'日期':<12} {'HTML':<6} {'提取':<5} {'文章数':<6} 标题" print(header) print("-" * len(header)) for iid in sorted(issues.keys(), key=lambda x: int(x) if str(x).isdigit() else 0): v = issues[iid] html_ok = "✅" if v.get("html_file") else "—" ext_ok = "✅" if v.get("extracted") and v.get("article_count", 0) > 0 else "—" print(f"{iid:>6} {v.get('pub_date',''):<12} {html_ok:<6} {ext_ok:<5} {v.get('article_count',0):<6} {v.get('title','')}") def cmd_issues(manifest): for iid in sorted(manifest["issues"].keys(), key=lambda x: int(x) if str(x).isdigit() else 0): v = manifest["issues"][iid] print(f"{iid}\t{v.get('pub_date','')}\t{v.get('title','')}\t{v.get('url','')}") # ---------------------------------------------------------------- main def main(): ap = argparse.ArgumentParser(description="SRE Weekly 抓取 + 文章提取(供 n8n 定时调用)") ap.add_argument("--feed", default=DEFAULT_FEED, help=f"RSS 地址(默认 {DEFAULT_FEED})") ap.add_argument("--dir", default=DEFAULT_DIR, help=f"工作目录(默认 {DEFAULT_DIR})") ap.add_argument("--check", action="store_true", help="只检查有无新期,输出 NEW:id,id 或 NONE") ap.add_argument("--extract", metavar="ID|latest|pending|all", help="提取文章 → markdown") ap.add_argument("--status", action="store_true", help="打印 manifest 状态表") ap.add_argument("--issues", action="store_true", help="列出已登记的所有期") ap.add_argument("--force", action="store_true", help="强制重新提取已提取过的期") args = ap.parse_args() base_dir = os.path.abspath(os.path.expanduser(args.dir)) os.makedirs(base_dir, exist_ok=True) if args.extract or args.status or args.issues: manifest = load_manifest(base_dir) if args.extract: cmd_extract(args, manifest, base_dir) if args.status: cmd_status(manifest) if args.issues: cmd_issues(manifest) elif args.check: manifest = load_manifest(base_dir) cmd_check(args, manifest) else: manifest = load_manifest(base_dir) cmd_scan(args, manifest, base_dir) return 0 if __name__ == "__main__": sys.exit(main())