diff --git a/.gitignore b/.gitignore index 74c13cb..4f12652 100644 --- a/.gitignore +++ b/.gitignore @@ -2,3 +2,5 @@ .DS_Store **/.DS_Store + +__pycache__/ \ No newline at end of file diff --git a/blogwatcher-daily/SKILL.md b/blogwatcher-daily/SKILL.md index 28f8fb8..b04b765 100644 --- a/blogwatcher-daily/SKILL.md +++ b/blogwatcher-daily/SKILL.md @@ -1,12 +1,12 @@ --- name: blogwatcher-daily -description: RSS 订阅监控 + 每日笔记生成。使用 RSSHub + feedparser 抓取 31 个订阅,自动去重并存入 SQLite,新文章追加写入 Markdown 笔记。 -version: 1.0 +description: RSS 订阅监控 + 文章导出。使用 RSSHub + feedparser 抓取 31 个订阅,自动去重并存入 SQLite;可将文章按 txt/markdown/html 三种格式导出。 +version: 1.3 category: custom -tags: [rss, blog, monitoring, automation] +tags: [rss, blog, monitoring, automation, export] metadata: author: Hermes Agent - last_updated: 2025-04-19 + last_updated: 2026-08-29 platform: macos, ubuntu custom_skill_path: /Users/weishen/.hermes/skills/custom/ installation_note: "blogwatcher-daily 脚本在 Mac mini 本地运行(依赖 feedparser)。需要 RSSHub 服务(http://192.168.3.45:1200)访问 YouTube/Bilibili 等被墙源。" @@ -14,7 +14,7 @@ metadata: # Blogwatcher Daily -RSS 订阅监控 + 每日笔记生成自动化。 +RSS 订阅监控自动化,抓取后去重入库,可按 txt / markdown / html 三种格式导出。 ## 依赖 @@ -22,7 +22,7 @@ RSS 订阅监控 + 每日笔记生成自动化。 pip3 install feedparser ``` -> feedparser 是 Python 最成熟的 RSS 解析库,支持 RSS 1.0/2.0/Atom、任意编码、畸形 XML。 +> feedparser 是 Python 最成熟的 RSS 解析库,支持 RSS 1.0/2.0/Atom、任意编码、畸形 XML。HTML 导出使用 Python 标准库 `html.escape`,无第三方依赖。 ## 实际抓取架构(重要) @@ -48,35 +48,13 @@ curl → 目标网站 - 已有的 RSSHub URL → 直接使用 - 其他 → 直接访问原始 URL -``` -YouTube 频道 URL - ↓ 脚本自动识别并转为 RSSHub 格式 -http://192.168.3.45:1200/youtube/channel/{id} - ↓ -curl → RSSHub → YouTube - -普通 RSS(Engadget, Slashdot 等) - ↓ 脚本直接访问(绕过 RSSHub /rss/ 路由) -原始 RSS URL - ↓ -curl → 目标网站 -``` - -**关键发现**:RSSHub 的 `/rss/{url}` 路由不稳定(返回 RSSHub 欢迎页), -因此普通 RSS 源直接访问,不走 RSSHub 代理。 - -脚本内部 `build_fetch_url()` 根据 URL 类型自动选择路由: -- YouTube → RSSHub -- 已有的 RSSHub URL → 直接使用 -- 其他 → 直接访问原始 URL - ## 目录结构 ``` ~/.hermes/skills/custom/blogwatcher-daily/ ├── SKILL.md # 本文件 ├── scripts/ -│ └── blogwatcher-daily.py # 主脚本 +│ └── blogwatcher-daily.py # 扫描入库 + 订阅管理 + 文章导出(单文件) ├── subscriptions.txt # 订阅列表(name|URL) └── blogwatcher.db # SQLite 数据库(自动创建) ``` @@ -90,7 +68,7 @@ python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py ``` - 扫描所有订阅,新增文章存入数据库 -- 生成今日笔记:`~/Workspace/nexus/ishenwei/blogwatcher/YYYY-MM-DD.md` +- 控制台打印每个订阅新增文章数量及总计 ### 添加订阅 @@ -115,24 +93,13 @@ python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --list ``` -### 仅扫描(不写文件) - -```bash -python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --scan-only -``` - ### 强制回扫(`--all`) ```bash -# 强制抓取每频道10篇(忽略已读状态),写入独立文件 +# 强制抓取每频道10篇(忽略已读状态) python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --all - -# 测试模式:只打印,不写文件 -python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --scan-only --all ``` -- 输出文件:`~/Workspace/nexus/ishenwei/blogwatcher/all-YYYY-MM-DD.md`(**覆盖模式**,每次运行覆盖) -- **不混入**每日报告 `YYYY-MM-DD.md` - 每频道最多10篇 - 用途:历史内容回扫、测试订阅状态 @@ -142,6 +109,51 @@ python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py - python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --mark-read ``` +### 导出文章(`--export`) + +从 SQLite 里读取文章并输出到 stdout,格式通过 `--export {txt,markdown,html}` 选择: + +```bash +# 今天所有文章 → Markdown +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --export markdown + +# 指定日期 → HTML +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --export html --date 2026-08-28 > digest.html + +# 全库最新 20 篇 → 纯文本 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --export txt --limit 20 +``` + +**过滤规则**(`--date` 与 `--limit` 的组合语义) + +| 参数组合 | 行为 | +|---------|------| +| `--export FMT`(默认) | 今天(本地时区)的所有文章 | +| `--export FMT --date X` | 指定日期 X 的所有文章 | +| `--export FMT --limit N` | 忽略日期,取**全库最新 N 篇** | +| `--export FMT --date X --limit N` | 日期 X 的最新 N 篇 | + +**参数速查** + +| 参数 | 说明 | +|------|------| +| `--export {txt,markdown,html}` | 触发导出模式并指定格式 | +| `--date YYYY-MM-DD` | 目标日期(本地时区,默认今天) | +| `--limit N` | 文章数上限;单独使用时忽略日期 | + +**输出去向** + +- 正文 → **stdout**(可直接 `> file.md` 重定向) +- 进度 → **stderr**(`📊 N 篇文章 → FMT`) + +**格式细节** + +- `txt`:分频道分组,纯文本无标记,适合终端查看或管道 +- `markdown`:`## 【频道】` + `- [title](link)` + 描述引用块 +- `html`:完整 `` 单文件,内嵌简洁 CSS,标题/URL/描述全部经 `html.escape` 转义(防 XSS) + ## 添加订阅示例 ### YouTube 频道 @@ -182,13 +194,12 @@ python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ | 配置项 | 默认值 | |--------|--------| | RSSHub 地址 | `http://192.168.3.45:1200`(可设置 `RSSHUB_URL` 环境变量覆盖) | -| 笔记输出目录 | `~/Workspace/nexus/ishenwei/blogwatcher/` | | 数据库 | `~/.hermes/skills/custom/blogwatcher-daily/blogwatcher.db` | | 订阅列表 | `~/.hermes/skills/custom/blogwatcher-daily/subscriptions.txt` | ## Cron Job 设置 -推荐每天早上 6:00 自动执行: +推荐每天早上 6:00 自动扫描: ```bash cronjob --create \ @@ -200,13 +211,22 @@ cronjob --create \ 执行步骤: 1. 加载 blogwatcher-daily 技能 -2. 运行:python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py -3. 检查输出,确认写入的笔记文件路径 -4. 如果有新文章,简短汇总发给我" +2. 运行扫描:python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py +3. 汇报本次新增文章数量" +``` + +如需**扫描 + 落盘 HTML 摘要**(例如放进 Obsidian vault 或自建 dashboard),可用原生 crontab 串起来: + +```cron +0 6 * * * python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py && \ + python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --export html > ~/blogwatcher/$(date +\%Y-\%m-\%d).html ``` ## 数据流程 +### 扫描入库(默认命令) + 1. **加载订阅**:从 `subscriptions.txt` 读取所有 name|URL 对 2. **URL 路由**:`build_fetch_url()` 自动判断: - YouTube → 转为 `http://192.168.3.45:1200/youtube/channel/{id}` @@ -216,9 +236,14 @@ cronjob --create \ 4. **解析**:feedparser 提取 title、link、description、pub_date 5. **去重**:按 link 去重,已存在则跳过 6. **存储**:新文章存入 SQLite -7. **输出**:**追加写入** Markdown 笔记(同一文件多次扫描不会覆盖,仅追加新链接) - - 读取已有文件,用正则提取现有链接 → 去重 - - 新文章追加 `## 📦 新增 N 篇` 分隔块 +7. **输出**:控制台打印每个订阅新增数量与总计(不生成文件) + +### 导出(`--export`) + +1. **查询**:`fetch_articles()` 按 `date(fetched_at, 'localtime')` 过滤 + 可选 `LIMIT N` +2. **分组**:`_group_by_channel()` 按 `channel_title` 分桶 +3. **格式化**:`FORMATTERS[fmt]` 分发到 `format_txt` / `format_markdown` / `format_html` +4. **输出**:正文写 stdout,`📊 N 篇文章 → FMT` 进度写 stderr ## 已知问题 @@ -236,4 +261,5 @@ cronjob --create \ - 每次最多取每源 10 篇(TED Talks 等除外,取 20 篇) - 数据库自动创建,无需手动初始化 - OPML 文件可用 Python 解析后批量导入(参考 /wiki-ingest 流程中的 OPML 解析代码) -- **Markdown 追加模式**:同一天多次扫描不会覆盖旧内容,仅追加新链接(去重基于 URL) +- 导出的"今天"按**本地时区**(SQLite `date(fetched_at, 'localtime')`),不是 UTC +- HTML 导出使用 `html.escape` 转义标题/链接/描述,可安全嵌入网页;CSS 内嵌,单文件独立可用 diff --git a/blogwatcher-daily/scripts/__pycache__/blogwatcher-daily.cpython-311.pyc b/blogwatcher-daily/scripts/__pycache__/blogwatcher-daily.cpython-311.pyc deleted file mode 100644 index 4cb6ed4..0000000 Binary files a/blogwatcher-daily/scripts/__pycache__/blogwatcher-daily.cpython-311.pyc and /dev/null differ diff --git a/blogwatcher-daily/scripts/blogwatcher-daily.py b/blogwatcher-daily/scripts/blogwatcher-daily.py index 7cd8d66..752ec25 100644 --- a/blogwatcher-daily/scripts/blogwatcher-daily.py +++ b/blogwatcher-daily/scripts/blogwatcher-daily.py @@ -4,20 +4,23 @@ Blogwatcher Daily - RSS Feed 监控脚本 放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py Usage: - python3 blogwatcher-daily.py # 扫描所有订阅,写入今日笔记 - python3 blogwatcher-daily.py --list # 列出所有订阅 + python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库 + python3 blogwatcher-daily.py --list # 列出所有订阅 python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅 - python3 blogwatcher-daily.py --scan-only # 只扫描不写入文件 - python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读 + python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读 + python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown + python3 blogwatcher-daily.py --export html --date 2026-08-28 + python3 blogwatcher-daily.py --export txt --limit 20 """ import os import re +import sys import sqlite3 import argparse import feedparser -from datetime import datetime -from html import unescape +from datetime import date +from html import escape, unescape from urllib.request import Request, urlopen from urllib.error import URLError, HTTPError import ssl @@ -27,7 +30,6 @@ SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) SKILL_DIR = os.path.dirname(SCRIPT_DIR) # .../research/blogwatcher-daily DB_PATH = os.path.join(SKILL_DIR, "blogwatcher.db") SUBSCRIPTIONS_FILE = os.path.join(SKILL_DIR, "subscriptions.txt") -OUTPUT_DIR = os.path.expanduser("~/Workspace/nexus/ishenwei/blogwatcher") # RSSHub 基础地址(可配置) RSSHUB_BASE = os.environ.get("RSSHUB_URL", "http://192.168.3.45:1200") @@ -235,99 +237,10 @@ def list_subscriptions(): print(f" [{i}] {sub['name']}") print(f" {sub['url']}\n") -# ========== Markdown 输出 ========== -def generate_markdown(results, date_str=None): - """生成 Markdown 格式""" - if date_str is None: - date_str = datetime.now().strftime("%Y-%m-%d") - - md = f"# Blogwatcher Daily - {date_str}\n\n" - md += f"_生成时间: {datetime.now().strftime('%H:%M:%S')}_\n\n" - md += "---\n\n" - - if not results: - md += "_今日无新文章_\n" - return md - - # 按频道分组 - channels = {} - for item in results: - ch = item['channel'] - if ch not in channels: - channels[ch] = [] - channels[ch].append(item) - - for channel, items in channels.items(): - md += f"## 【{channel}】\n\n" - for item in items: - md += f"- [{item['title']}]({item['link']})\n" - if item['description']: - md += f" {item['description'][:150]}...\n" - md += "\n" - md += "---\n\n" - - return md - -def save_daily_report(new_results, date_str=None): - """保存每日报告(追加模式,避免覆盖)""" - if date_str is None: - date_str = datetime.now().strftime("%Y-%m-%d") - - os.makedirs(OUTPUT_DIR, exist_ok=True) - filepath = os.path.join(OUTPUT_DIR, f"{date_str}.md") - - # 读取已有内容,提取已有链接避免重复 - existing_links = set() - if os.path.exists(filepath): - with open(filepath, 'r') as f: - existing_content = f.read() - # 简单提取已有链接(用于去重),去掉末尾的 ) 等标点 - import re as _re - raw_links = _re.findall(r'https?://\S+', existing_content) - existing_links = set() - for l in raw_links: - # 去掉末尾的 ) 等非 URL 字符 - while l and l[-1] in '),;:': - l = l[:-1] - existing_links.add(l) - else: - existing_content = "" - - # 如果没有新结果,且文件已存在,直接返回 - if not new_results and existing_content: - return filepath - - # 生成新内容的 header - new_md = "" - new_articles = [a for a in new_results if a['link'] not in existing_links] - - if new_articles: - new_md += f"\n## 📦 新增 {len(new_articles)} 篇 ({datetime.now().strftime('%H:%M:%S')})\n\n" - channels = {} - for item in new_articles: - ch = item['channel'] - if ch not in channels: - channels[ch] = [] - channels[ch].append(item) - for channel, items in channels.items(): - new_md += f"### 【{channel}】\n\n" - for item in items: - new_md += f"- [{item['title']}]({item['link']})\n" - if item['description']: - new_md += f" {item['description'][:150]}...\n" - new_md += "\n" - - # 追加写入 - with open(filepath, 'a') as f: - f.write(new_md) - - return filepath - # ========== 主流程 ========== -def scan_all(force_all=False, write_file=True): +def scan_all(force_all=False): """扫描所有订阅 force_all: True 则忽略已读状态,每个频道强制抓10篇 - write_file: True 则写入文件 """ print("=" * 50) print("Blogwatcher Daily Scan") @@ -375,32 +288,149 @@ def scan_all(force_all=False, write_file=True): print("-" * 50) print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n") - if all_new_articles and write_file: - if force_all: - # --all 模式:写入独立文件(不追加日常报告) - md = generate_markdown(all_new_articles) - filepath = os.path.join(OUTPUT_DIR, f"all-{datetime.now().strftime('%Y-%m-%d')}.md") - with open(filepath, 'w') as f: - f.write(md) - print(f"📝 已写入(force-all 模式): {filepath}") - else: - filepath = save_daily_report(all_new_articles) - print(f"📝 已写入: {filepath}") - elif not all_new_articles: + if not all_new_articles: print("📭 今日无新文章") return all_new_articles +# ========== 导出 ========== +def fetch_articles(target_date=None, limit=None): + """从 articles 表读取;可选按本地日期过滤(date(fetched_at,'localtime'))+ 可选限制数量""" + if not os.path.exists(DB_PATH): + print(f"❌ 数据库不存在: {DB_PATH}", file=sys.stderr) + return [] + + conn = sqlite3.connect(DB_PATH) + conn.row_factory = sqlite3.Row + + where = "" + params = [] + if target_date: + where = "WHERE date(fetched_at, 'localtime') = ?" + params.append(target_date) + + query = f""" + SELECT channel_title, title, link, description, pub_date, fetched_at + FROM articles + {where} + ORDER BY fetched_at DESC, channel_title + """ + if limit: + query += " LIMIT ?" + params.append(limit) + + cursor = conn.execute(query, params) + rows = [dict(r) for r in cursor.fetchall()] + conn.close() + return rows + +def _group_by_channel(articles): + grouped = {} + for a in articles: + ch = a["channel_title"] or "Unknown" + grouped.setdefault(ch, []).append(a) + return grouped + +def format_txt(articles, label): + lines = [f"Blogwatcher Daily — {label}", "=" * 50, ""] + if not articles: + lines.append("(无文章)") + return "\n".join(lines) + "\n" + lines.append(f"共 {len(articles)} 篇文章") + lines.append("") + for channel, items in _group_by_channel(articles).items(): + lines.append(f"[{channel}]") + for it in items: + title = (it["title"] or "Untitled").strip() + link = (it["link"] or "").strip() + lines.append(f" - {title}") + if link: + lines.append(f" {link}") + desc = (it["description"] or "").strip() + if desc: + lines.append(f" {desc[:200]}") + lines.append("") + return "\n".join(lines) + "\n" + +def format_markdown(articles, label): + md = f"# Blogwatcher Daily — {label}\n\n" + if not articles: + return md + "_无文章_\n" + md += f"共 **{len(articles)}** 篇文章。\n\n---\n\n" + for channel, items in _group_by_channel(articles).items(): + md += f"## 【{channel}】\n\n" + for it in items: + title = (it["title"] or "Untitled").strip() + link = (it["link"] or "").strip() + md += f"- [{title}]({link})\n" + desc = (it["description"] or "").strip() + if desc: + md += f" > {desc[:200]}\n" + md += "\n" + return md + +def format_html(articles, label): + parts = [ + "", + '
', + f"无文章
") + return "\n".join(parts) + "\n" + parts.append(f"共 {len(articles)} 篇文章。
{escape(desc[:200])}") + parts.append("