From 146350565c2c1a1274a2cbe93076249d96b655c5 Mon Sep 17 00:00:00 2001 From: admin Date: Sun, 30 Aug 2026 07:42:25 +0800 Subject: [PATCH] =?UTF-8?q?=E6=9B=B4=E6=96=B0=E8=84=9A=E6=9C=AC=E5=92=8C?= =?UTF-8?q?=E6=8A=80=E8=83=BD?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- blogwatcher-daily/SKILL.md | 197 ++++++++++++- .../scripts/blogwatcher-daily.py | 273 +++++++++++++++--- 2 files changed, 421 insertions(+), 49 deletions(-) diff --git a/blogwatcher-daily/SKILL.md b/blogwatcher-daily/SKILL.md index b04b765..b4e67f1 100644 --- a/blogwatcher-daily/SKILL.md +++ b/blogwatcher-daily/SKILL.md @@ -1,12 +1,12 @@ --- name: blogwatcher-daily -description: RSS 订阅监控 + 文章导出。使用 RSSHub + feedparser 抓取 31 个订阅,自动去重并存入 SQLite;可将文章按 txt/markdown/html 三种格式导出。 -version: 1.3 +description: RSS 订阅监控 + 文章管理 + 导出。使用 RSSHub + feedparser 抓取订阅,自动去重入库 SQLite;支持按单个订阅更新/查看/标记已读/删除文章,可将文章按 txt/markdown/html 导出。附「自然语言操作指南」,AI 智能体可直接把用户口语映射到 CLI 命令。 +version: 1.4 category: custom tags: [rss, blog, monitoring, automation, export] metadata: author: Hermes Agent - last_updated: 2026-08-29 + last_updated: 2026-08-30 platform: macos, ubuntu custom_skill_path: /Users/weishen/.hermes/skills/custom/ installation_note: "blogwatcher-daily 脚本在 Mac mini 本地运行(依赖 feedparser)。需要 RSSHub 服务(http://192.168.3.45:1200)访问 YouTube/Bilibili 等被墙源。" @@ -103,12 +103,107 @@ python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py - - 每频道最多10篇 - 用途:历史内容回扫、测试订阅状态 -### 标记已读 +### 更新单个订阅(`--update SUB`) + +只抓某一个订阅,不动其它源。SUB 可传订阅名或 `--list` 中的序号(详见下方「订阅标识(SUB)」)。 ```bash -python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --mark-read +# 按名字更新 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --update "Tech With Tim" + +# 按 --list 里的序号更新(例如第 3 个订阅) +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --update 3 + +# 强制回扫单个订阅最新 10 篇(叠加 --all) +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --update 3 --all ``` +- 抓取失败 → 打印错误、退出 0(不打断脚本管道) +- 订阅未找到 → 退出 1 + +### 查看某订阅的文章(`--articles SUB`) + +从数据库读文章(不联网),可选只看未读: + +```bash +# 该订阅全部文章(按 fetched_at 降序) +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --articles "Engadget" + +# 只看未读 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --articles "Engadget" --unread + +# 也可用序号 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --articles 22 --unread +``` + +输出示例: + +``` +📄 [Engadget] 未读文章 (244 篇): + + [1] 🆕 How long is a new Fire TV Stick actually supposed to last? + 🔗 https://www.engadget.com/... + 📅 Sat, 30 Aug 2026 12:00:00 GMT + ... +``` + +- 已读用 `📖`,未读用 `🆕` +- 每篇左侧的 `[N]` 是**订阅内 1-based 序号**,可直接用于下方 `--delete --article-id N` + +### 标记已读(`--mark-read [SUB]`) + +不带参数 → 全库标记;带 SUB → 只标记指定订阅。 + +```bash +# 全库标记为已读 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py --mark-read + +# 只标记某个订阅 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --mark-read "Tech With Tim" + +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --mark-read 3 +``` + +### 删除文章(`--delete SUB [--article-id N]`) + +删除该订阅下**所有文章**或**单篇**。序号 N 与 `--articles` 显示顺序一致。 + +```bash +# 删除某订阅下所有文章 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --delete "Jon Law" + +# 删除某订阅下第 2 篇(先用 --articles 查序号) +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --delete "Jon Law" --article-id 2 + +# 序号也支持 +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ + --delete 2 --article-id 1 +``` + +- 空库删除是幂等 no-op,退出 0 +- 序号越界 → 退出 1,不改动 DB +- 删除**不可撤销**,请配合 `--articles SUB` 先看清再删 + +### 订阅标识(SUB) + +`--update / --articles / --mark-read / --delete` 的 SUB 参数按以下顺序匹配: + +1. 数字 → 视为 `--list` 中的 1-based 序号 +2. 精确名称(区分大小写) +3. 忽略大小写精确匹配 +4. 忽略大小写子串匹配 + +匹配到多条时脚本报错并列出候选:AI 智能体应向用户回问「你指的是 X 还是 Y?」或改用序号消歧。 + ### 导出文章(`--export`) 从 SQLite 里读取文章并输出到 stdout,格式通过 `--export {txt,markdown,html}` 选择: @@ -154,6 +249,98 @@ python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py \ - `markdown`:`## 【频道】` + `- [title](link)` + 描述引用块 - `html`:完整 `` 单文件,内嵌简洁 CSS,标题/URL/描述全部经 `html.escape` 转义(防 XSS) +## 自然语言操作指南(AI 智能体路由表) + +用户通常不会记 CLI 参数。下面是常见口语意图 → 应执行命令的映射,AI 智能体应据此选择命令。 + +命令前缀统一是: +``` +python3 ~/.hermes/skills/custom/blogwatcher-daily/scripts/blogwatcher-daily.py +``` +下表省略前缀,只列参数。 + +### 1. 抓取 / 更新 + +| 用户可能说 | 命令 | 备注 | +|---|---|---| +| "刷新所有博客"、"扫一下今天有什么新的"、"跑一下每日抓取"、"更新一下" | *(无参数)* | 全量扫描所有订阅 | +| "更新 Tech With Tim"、"抓一下 Engadget 最新的"、"看看第 3 个订阅有没有更新" | `--update "SUB"` | SUB = 名字或序号 | +| "强制回扫 X 最新 10 篇"、"忽略已读重新抓 X"、"重新拉 X 全部" | `--update "SUB" --all` | 忽略去重、强抓 10 篇 | +| "强制回扫所有订阅"、"忽略已读全部重抓" | `--all` | 全量强抓 | + +### 2. 查看订阅 + +| 用户可能说 | 命令 | +|---|---| +| "列出所有订阅"、"我订了哪些博客"、"有哪些频道"、"看订阅列表" | `--list` | + +### 3. 查看文章 + +| 用户可能说 | 命令 | +|---|---| +| "看看 Engadget 里都有什么文章"、"X 的所有文章"、"列出 X 的文章" | `--articles "SUB"` | +| "X 还没看的"、"X 有哪些未读"、"X 的未读列表"、"哪些没读过" | `--articles "SUB" --unread` | +| "第 3 个订阅有什么文章" | `--articles 3` | + +> AI 智能体判断"未读"意图的关键词:**未读 / 没看 / 还没读 / 新的 / unread / 待看**。 + +### 4. 标记已读 + +| 用户可能说 | 命令 | +|---|---| +| "全部标为已读"、"清一下未读"、"我都看过了"、"全部当作看过" | `--mark-read` | +| "把 X 都当作看过了"、"X 全部标为已读"、"这个频道我不感兴趣了先标已读" | `--mark-read "SUB"` | + +### 5. 删除文章 + +| 用户可能说 | 命令 | 危险度 | +|---|---|---| +| "删掉 X 的所有文章"、"清空 X 的历史"、"重置 X" | `--delete "SUB"` | 高 — 建议先 `--articles SUB` 确认 | +| "删除 X 里第 3 篇"、"把 X 的第 2 条去掉"、"X 的 [5] 号删了" | `--delete "SUB" --article-id N` | 中 — 序号来自 `--articles` 输出 | + +> **重要**:删除不可逆。AI 智能体收到"删除 / 清空 / 移除"类意图时,除非用户明确肯定,应先执行 `--articles SUB` 展示待删内容再回问一次。 + +### 6. 添加订阅 + +| 用户可能说 | 命令 | +|---|---| +| "订阅 X"、"添加频道 X"、"加个 RSS"、"帮我关注 X"(附 URL) | `--add "NAME" "URL"` | + +- YouTube URL、RSSHub URL、原生 RSS URL 都直接传,脚本内部自动路由。 + +### 7. 导出 + +| 用户可能说 | 命令 | +|---|---| +| "把今天的文章导出成 markdown"、"生成今日摘要 md" | `--export markdown` | +| "导出 2026-08-28 的 HTML"、"我要那天的网页版" | `--export html --date 2026-08-28` | +| "最新 20 篇导出为 txt"、"给我全库最新 20 篇" | `--export txt --limit 20` | +| "今天的文章导 html 到桌面" | `--export html > ~/Desktop/$(date +%Y-%m-%d).html` | + +### 订阅标识(SUB)如何解析 + +`--update / --articles / --mark-read / --delete` 的 SUB 参数按下列顺序匹配: + +1. **数字** → 视为 `--list` 中的 1-based 序号 +2. **精确名称**(区分大小写) +3. **忽略大小写精确** +4. **忽略大小写子串** + +匹配到多条时脚本会打印候选并退出 1。此时 AI 智能体应: +- 向用户回问「你指的是 X 还是 Y?」 +- 或改用 `--list` 里的序号做无歧义定位 + +匹配不到时也退出 1。 + +### AI 智能体决策要点 + +1. **看/查询类**(无副作用):`--list`、`--articles`。可直接执行。 +2. **抓取类**(可能改 DB):`--update`、无参扫描、`--all`。安全,直接跑。 +3. **写入类**(会改 DB):`--mark-read`、`--add`。可直接执行,但结束后简报结果。 +4. **删除类**(不可逆):`--delete`。**先展示要删的内容再执行**,除非用户在同一轮明确肯定。 +5. **导出类**(只读):`--export`。stdout 输出,AI 应帮用户决定重定向到哪个文件。 +6. **SUB 消歧**:能用序号就优先用序号(`--list` 是廉价的),避免多义。 + ## 添加订阅示例 ### YouTube 频道 diff --git a/blogwatcher-daily/scripts/blogwatcher-daily.py b/blogwatcher-daily/scripts/blogwatcher-daily.py index 752ec25..463d575 100644 --- a/blogwatcher-daily/scripts/blogwatcher-daily.py +++ b/blogwatcher-daily/scripts/blogwatcher-daily.py @@ -4,13 +4,34 @@ Blogwatcher Daily - RSS Feed 监控脚本 放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py Usage: - python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库 - python3 blogwatcher-daily.py --list # 列出所有订阅 - python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅 - python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读 - python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown + # 扫描 / 抓取 + python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库 + python3 blogwatcher-daily.py --update "频道名" # 只更新指定订阅(可用 --list 序号或名称) + python3 blogwatcher-daily.py --update 3 --all # 强制回扫指定订阅最新 10 篇 + + # 订阅管理 + python3 blogwatcher-daily.py --list # 列出所有订阅 + python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅 + + # 查看文章 + python3 blogwatcher-daily.py --articles "频道名" # 列出该订阅所有文章 + python3 blogwatcher-daily.py --articles 3 --unread # 只列该订阅的未读文章 + + # 标记已读 + python3 blogwatcher-daily.py --mark-read # 全库标记为已读 + python3 blogwatcher-daily.py --mark-read "频道名" # 只标记指定订阅 + + # 删除 + python3 blogwatcher-daily.py --delete "频道名" # 删除该订阅下所有文章 + python3 blogwatcher-daily.py --delete 3 --article-id 2 # 删除该订阅下第 2 篇(1-based,顺序同 --articles) + + # 导出 + python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown python3 blogwatcher-daily.py --export html --date 2026-08-28 python3 blogwatcher-daily.py --export txt --limit 20 + +订阅标识(SUB): + 可传 --list 输出中的整数序号(如 3),也可传订阅名(精确 → 忽略大小写精确 → 忽略大小写包含)。 """ import os @@ -237,7 +258,57 @@ def list_subscriptions(): print(f" [{i}] {sub['name']}") print(f" {sub['url']}\n") +def find_subscription(identifier): + """按序号(--list 中的 1-based 索引)或名称定位订阅。 + 匹配顺序:数字 → 序号;否则 精确 → 忽略大小写精确 → 忽略大小写包含。 + 多条匹配时打印候选并返回 None。 + """ + subs = load_subscriptions() + if not subs: + print("❌ 订阅列表为空", file=sys.stderr) + return None + + if identifier.isdigit(): + idx = int(identifier) + if 1 <= idx <= len(subs): + return subs[idx - 1] + + for sub in subs: + if sub['name'] == identifier: + return sub + + lower = identifier.lower() + exact_ci = [s for s in subs if s['name'].lower() == lower] + if len(exact_ci) == 1: + return exact_ci[0] + if len(exact_ci) > 1: + print(f"❌ 忽略大小写下匹配到多个订阅: {[s['name'] for s in exact_ci]}", file=sys.stderr) + return None + + substr = [s for s in subs if lower in s['name'].lower()] + if len(substr) == 1: + return substr[0] + if len(substr) > 1: + print(f"❌ 匹配到多个订阅: {[s['name'] for s in substr]}", file=sys.stderr) + return None + + print(f"❌ 未找到订阅: {identifier}", file=sys.stderr) + return None + # ========== 主流程 ========== +def _scan_one(conn, sub, force_all=False): + """扫描单个订阅(内部辅助)。返回 (channel_title 或 None, 抓取总数, 新增数量)。""" + channel_title, items = fetch_rss(sub['url']) + if items is None: + return None, 0, 0 + + items_to_save = items[:10] if force_all else [i for i in items[:10] if is_article_new(conn, i['link'])] + for item in items_to_save: + save_article(conn, sub['url'], channel_title, item['title'], item['link'], + item['description'], item['pub_date']) + return channel_title, len(items), len(items_to_save) + + def scan_all(force_all=False): """扫描所有订阅 force_all: True 则忽略已读状态,每个频道强制抓10篇 @@ -253,45 +324,117 @@ def scan_all(force_all=False): return conn = get_db() - all_new_articles = [] new_count = 0 print(f"\n📡 开始扫描 {len(subs)} 个订阅...\n") for sub in subs: print(f"🔍 扫描: {sub['name']}") - - channel_title, items = fetch_rss(sub['url']) - - if items is None: + channel_title, total, new_in_feed = _scan_one(conn, sub, force_all=force_all) + if channel_title is None: print(f" ⏭️ 跳过\n") continue - - new_in_feed = 0 - items_to_save = items[:10] if force_all else [item for item in items[:10] if is_article_new(conn, item['link'])] - - for item in items_to_save: - save_article(conn, sub['url'], channel_title, item['title'], item['link'], - item['description'], item['pub_date']) - all_new_articles.append({ - 'channel': channel_title, - **item - }) - new_in_feed += 1 - new_count += 1 - - print(f" ✅ {channel_title}: {len(items)} 篇, 新增 {new_in_feed} 篇\n") + new_count += new_in_feed + print(f" ✅ {channel_title}: {total} 篇, 新增 {new_in_feed} 篇\n") conn.close() - # 生成报告 print("-" * 50) print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n") - - if not all_new_articles: + if new_count == 0: print("📭 今日无新文章") - - return all_new_articles + + +def scan_subscription(sub, force_all=False): + print("=" * 50) + print(f"Blogwatcher — 更新单个订阅: {sub['name']}") + print("=" * 50) + + conn = get_db() + print(f"\n🔍 扫描: {sub['name']}") + channel_title, total, new_in_feed = _scan_one(conn, sub, force_all=force_all) + conn.close() + + if channel_title is None: + print(f" ⏭️ 跳过\n") + print("-" * 50) + print("📊 扫描完成: 未获取到新文章") + return + + print(f" ✅ {channel_title}: {total} 篇, 新增 {new_in_feed} 篇\n") + print("-" * 50) + print(f"📊 扫描完成: 新增 {new_in_feed} 篇文章") + +# ========== 文章管理 ========== +def _fetch_articles_for_feed(conn, feed_url, unread_only=False): + conn.row_factory = sqlite3.Row + query = """ + SELECT id, title, link, description, pub_date, is_read, fetched_at + FROM articles + WHERE feed_url = ? + """ + params = [feed_url] + if unread_only: + query += " AND is_read = 0" + query += " ORDER BY fetched_at DESC, id DESC" + return conn.execute(query, params).fetchall() + + +def list_articles_for_subscription(sub, unread_only=False): + conn = get_db() + rows = _fetch_articles_for_feed(conn, sub['url'], unread_only=unread_only) + conn.close() + + label = "未读" if unread_only else "全部" + print(f"\n📄 [{sub['name']}] {label}文章 ({len(rows)} 篇):\n") + if not rows: + print(" (无文章)") + return + + for i, row in enumerate(rows, 1): + mark = "🆕" if not row['is_read'] else "📖" + title = (row['title'] or 'Untitled').strip() + print(f" [{i}] {mark} {title}") + if row['link']: + print(f" 🔗 {row['link']}") + if row['pub_date']: + print(f" 📅 {row['pub_date']}") + print() + + +def delete_articles_for_subscription(sub, article_index=None): + """删除该订阅下文章。 + article_index=None → 删除全部;否则按 --articles 显示顺序删除第 article_index 篇(1-based)。 + 返回 True 表示删除操作成功(或空库时的幂等 no-op),False 表示失败(如序号越界)。 + """ + conn = get_db() + + if article_index is None: + count = conn.execute( + "SELECT COUNT(*) FROM articles WHERE feed_url = ?", (sub['url'],) + ).fetchone()[0] + if count == 0: + print(f"📭 [{sub['name']}] 无文章可删除") + conn.close() + return True + conn.execute("DELETE FROM articles WHERE feed_url = ?", (sub['url'],)) + conn.commit() + conn.close() + print(f"🗑️ 已删除 [{sub['name']}] 全部 {count} 篇文章") + return True + + rows = _fetch_articles_for_feed(conn, sub['url']) + if article_index < 1 or article_index > len(rows): + print(f"❌ 无效序号 {article_index},[{sub['name']}] 当前有 {len(rows)} 篇文章", file=sys.stderr) + conn.close() + return False + + victim = rows[article_index - 1] + conn.execute("DELETE FROM articles WHERE id = ?", (victim['id'],)) + conn.commit() + conn.close() + print(f"🗑️ 已删除 [{sub['name']}] 第 {article_index} 篇: {(victim['title'] or 'Untitled').strip()}") + return True # ========== 导出 ========== def fetch_articles(target_date=None, limit=None): @@ -423,7 +566,18 @@ def main(): parser = argparse.ArgumentParser(description='Blogwatcher Daily RSS 监控脚本') parser.add_argument('--list', '-l', action='store_true', help='列出所有订阅') parser.add_argument('--add', nargs=2, metavar=('NAME', 'URL'), help='添加订阅') - parser.add_argument('--mark-read', action='store_true', help='标记所有为已读') + parser.add_argument('--update', metavar='SUB', + help='更新指定订阅(--list 序号或名称),拉取最新文章入库;可搭配 --all') + parser.add_argument('--articles', metavar='SUB', + help='列出指定订阅的文章;可搭配 --unread 只看未读') + parser.add_argument('--unread', action='store_true', + help='与 --articles 搭配:只列出未读文章') + parser.add_argument('--mark-read', nargs='?', const='__ALL__', metavar='SUB', + help='标记为已读:不带参数=全库;带 SUB=只标记指定订阅') + parser.add_argument('--delete', metavar='SUB', + help='删除指定订阅下的所有文章;可搭配 --article-id 删除单篇') + parser.add_argument('--article-id', type=int, metavar='N', + help='与 --delete 搭配:删除该订阅下第 N 篇(1-based,顺序同 --articles)') parser.add_argument('--rsshub', help='设置 RSSHub 地址') parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇') parser.add_argument('--export', choices=['txt', 'markdown', 'html'], @@ -433,29 +587,60 @@ def main(): help='导出的文章数上限;单独使用时忽略日期取全库最新 N 篇(仅 --export 有效)') args = parser.parse_args() - + + if args.unread and not args.articles: + parser.error("--unread 必须与 --articles 一起使用") + if args.article_id is not None and not args.delete: + parser.error("--article-id 必须与 --delete 一起使用") + global RSSHUB_BASE if args.rsshub: RSSHUB_BASE = args.rsshub - + if args.list: list_subscriptions() - + elif args.add: name, url = args.add - # 如果 URL 不是完整地址,尝试添加 RSSHub 前缀 if not url.startswith('http'): url = f"{RSSHUB_BASE}/{url}" - # YouTube URL 自动转为 RSSHub 格式 stored_url = convert_to_stored_url(url) save_subscription(name, stored_url) - - elif args.mark_read: - conn = get_db() - mark_all_read(conn) - conn.close() - print("✅ 已标记所有文章为已读") - + + elif args.update: + sub = find_subscription(args.update) + if not sub: + sys.exit(1) + scan_subscription(sub, force_all=args.all) + + elif args.articles: + sub = find_subscription(args.articles) + if not sub: + sys.exit(1) + list_articles_for_subscription(sub, unread_only=args.unread) + + elif args.mark_read is not None: + if args.mark_read == '__ALL__': + conn = get_db() + mark_all_read(conn) + conn.close() + print("✅ 已标记所有文章为已读") + else: + sub = find_subscription(args.mark_read) + if not sub: + sys.exit(1) + conn = get_db() + mark_all_read(conn, feed_url=sub['url']) + conn.close() + print(f"✅ 已标记 [{sub['name']}] 所有文章为已读") + + elif args.delete: + sub = find_subscription(args.delete) + if not sub: + sys.exit(1) + if not delete_articles_for_subscription(sub, article_index=args.article_id): + sys.exit(1) + elif args.export: if args.date: target_date = args.date @@ -464,7 +649,7 @@ def main(): else: target_date = date.today().isoformat() export_articles(args.export, target_date, args.limit) - + else: scan_all(force_all=args.all)