更新脚本和技能
This commit is contained in:
@@ -4,13 +4,34 @@ Blogwatcher Daily - RSS Feed 监控脚本
|
||||
放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py
|
||||
|
||||
Usage:
|
||||
python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库
|
||||
python3 blogwatcher-daily.py --list # 列出所有订阅
|
||||
python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅
|
||||
python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读
|
||||
python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown
|
||||
# 扫描 / 抓取
|
||||
python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库
|
||||
python3 blogwatcher-daily.py --update "频道名" # 只更新指定订阅(可用 --list 序号或名称)
|
||||
python3 blogwatcher-daily.py --update 3 --all # 强制回扫指定订阅最新 10 篇
|
||||
|
||||
# 订阅管理
|
||||
python3 blogwatcher-daily.py --list # 列出所有订阅
|
||||
python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅
|
||||
|
||||
# 查看文章
|
||||
python3 blogwatcher-daily.py --articles "频道名" # 列出该订阅所有文章
|
||||
python3 blogwatcher-daily.py --articles 3 --unread # 只列该订阅的未读文章
|
||||
|
||||
# 标记已读
|
||||
python3 blogwatcher-daily.py --mark-read # 全库标记为已读
|
||||
python3 blogwatcher-daily.py --mark-read "频道名" # 只标记指定订阅
|
||||
|
||||
# 删除
|
||||
python3 blogwatcher-daily.py --delete "频道名" # 删除该订阅下所有文章
|
||||
python3 blogwatcher-daily.py --delete 3 --article-id 2 # 删除该订阅下第 2 篇(1-based,顺序同 --articles)
|
||||
|
||||
# 导出
|
||||
python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown
|
||||
python3 blogwatcher-daily.py --export html --date 2026-08-28
|
||||
python3 blogwatcher-daily.py --export txt --limit 20
|
||||
|
||||
订阅标识(SUB):
|
||||
可传 --list 输出中的整数序号(如 3),也可传订阅名(精确 → 忽略大小写精确 → 忽略大小写包含)。
|
||||
"""
|
||||
|
||||
import os
|
||||
@@ -237,7 +258,57 @@ def list_subscriptions():
|
||||
print(f" [{i}] {sub['name']}")
|
||||
print(f" {sub['url']}\n")
|
||||
|
||||
def find_subscription(identifier):
|
||||
"""按序号(--list 中的 1-based 索引)或名称定位订阅。
|
||||
匹配顺序:数字 → 序号;否则 精确 → 忽略大小写精确 → 忽略大小写包含。
|
||||
多条匹配时打印候选并返回 None。
|
||||
"""
|
||||
subs = load_subscriptions()
|
||||
if not subs:
|
||||
print("❌ 订阅列表为空", file=sys.stderr)
|
||||
return None
|
||||
|
||||
if identifier.isdigit():
|
||||
idx = int(identifier)
|
||||
if 1 <= idx <= len(subs):
|
||||
return subs[idx - 1]
|
||||
|
||||
for sub in subs:
|
||||
if sub['name'] == identifier:
|
||||
return sub
|
||||
|
||||
lower = identifier.lower()
|
||||
exact_ci = [s for s in subs if s['name'].lower() == lower]
|
||||
if len(exact_ci) == 1:
|
||||
return exact_ci[0]
|
||||
if len(exact_ci) > 1:
|
||||
print(f"❌ 忽略大小写下匹配到多个订阅: {[s['name'] for s in exact_ci]}", file=sys.stderr)
|
||||
return None
|
||||
|
||||
substr = [s for s in subs if lower in s['name'].lower()]
|
||||
if len(substr) == 1:
|
||||
return substr[0]
|
||||
if len(substr) > 1:
|
||||
print(f"❌ 匹配到多个订阅: {[s['name'] for s in substr]}", file=sys.stderr)
|
||||
return None
|
||||
|
||||
print(f"❌ 未找到订阅: {identifier}", file=sys.stderr)
|
||||
return None
|
||||
|
||||
# ========== 主流程 ==========
|
||||
def _scan_one(conn, sub, force_all=False):
|
||||
"""扫描单个订阅(内部辅助)。返回 (channel_title 或 None, 抓取总数, 新增数量)。"""
|
||||
channel_title, items = fetch_rss(sub['url'])
|
||||
if items is None:
|
||||
return None, 0, 0
|
||||
|
||||
items_to_save = items[:10] if force_all else [i for i in items[:10] if is_article_new(conn, i['link'])]
|
||||
for item in items_to_save:
|
||||
save_article(conn, sub['url'], channel_title, item['title'], item['link'],
|
||||
item['description'], item['pub_date'])
|
||||
return channel_title, len(items), len(items_to_save)
|
||||
|
||||
|
||||
def scan_all(force_all=False):
|
||||
"""扫描所有订阅
|
||||
force_all: True 则忽略已读状态,每个频道强制抓10篇
|
||||
@@ -253,45 +324,117 @@ def scan_all(force_all=False):
|
||||
return
|
||||
|
||||
conn = get_db()
|
||||
all_new_articles = []
|
||||
new_count = 0
|
||||
|
||||
print(f"\n📡 开始扫描 {len(subs)} 个订阅...\n")
|
||||
|
||||
for sub in subs:
|
||||
print(f"🔍 扫描: {sub['name']}")
|
||||
|
||||
channel_title, items = fetch_rss(sub['url'])
|
||||
|
||||
if items is None:
|
||||
channel_title, total, new_in_feed = _scan_one(conn, sub, force_all=force_all)
|
||||
if channel_title is None:
|
||||
print(f" ⏭️ 跳过\n")
|
||||
continue
|
||||
|
||||
new_in_feed = 0
|
||||
items_to_save = items[:10] if force_all else [item for item in items[:10] if is_article_new(conn, item['link'])]
|
||||
|
||||
for item in items_to_save:
|
||||
save_article(conn, sub['url'], channel_title, item['title'], item['link'],
|
||||
item['description'], item['pub_date'])
|
||||
all_new_articles.append({
|
||||
'channel': channel_title,
|
||||
**item
|
||||
})
|
||||
new_in_feed += 1
|
||||
new_count += 1
|
||||
|
||||
print(f" ✅ {channel_title}: {len(items)} 篇, 新增 {new_in_feed} 篇\n")
|
||||
new_count += new_in_feed
|
||||
print(f" ✅ {channel_title}: {total} 篇, 新增 {new_in_feed} 篇\n")
|
||||
|
||||
conn.close()
|
||||
|
||||
# 生成报告
|
||||
print("-" * 50)
|
||||
print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n")
|
||||
|
||||
if not all_new_articles:
|
||||
if new_count == 0:
|
||||
print("📭 今日无新文章")
|
||||
|
||||
return all_new_articles
|
||||
|
||||
|
||||
def scan_subscription(sub, force_all=False):
|
||||
print("=" * 50)
|
||||
print(f"Blogwatcher — 更新单个订阅: {sub['name']}")
|
||||
print("=" * 50)
|
||||
|
||||
conn = get_db()
|
||||
print(f"\n🔍 扫描: {sub['name']}")
|
||||
channel_title, total, new_in_feed = _scan_one(conn, sub, force_all=force_all)
|
||||
conn.close()
|
||||
|
||||
if channel_title is None:
|
||||
print(f" ⏭️ 跳过\n")
|
||||
print("-" * 50)
|
||||
print("📊 扫描完成: 未获取到新文章")
|
||||
return
|
||||
|
||||
print(f" ✅ {channel_title}: {total} 篇, 新增 {new_in_feed} 篇\n")
|
||||
print("-" * 50)
|
||||
print(f"📊 扫描完成: 新增 {new_in_feed} 篇文章")
|
||||
|
||||
# ========== 文章管理 ==========
|
||||
def _fetch_articles_for_feed(conn, feed_url, unread_only=False):
|
||||
conn.row_factory = sqlite3.Row
|
||||
query = """
|
||||
SELECT id, title, link, description, pub_date, is_read, fetched_at
|
||||
FROM articles
|
||||
WHERE feed_url = ?
|
||||
"""
|
||||
params = [feed_url]
|
||||
if unread_only:
|
||||
query += " AND is_read = 0"
|
||||
query += " ORDER BY fetched_at DESC, id DESC"
|
||||
return conn.execute(query, params).fetchall()
|
||||
|
||||
|
||||
def list_articles_for_subscription(sub, unread_only=False):
|
||||
conn = get_db()
|
||||
rows = _fetch_articles_for_feed(conn, sub['url'], unread_only=unread_only)
|
||||
conn.close()
|
||||
|
||||
label = "未读" if unread_only else "全部"
|
||||
print(f"\n📄 [{sub['name']}] {label}文章 ({len(rows)} 篇):\n")
|
||||
if not rows:
|
||||
print(" (无文章)")
|
||||
return
|
||||
|
||||
for i, row in enumerate(rows, 1):
|
||||
mark = "🆕" if not row['is_read'] else "📖"
|
||||
title = (row['title'] or 'Untitled').strip()
|
||||
print(f" [{i}] {mark} {title}")
|
||||
if row['link']:
|
||||
print(f" 🔗 {row['link']}")
|
||||
if row['pub_date']:
|
||||
print(f" 📅 {row['pub_date']}")
|
||||
print()
|
||||
|
||||
|
||||
def delete_articles_for_subscription(sub, article_index=None):
|
||||
"""删除该订阅下文章。
|
||||
article_index=None → 删除全部;否则按 --articles 显示顺序删除第 article_index 篇(1-based)。
|
||||
返回 True 表示删除操作成功(或空库时的幂等 no-op),False 表示失败(如序号越界)。
|
||||
"""
|
||||
conn = get_db()
|
||||
|
||||
if article_index is None:
|
||||
count = conn.execute(
|
||||
"SELECT COUNT(*) FROM articles WHERE feed_url = ?", (sub['url'],)
|
||||
).fetchone()[0]
|
||||
if count == 0:
|
||||
print(f"📭 [{sub['name']}] 无文章可删除")
|
||||
conn.close()
|
||||
return True
|
||||
conn.execute("DELETE FROM articles WHERE feed_url = ?", (sub['url'],))
|
||||
conn.commit()
|
||||
conn.close()
|
||||
print(f"🗑️ 已删除 [{sub['name']}] 全部 {count} 篇文章")
|
||||
return True
|
||||
|
||||
rows = _fetch_articles_for_feed(conn, sub['url'])
|
||||
if article_index < 1 or article_index > len(rows):
|
||||
print(f"❌ 无效序号 {article_index},[{sub['name']}] 当前有 {len(rows)} 篇文章", file=sys.stderr)
|
||||
conn.close()
|
||||
return False
|
||||
|
||||
victim = rows[article_index - 1]
|
||||
conn.execute("DELETE FROM articles WHERE id = ?", (victim['id'],))
|
||||
conn.commit()
|
||||
conn.close()
|
||||
print(f"🗑️ 已删除 [{sub['name']}] 第 {article_index} 篇: {(victim['title'] or 'Untitled').strip()}")
|
||||
return True
|
||||
|
||||
# ========== 导出 ==========
|
||||
def fetch_articles(target_date=None, limit=None):
|
||||
@@ -423,7 +566,18 @@ def main():
|
||||
parser = argparse.ArgumentParser(description='Blogwatcher Daily RSS 监控脚本')
|
||||
parser.add_argument('--list', '-l', action='store_true', help='列出所有订阅')
|
||||
parser.add_argument('--add', nargs=2, metavar=('NAME', 'URL'), help='添加订阅')
|
||||
parser.add_argument('--mark-read', action='store_true', help='标记所有为已读')
|
||||
parser.add_argument('--update', metavar='SUB',
|
||||
help='更新指定订阅(--list 序号或名称),拉取最新文章入库;可搭配 --all')
|
||||
parser.add_argument('--articles', metavar='SUB',
|
||||
help='列出指定订阅的文章;可搭配 --unread 只看未读')
|
||||
parser.add_argument('--unread', action='store_true',
|
||||
help='与 --articles 搭配:只列出未读文章')
|
||||
parser.add_argument('--mark-read', nargs='?', const='__ALL__', metavar='SUB',
|
||||
help='标记为已读:不带参数=全库;带 SUB=只标记指定订阅')
|
||||
parser.add_argument('--delete', metavar='SUB',
|
||||
help='删除指定订阅下的所有文章;可搭配 --article-id 删除单篇')
|
||||
parser.add_argument('--article-id', type=int, metavar='N',
|
||||
help='与 --delete 搭配:删除该订阅下第 N 篇(1-based,顺序同 --articles)')
|
||||
parser.add_argument('--rsshub', help='设置 RSSHub 地址')
|
||||
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇')
|
||||
parser.add_argument('--export', choices=['txt', 'markdown', 'html'],
|
||||
@@ -433,29 +587,60 @@ def main():
|
||||
help='导出的文章数上限;单独使用时忽略日期取全库最新 N 篇(仅 --export 有效)')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
|
||||
if args.unread and not args.articles:
|
||||
parser.error("--unread 必须与 --articles 一起使用")
|
||||
if args.article_id is not None and not args.delete:
|
||||
parser.error("--article-id 必须与 --delete 一起使用")
|
||||
|
||||
global RSSHUB_BASE
|
||||
if args.rsshub:
|
||||
RSSHUB_BASE = args.rsshub
|
||||
|
||||
|
||||
if args.list:
|
||||
list_subscriptions()
|
||||
|
||||
|
||||
elif args.add:
|
||||
name, url = args.add
|
||||
# 如果 URL 不是完整地址,尝试添加 RSSHub 前缀
|
||||
if not url.startswith('http'):
|
||||
url = f"{RSSHUB_BASE}/{url}"
|
||||
# YouTube URL 自动转为 RSSHub 格式
|
||||
stored_url = convert_to_stored_url(url)
|
||||
save_subscription(name, stored_url)
|
||||
|
||||
elif args.mark_read:
|
||||
conn = get_db()
|
||||
mark_all_read(conn)
|
||||
conn.close()
|
||||
print("✅ 已标记所有文章为已读")
|
||||
|
||||
|
||||
elif args.update:
|
||||
sub = find_subscription(args.update)
|
||||
if not sub:
|
||||
sys.exit(1)
|
||||
scan_subscription(sub, force_all=args.all)
|
||||
|
||||
elif args.articles:
|
||||
sub = find_subscription(args.articles)
|
||||
if not sub:
|
||||
sys.exit(1)
|
||||
list_articles_for_subscription(sub, unread_only=args.unread)
|
||||
|
||||
elif args.mark_read is not None:
|
||||
if args.mark_read == '__ALL__':
|
||||
conn = get_db()
|
||||
mark_all_read(conn)
|
||||
conn.close()
|
||||
print("✅ 已标记所有文章为已读")
|
||||
else:
|
||||
sub = find_subscription(args.mark_read)
|
||||
if not sub:
|
||||
sys.exit(1)
|
||||
conn = get_db()
|
||||
mark_all_read(conn, feed_url=sub['url'])
|
||||
conn.close()
|
||||
print(f"✅ 已标记 [{sub['name']}] 所有文章为已读")
|
||||
|
||||
elif args.delete:
|
||||
sub = find_subscription(args.delete)
|
||||
if not sub:
|
||||
sys.exit(1)
|
||||
if not delete_articles_for_subscription(sub, article_index=args.article_id):
|
||||
sys.exit(1)
|
||||
|
||||
elif args.export:
|
||||
if args.date:
|
||||
target_date = args.date
|
||||
@@ -464,7 +649,7 @@ def main():
|
||||
else:
|
||||
target_date = date.today().isoformat()
|
||||
export_articles(args.export, target_date, args.limit)
|
||||
|
||||
|
||||
else:
|
||||
scan_all(force_all=args.all)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user