Files
atlas/blogwatcher-daily/scripts/blogwatcher-daily.py
2026-08-30 07:42:25 +08:00

658 lines
24 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Blogwatcher Daily - RSS Feed 监控脚本
放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py
Usage:
# 扫描 / 抓取
python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库
python3 blogwatcher-daily.py --update "频道名" # 只更新指定订阅(可用 --list 序号或名称)
python3 blogwatcher-daily.py --update 3 --all # 强制回扫指定订阅最新 10 篇
# 订阅管理
python3 blogwatcher-daily.py --list # 列出所有订阅
python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅
# 查看文章
python3 blogwatcher-daily.py --articles "频道名" # 列出该订阅所有文章
python3 blogwatcher-daily.py --articles 3 --unread # 只列该订阅的未读文章
# 标记已读
python3 blogwatcher-daily.py --mark-read # 全库标记为已读
python3 blogwatcher-daily.py --mark-read "频道名" # 只标记指定订阅
# 删除
python3 blogwatcher-daily.py --delete "频道名" # 删除该订阅下所有文章
python3 blogwatcher-daily.py --delete 3 --article-id 2 # 删除该订阅下第 2 篇(1-based,顺序同 --articles)
# 导出
python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown
python3 blogwatcher-daily.py --export html --date 2026-08-28
python3 blogwatcher-daily.py --export txt --limit 20
订阅标识(SUB):
可传 --list 输出中的整数序号(如 3),也可传订阅名(精确 → 忽略大小写精确 → 忽略大小写包含)。
"""
import os
import re
import sys
import sqlite3
import argparse
import feedparser
from datetime import date
from html import escape, unescape
from urllib.request import Request, urlopen
from urllib.error import URLError, HTTPError
import ssl
# ========== 配置 ==========
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
SKILL_DIR = os.path.dirname(SCRIPT_DIR) # .../research/blogwatcher-daily
DB_PATH = os.path.join(SKILL_DIR, "blogwatcher.db")
SUBSCRIPTIONS_FILE = os.path.join(SKILL_DIR, "subscriptions.txt")
# RSSHub 基础地址(可配置)
RSSHUB_BASE = os.environ.get("RSSHUB_URL", "http://192.168.3.45:1200")
# YouTube channel ID 提取正则
YOUTUBE_CHANNEL_RE = re.compile(r'youtube\.com/channel/([\w-]+)', re.IGNORECASE)
YOUTUBE_FEED_RE = re.compile(r'youtube\.com/feeds/videos\.xml\?channel_id=([\w-]+)', re.IGNORECASE)
YOUTUBE_USER_RE = re.compile(r'youtube\.com/user/([\w-]+)', re.IGNORECASE)
# ========== 数据库 ==========
def get_db():
"""获取数据库连接"""
os.makedirs(os.path.dirname(DB_PATH), exist_ok=True)
conn = sqlite3.connect(DB_PATH)
conn.execute("""
CREATE TABLE IF NOT EXISTS articles (
id INTEGER PRIMARY KEY AUTOINCREMENT,
feed_url TEXT NOT NULL,
channel_title TEXT,
title TEXT NOT NULL,
link TEXT UNIQUE NOT NULL,
description TEXT,
pub_date TEXT,
is_read INTEGER DEFAULT 0,
fetched_at TEXT DEFAULT CURRENT_TIMESTAMP
)
""")
conn.execute("""
CREATE TABLE IF NOT EXISTS subscriptions (
id INTEGER PRIMARY KEY AUTOINCREMENT,
name TEXT NOT NULL,
feed_url TEXT UNIQUE NOT NULL,
enabled INTEGER DEFAULT 1,
created_at TEXT DEFAULT CURRENT_TIMESTAMP
)
""")
return conn
def is_article_new(conn, link):
"""检查文章是否已存在"""
cursor = conn.execute("SELECT is_read FROM articles WHERE link = ?", (link,))
row = cursor.fetchone()
return row is None
def save_article(conn, feed_url, channel_title, title, link, description, pub_date):
"""保存文章到数据库"""
try:
conn.execute("""
INSERT OR IGNORE INTO articles
(feed_url, channel_title, title, link, description, pub_date)
VALUES (?, ?, ?, ?, ?, ?)
""", (feed_url, channel_title, title, link, description, pub_date))
conn.commit()
except Exception as e:
print(f" ⚠️ 保存失败: {e}")
def mark_all_read(conn, feed_url=None):
"""标记所有文章为已读"""
if feed_url:
conn.execute("UPDATE articles SET is_read = 1 WHERE feed_url = ?", (feed_url,))
else:
conn.execute("UPDATE articles SET is_read = 1")
conn.commit()
# ========== RSS 获取 ==========
def build_fetch_url(url):
"""根据 URL 类型决定获取方式:
- YouTube channel/user/feed URL → 路由到 RSSHub
- RSSHub URL → 直接使用
- 其他 → 直接访问(绕过 RSSHub /rss/ 路由不稳定)
"""
if url.startswith(RSSHUB_BASE):
return url
# YouTube channel
m = YOUTUBE_CHANNEL_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
# YouTube feed
m = YOUTUBE_FEED_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
# YouTube user
m = YOUTUBE_USER_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}"
return url
def convert_to_stored_url(url):
"""将用户输入的 YouTube URL 转换为 RSSHub 存储格式"""
if url.startswith(RSSHUB_BASE):
return url
m = YOUTUBE_CHANNEL_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
m = YOUTUBE_FEED_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
m = YOUTUBE_USER_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}"
return url
def fetch_rss(url, timeout=15):
"""获取并解析 RSS feed"""
ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE
fetch_url = build_fetch_url(url)
req = Request(fetch_url, headers={'User-Agent': 'Mozilla/5.0 (compatible; Blogwatcher/1.0)'})
try:
with urlopen(req, timeout=timeout, context=ctx) as response:
content = response.read().decode('utf-8', errors='replace')
return parse_rss(content, fetch_url)
except (URLError, HTTPError, Exception) as e:
print(f" ❌ 获取失败: {e}")
return None, []
def parse_rss(xml_content, fetch_url):
"""解析 RSS/Atom XML(支持 RSS 1.0/2.0/Atom)"""
parsed = feedparser.parse(xml_content, response_headers={'fetch_url': fetch_url})
channel_title = parsed.feed.get('title', 'Unknown') if parsed.feed else 'Unknown'
items = []
for entry in parsed.entries[:20]: # 最多取20篇
# 提取 link
link_text = ''
if hasattr(entry, 'link') and entry.link:
link_text = entry.link
elif hasattr(entry, 'id') and entry.id:
link_text = entry.id
# 提取 title
title_text = entry.get('title', 'No Title') or 'No Title'
# 提取 description/summary
desc_text = ''
if hasattr(entry, 'summary') and entry.summary:
desc_text = re.sub(r'<[^>]+>', '', entry.summary)
desc_text = unescape(desc_text)
desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200]
elif hasattr(entry, 'description') and entry.description:
desc_text = re.sub(r'<[^>]+>', '', entry.description)
desc_text = unescape(desc_text)
desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200]
# 提取 pubDate
pub_text = ''
if hasattr(entry, 'published') and entry.published:
pub_text = entry.published
elif hasattr(entry, 'updated') and entry.updated:
pub_text = entry.updated
items.append({
'title': title_text,
'link': link_text,
'description': desc_text,
'pub_date': pub_text
})
return channel_title, items
# ========== 订阅管理 ==========
def load_subscriptions():
"""从文件加载订阅列表"""
subs = []
if os.path.exists(SUBSCRIPTIONS_FILE):
with open(SUBSCRIPTIONS_FILE, 'r') as f:
for line in f:
line = line.strip()
if line and not line.startswith('#'):
parts = line.split('|')
if len(parts) >= 2:
name = parts[0].strip()
url = parts[1].strip()
enabled = parts[2].strip() != '0' if len(parts) > 2 else True
if enabled:
subs.append({'name': name, 'url': url})
return subs
def save_subscription(name, url):
"""添加订阅到文件"""
# 检查是否已存在
subs = load_subscriptions()
for sub in subs:
if sub['url'] == url:
print(f"⚠️ 订阅已存在: {name}")
return False
with open(SUBSCRIPTIONS_FILE, 'a') as f:
f.write(f"{name}|{url}\n")
print(f"✅ 已添加订阅: {name}")
return True
def list_subscriptions():
"""列出所有订阅"""
subs = load_subscriptions()
if not subs:
print("📭 暂无订阅")
return
print(f"\n📡 当前订阅 ({len(subs)} 个):\n")
for i, sub in enumerate(subs, 1):
print(f" [{i}] {sub['name']}")
print(f" {sub['url']}\n")
def find_subscription(identifier):
"""按序号(--list 中的 1-based 索引)或名称定位订阅。
匹配顺序:数字 → 序号;否则 精确 → 忽略大小写精确 → 忽略大小写包含。
多条匹配时打印候选并返回 None。
"""
subs = load_subscriptions()
if not subs:
print("❌ 订阅列表为空", file=sys.stderr)
return None
if identifier.isdigit():
idx = int(identifier)
if 1 <= idx <= len(subs):
return subs[idx - 1]
for sub in subs:
if sub['name'] == identifier:
return sub
lower = identifier.lower()
exact_ci = [s for s in subs if s['name'].lower() == lower]
if len(exact_ci) == 1:
return exact_ci[0]
if len(exact_ci) > 1:
print(f"❌ 忽略大小写下匹配到多个订阅: {[s['name'] for s in exact_ci]}", file=sys.stderr)
return None
substr = [s for s in subs if lower in s['name'].lower()]
if len(substr) == 1:
return substr[0]
if len(substr) > 1:
print(f"❌ 匹配到多个订阅: {[s['name'] for s in substr]}", file=sys.stderr)
return None
print(f"❌ 未找到订阅: {identifier}", file=sys.stderr)
return None
# ========== 主流程 ==========
def _scan_one(conn, sub, force_all=False):
"""扫描单个订阅(内部辅助)。返回 (channel_title 或 None, 抓取总数, 新增数量)。"""
channel_title, items = fetch_rss(sub['url'])
if items is None:
return None, 0, 0
items_to_save = items[:10] if force_all else [i for i in items[:10] if is_article_new(conn, i['link'])]
for item in items_to_save:
save_article(conn, sub['url'], channel_title, item['title'], item['link'],
item['description'], item['pub_date'])
return channel_title, len(items), len(items_to_save)
def scan_all(force_all=False):
"""扫描所有订阅
force_all: True 则忽略已读状态,每个频道强制抓10篇
"""
print("=" * 50)
print("Blogwatcher Daily Scan")
print("=" * 50)
subs = load_subscriptions()
if not subs:
print("\n📭 暂无订阅,请先添加:")
print(f" python3 {__file__} --add \"频道名\" \"RSS URL\"")
return
conn = get_db()
new_count = 0
print(f"\n📡 开始扫描 {len(subs)} 个订阅...\n")
for sub in subs:
print(f"🔍 扫描: {sub['name']}")
channel_title, total, new_in_feed = _scan_one(conn, sub, force_all=force_all)
if channel_title is None:
print(f" ⏭️ 跳过\n")
continue
new_count += new_in_feed
print(f" ✅ {channel_title}: {total} 篇, 新增 {new_in_feed} 篇\n")
conn.close()
print("-" * 50)
print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n")
if new_count == 0:
print("📭 今日无新文章")
def scan_subscription(sub, force_all=False):
print("=" * 50)
print(f"Blogwatcher — 更新单个订阅: {sub['name']}")
print("=" * 50)
conn = get_db()
print(f"\n🔍 扫描: {sub['name']}")
channel_title, total, new_in_feed = _scan_one(conn, sub, force_all=force_all)
conn.close()
if channel_title is None:
print(f" ⏭️ 跳过\n")
print("-" * 50)
print("📊 扫描完成: 未获取到新文章")
return
print(f" ✅ {channel_title}: {total} 篇, 新增 {new_in_feed} 篇\n")
print("-" * 50)
print(f"📊 扫描完成: 新增 {new_in_feed} 篇文章")
# ========== 文章管理 ==========
def _fetch_articles_for_feed(conn, feed_url, unread_only=False):
conn.row_factory = sqlite3.Row
query = """
SELECT id, title, link, description, pub_date, is_read, fetched_at
FROM articles
WHERE feed_url = ?
"""
params = [feed_url]
if unread_only:
query += " AND is_read = 0"
query += " ORDER BY fetched_at DESC, id DESC"
return conn.execute(query, params).fetchall()
def list_articles_for_subscription(sub, unread_only=False):
conn = get_db()
rows = _fetch_articles_for_feed(conn, sub['url'], unread_only=unread_only)
conn.close()
label = "未读" if unread_only else "全部"
print(f"\n📄 [{sub['name']}] {label}文章 ({len(rows)} 篇):\n")
if not rows:
print(" (无文章)")
return
for i, row in enumerate(rows, 1):
mark = "🆕" if not row['is_read'] else "📖"
title = (row['title'] or 'Untitled').strip()
print(f" [{i}] {mark} {title}")
if row['link']:
print(f" 🔗 {row['link']}")
if row['pub_date']:
print(f" 📅 {row['pub_date']}")
print()
def delete_articles_for_subscription(sub, article_index=None):
"""删除该订阅下文章。
article_index=None → 删除全部;否则按 --articles 显示顺序删除第 article_index 篇(1-based)。
返回 True 表示删除操作成功(或空库时的幂等 no-op),False 表示失败(如序号越界)。
"""
conn = get_db()
if article_index is None:
count = conn.execute(
"SELECT COUNT(*) FROM articles WHERE feed_url = ?", (sub['url'],)
).fetchone()[0]
if count == 0:
print(f"📭 [{sub['name']}] 无文章可删除")
conn.close()
return True
conn.execute("DELETE FROM articles WHERE feed_url = ?", (sub['url'],))
conn.commit()
conn.close()
print(f"🗑️ 已删除 [{sub['name']}] 全部 {count} 篇文章")
return True
rows = _fetch_articles_for_feed(conn, sub['url'])
if article_index < 1 or article_index > len(rows):
print(f"❌ 无效序号 {article_index},[{sub['name']}] 当前有 {len(rows)} 篇文章", file=sys.stderr)
conn.close()
return False
victim = rows[article_index - 1]
conn.execute("DELETE FROM articles WHERE id = ?", (victim['id'],))
conn.commit()
conn.close()
print(f"🗑️ 已删除 [{sub['name']}] 第 {article_index} 篇: {(victim['title'] or 'Untitled').strip()}")
return True
# ========== 导出 ==========
def fetch_articles(target_date=None, limit=None):
"""从 articles 表读取;可选按本地日期过滤(date(fetched_at,'localtime'))+ 可选限制数量"""
if not os.path.exists(DB_PATH):
print(f"❌ 数据库不存在: {DB_PATH}", file=sys.stderr)
return []
conn = sqlite3.connect(DB_PATH)
conn.row_factory = sqlite3.Row
where = ""
params = []
if target_date:
where = "WHERE date(fetched_at, 'localtime') = ?"
params.append(target_date)
query = f"""
SELECT channel_title, title, link, description, pub_date, fetched_at
FROM articles
{where}
ORDER BY fetched_at DESC, channel_title
"""
if limit:
query += " LIMIT ?"
params.append(limit)
cursor = conn.execute(query, params)
rows = [dict(r) for r in cursor.fetchall()]
conn.close()
return rows
def _group_by_channel(articles):
grouped = {}
for a in articles:
ch = a["channel_title"] or "Unknown"
grouped.setdefault(ch, []).append(a)
return grouped
def format_txt(articles, label):
lines = [f"Blogwatcher Daily — {label}", "=" * 50, ""]
if not articles:
lines.append("(无文章)")
return "\n".join(lines) + "\n"
lines.append(f"共 {len(articles)} 篇文章")
lines.append("")
for channel, items in _group_by_channel(articles).items():
lines.append(f"[{channel}]")
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
lines.append(f" - {title}")
if link:
lines.append(f" {link}")
desc = (it["description"] or "").strip()
if desc:
lines.append(f" {desc[:200]}")
lines.append("")
return "\n".join(lines) + "\n"
def format_markdown(articles, label):
md = f"# Blogwatcher Daily — {label}\n\n"
if not articles:
return md + "_无文章_\n"
md += f"共 **{len(articles)}** 篇文章。\n\n---\n\n"
for channel, items in _group_by_channel(articles).items():
md += f"## 【{channel}】\n\n"
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
md += f"- [{title}]({link})\n"
desc = (it["description"] or "").strip()
if desc:
md += f" > {desc[:200]}\n"
md += "\n"
return md
def format_html(articles, label):
parts = [
"<!DOCTYPE html>",
'<html lang="zh-CN"><head><meta charset="utf-8">',
f"<title>Blogwatcher Daily — {escape(label)}</title>",
"<style>body{font-family:-apple-system,sans-serif;max-width:800px;"
"margin:2em auto;padding:0 1em;line-height:1.5}"
"h2{border-bottom:1px solid #ddd;padding-bottom:0.3em;margin-top:2em}"
"blockquote{color:#555;border-left:3px solid #ccc;margin:0.3em 0 0.6em;"
"padding:0.2em 1em}"
"li{margin-bottom:0.5em}</style>",
"</head><body>",
f"<h1>Blogwatcher Daily — {escape(label)}</h1>",
]
if not articles:
parts.append("<p><em>无文章</em></p></body></html>")
return "\n".join(parts) + "\n"
parts.append(f"<p>共 <strong>{len(articles)}</strong> 篇文章。</p><hr>")
for channel, items in _group_by_channel(articles).items():
parts.append(f"<h2>【{escape(channel)}】</h2>\n<ul>")
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
parts.append(f' <li><a href="{escape(link)}">{escape(title)}</a>')
desc = (it["description"] or "").strip()
if desc:
parts.append(f" <blockquote>{escape(desc[:200])}</blockquote>")
parts.append(" </li>")
parts.append("</ul>")
parts.append("</body></html>")
return "\n".join(parts) + "\n"
FORMATTERS = {
"txt": format_txt,
"markdown": format_markdown,
"html": format_html,
}
def export_articles(fmt, target_date, limit):
articles = fetch_articles(target_date, limit)
if target_date:
label = target_date
elif limit:
label = f"最新 {limit} 篇"
else:
label = "全部"
print(f"📊 {len(articles)} 篇文章 → {fmt}", file=sys.stderr)
sys.stdout.write(FORMATTERS[fmt](articles, label))
# ========== CLI ==========
def main():
parser = argparse.ArgumentParser(description='Blogwatcher Daily RSS 监控脚本')
parser.add_argument('--list', '-l', action='store_true', help='列出所有订阅')
parser.add_argument('--add', nargs=2, metavar=('NAME', 'URL'), help='添加订阅')
parser.add_argument('--update', metavar='SUB',
help='更新指定订阅(--list 序号或名称),拉取最新文章入库;可搭配 --all')
parser.add_argument('--articles', metavar='SUB',
help='列出指定订阅的文章;可搭配 --unread 只看未读')
parser.add_argument('--unread', action='store_true',
help='与 --articles 搭配:只列出未读文章')
parser.add_argument('--mark-read', nargs='?', const='__ALL__', metavar='SUB',
help='标记为已读:不带参数=全库;带 SUB=只标记指定订阅')
parser.add_argument('--delete', metavar='SUB',
help='删除指定订阅下的所有文章;可搭配 --article-id 删除单篇')
parser.add_argument('--article-id', type=int, metavar='N',
help='与 --delete 搭配:删除该订阅下第 N 篇(1-based,顺序同 --articles)')
parser.add_argument('--rsshub', help='设置 RSSHub 地址')
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇')
parser.add_argument('--export', choices=['txt', 'markdown', 'html'],
help='导出文章为指定格式,写到 stdout(可搭配 --date/--limit)')
parser.add_argument('--date', help='导出的目标日期 YYYY-MM-DD(默认今天,仅 --export 有效)')
parser.add_argument('--limit', type=int,
help='导出的文章数上限;单独使用时忽略日期取全库最新 N 篇(仅 --export 有效)')
args = parser.parse_args()
if args.unread and not args.articles:
parser.error("--unread 必须与 --articles 一起使用")
if args.article_id is not None and not args.delete:
parser.error("--article-id 必须与 --delete 一起使用")
global RSSHUB_BASE
if args.rsshub:
RSSHUB_BASE = args.rsshub
if args.list:
list_subscriptions()
elif args.add:
name, url = args.add
if not url.startswith('http'):
url = f"{RSSHUB_BASE}/{url}"
stored_url = convert_to_stored_url(url)
save_subscription(name, stored_url)
elif args.update:
sub = find_subscription(args.update)
if not sub:
sys.exit(1)
scan_subscription(sub, force_all=args.all)
elif args.articles:
sub = find_subscription(args.articles)
if not sub:
sys.exit(1)
list_articles_for_subscription(sub, unread_only=args.unread)
elif args.mark_read is not None:
if args.mark_read == '__ALL__':
conn = get_db()
mark_all_read(conn)
conn.close()
print("✅ 已标记所有文章为已读")
else:
sub = find_subscription(args.mark_read)
if not sub:
sys.exit(1)
conn = get_db()
mark_all_read(conn, feed_url=sub['url'])
conn.close()
print(f"✅ 已标记 [{sub['name']}] 所有文章为已读")
elif args.delete:
sub = find_subscription(args.delete)
if not sub:
sys.exit(1)
if not delete_articles_for_subscription(sub, article_index=args.article_id):
sys.exit(1)
elif args.export:
if args.date:
target_date = args.date
elif args.limit:
target_date = None
else:
target_date = date.today().isoformat()
export_articles(args.export, target_date, args.limit)
else:
scan_all(force_all=args.all)
if __name__ == "__main__":
main()