Files
atlas/blogwatcher-daily/scripts/blogwatcher-daily.py
2026-08-29 21:18:59 +08:00

473 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Blogwatcher Daily - RSS Feed 监控脚本
放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py
Usage:
python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库
python3 blogwatcher-daily.py --list # 列出所有订阅
python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅
python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读
python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown
python3 blogwatcher-daily.py --export html --date 2026-08-28
python3 blogwatcher-daily.py --export txt --limit 20
"""
import os
import re
import sys
import sqlite3
import argparse
import feedparser
from datetime import date
from html import escape, unescape
from urllib.request import Request, urlopen
from urllib.error import URLError, HTTPError
import ssl
# ========== 配置 ==========
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
SKILL_DIR = os.path.dirname(SCRIPT_DIR) # .../research/blogwatcher-daily
DB_PATH = os.path.join(SKILL_DIR, "blogwatcher.db")
SUBSCRIPTIONS_FILE = os.path.join(SKILL_DIR, "subscriptions.txt")
# RSSHub 基础地址(可配置)
RSSHUB_BASE = os.environ.get("RSSHUB_URL", "http://192.168.3.45:1200")
# YouTube channel ID 提取正则
YOUTUBE_CHANNEL_RE = re.compile(r'youtube\.com/channel/([\w-]+)', re.IGNORECASE)
YOUTUBE_FEED_RE = re.compile(r'youtube\.com/feeds/videos\.xml\?channel_id=([\w-]+)', re.IGNORECASE)
YOUTUBE_USER_RE = re.compile(r'youtube\.com/user/([\w-]+)', re.IGNORECASE)
# ========== 数据库 ==========
def get_db():
"""获取数据库连接"""
os.makedirs(os.path.dirname(DB_PATH), exist_ok=True)
conn = sqlite3.connect(DB_PATH)
conn.execute("""
CREATE TABLE IF NOT EXISTS articles (
id INTEGER PRIMARY KEY AUTOINCREMENT,
feed_url TEXT NOT NULL,
channel_title TEXT,
title TEXT NOT NULL,
link TEXT UNIQUE NOT NULL,
description TEXT,
pub_date TEXT,
is_read INTEGER DEFAULT 0,
fetched_at TEXT DEFAULT CURRENT_TIMESTAMP
)
""")
conn.execute("""
CREATE TABLE IF NOT EXISTS subscriptions (
id INTEGER PRIMARY KEY AUTOINCREMENT,
name TEXT NOT NULL,
feed_url TEXT UNIQUE NOT NULL,
enabled INTEGER DEFAULT 1,
created_at TEXT DEFAULT CURRENT_TIMESTAMP
)
""")
return conn
def is_article_new(conn, link):
"""检查文章是否已存在"""
cursor = conn.execute("SELECT is_read FROM articles WHERE link = ?", (link,))
row = cursor.fetchone()
return row is None
def save_article(conn, feed_url, channel_title, title, link, description, pub_date):
"""保存文章到数据库"""
try:
conn.execute("""
INSERT OR IGNORE INTO articles
(feed_url, channel_title, title, link, description, pub_date)
VALUES (?, ?, ?, ?, ?, ?)
""", (feed_url, channel_title, title, link, description, pub_date))
conn.commit()
except Exception as e:
print(f" ⚠️ 保存失败: {e}")
def mark_all_read(conn, feed_url=None):
"""标记所有文章为已读"""
if feed_url:
conn.execute("UPDATE articles SET is_read = 1 WHERE feed_url = ?", (feed_url,))
else:
conn.execute("UPDATE articles SET is_read = 1")
conn.commit()
# ========== RSS 获取 ==========
def build_fetch_url(url):
"""根据 URL 类型决定获取方式:
- YouTube channel/user/feed URL → 路由到 RSSHub
- RSSHub URL → 直接使用
- 其他 → 直接访问(绕过 RSSHub /rss/ 路由不稳定)
"""
if url.startswith(RSSHUB_BASE):
return url
# YouTube channel
m = YOUTUBE_CHANNEL_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
# YouTube feed
m = YOUTUBE_FEED_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
# YouTube user
m = YOUTUBE_USER_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}"
return url
def convert_to_stored_url(url):
"""将用户输入的 YouTube URL 转换为 RSSHub 存储格式"""
if url.startswith(RSSHUB_BASE):
return url
m = YOUTUBE_CHANNEL_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
m = YOUTUBE_FEED_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
m = YOUTUBE_USER_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}"
return url
def fetch_rss(url, timeout=15):
"""获取并解析 RSS feed"""
ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE
fetch_url = build_fetch_url(url)
req = Request(fetch_url, headers={'User-Agent': 'Mozilla/5.0 (compatible; Blogwatcher/1.0)'})
try:
with urlopen(req, timeout=timeout, context=ctx) as response:
content = response.read().decode('utf-8', errors='replace')
return parse_rss(content, fetch_url)
except (URLError, HTTPError, Exception) as e:
print(f" ❌ 获取失败: {e}")
return None, []
def parse_rss(xml_content, fetch_url):
"""解析 RSS/Atom XML(支持 RSS 1.0/2.0/Atom)"""
parsed = feedparser.parse(xml_content, response_headers={'fetch_url': fetch_url})
channel_title = parsed.feed.get('title', 'Unknown') if parsed.feed else 'Unknown'
items = []
for entry in parsed.entries[:20]: # 最多取20篇
# 提取 link
link_text = ''
if hasattr(entry, 'link') and entry.link:
link_text = entry.link
elif hasattr(entry, 'id') and entry.id:
link_text = entry.id
# 提取 title
title_text = entry.get('title', 'No Title') or 'No Title'
# 提取 description/summary
desc_text = ''
if hasattr(entry, 'summary') and entry.summary:
desc_text = re.sub(r'<[^>]+>', '', entry.summary)
desc_text = unescape(desc_text)
desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200]
elif hasattr(entry, 'description') and entry.description:
desc_text = re.sub(r'<[^>]+>', '', entry.description)
desc_text = unescape(desc_text)
desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200]
# 提取 pubDate
pub_text = ''
if hasattr(entry, 'published') and entry.published:
pub_text = entry.published
elif hasattr(entry, 'updated') and entry.updated:
pub_text = entry.updated
items.append({
'title': title_text,
'link': link_text,
'description': desc_text,
'pub_date': pub_text
})
return channel_title, items
# ========== 订阅管理 ==========
def load_subscriptions():
"""从文件加载订阅列表"""
subs = []
if os.path.exists(SUBSCRIPTIONS_FILE):
with open(SUBSCRIPTIONS_FILE, 'r') as f:
for line in f:
line = line.strip()
if line and not line.startswith('#'):
parts = line.split('|')
if len(parts) >= 2:
name = parts[0].strip()
url = parts[1].strip()
enabled = parts[2].strip() != '0' if len(parts) > 2 else True
if enabled:
subs.append({'name': name, 'url': url})
return subs
def save_subscription(name, url):
"""添加订阅到文件"""
# 检查是否已存在
subs = load_subscriptions()
for sub in subs:
if sub['url'] == url:
print(f"⚠️ 订阅已存在: {name}")
return False
with open(SUBSCRIPTIONS_FILE, 'a') as f:
f.write(f"{name}|{url}\n")
print(f"✅ 已添加订阅: {name}")
return True
def list_subscriptions():
"""列出所有订阅"""
subs = load_subscriptions()
if not subs:
print("📭 暂无订阅")
return
print(f"\n📡 当前订阅 ({len(subs)} 个):\n")
for i, sub in enumerate(subs, 1):
print(f" [{i}] {sub['name']}")
print(f" {sub['url']}\n")
# ========== 主流程 ==========
def scan_all(force_all=False):
"""扫描所有订阅
force_all: True 则忽略已读状态,每个频道强制抓10篇
"""
print("=" * 50)
print("Blogwatcher Daily Scan")
print("=" * 50)
subs = load_subscriptions()
if not subs:
print("\n📭 暂无订阅,请先添加:")
print(f" python3 {__file__} --add \"频道名\" \"RSS URL\"")
return
conn = get_db()
all_new_articles = []
new_count = 0
print(f"\n📡 开始扫描 {len(subs)} 个订阅...\n")
for sub in subs:
print(f"🔍 扫描: {sub['name']}")
channel_title, items = fetch_rss(sub['url'])
if items is None:
print(f" ⏭️ 跳过\n")
continue
new_in_feed = 0
items_to_save = items[:10] if force_all else [item for item in items[:10] if is_article_new(conn, item['link'])]
for item in items_to_save:
save_article(conn, sub['url'], channel_title, item['title'], item['link'],
item['description'], item['pub_date'])
all_new_articles.append({
'channel': channel_title,
**item
})
new_in_feed += 1
new_count += 1
print(f" ✅ {channel_title}: {len(items)} 篇, 新增 {new_in_feed} 篇\n")
conn.close()
# 生成报告
print("-" * 50)
print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n")
if not all_new_articles:
print("📭 今日无新文章")
return all_new_articles
# ========== 导出 ==========
def fetch_articles(target_date=None, limit=None):
"""从 articles 表读取;可选按本地日期过滤(date(fetched_at,'localtime'))+ 可选限制数量"""
if not os.path.exists(DB_PATH):
print(f"❌ 数据库不存在: {DB_PATH}", file=sys.stderr)
return []
conn = sqlite3.connect(DB_PATH)
conn.row_factory = sqlite3.Row
where = ""
params = []
if target_date:
where = "WHERE date(fetched_at, 'localtime') = ?"
params.append(target_date)
query = f"""
SELECT channel_title, title, link, description, pub_date, fetched_at
FROM articles
{where}
ORDER BY fetched_at DESC, channel_title
"""
if limit:
query += " LIMIT ?"
params.append(limit)
cursor = conn.execute(query, params)
rows = [dict(r) for r in cursor.fetchall()]
conn.close()
return rows
def _group_by_channel(articles):
grouped = {}
for a in articles:
ch = a["channel_title"] or "Unknown"
grouped.setdefault(ch, []).append(a)
return grouped
def format_txt(articles, label):
lines = [f"Blogwatcher Daily — {label}", "=" * 50, ""]
if not articles:
lines.append("(无文章)")
return "\n".join(lines) + "\n"
lines.append(f"共 {len(articles)} 篇文章")
lines.append("")
for channel, items in _group_by_channel(articles).items():
lines.append(f"[{channel}]")
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
lines.append(f" - {title}")
if link:
lines.append(f" {link}")
desc = (it["description"] or "").strip()
if desc:
lines.append(f" {desc[:200]}")
lines.append("")
return "\n".join(lines) + "\n"
def format_markdown(articles, label):
md = f"# Blogwatcher Daily — {label}\n\n"
if not articles:
return md + "_无文章_\n"
md += f"共 **{len(articles)}** 篇文章。\n\n---\n\n"
for channel, items in _group_by_channel(articles).items():
md += f"## 【{channel}】\n\n"
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
md += f"- [{title}]({link})\n"
desc = (it["description"] or "").strip()
if desc:
md += f" > {desc[:200]}\n"
md += "\n"
return md
def format_html(articles, label):
parts = [
"<!DOCTYPE html>",
'<html lang="zh-CN"><head><meta charset="utf-8">',
f"<title>Blogwatcher Daily — {escape(label)}</title>",
"<style>body{font-family:-apple-system,sans-serif;max-width:800px;"
"margin:2em auto;padding:0 1em;line-height:1.5}"
"h2{border-bottom:1px solid #ddd;padding-bottom:0.3em;margin-top:2em}"
"blockquote{color:#555;border-left:3px solid #ccc;margin:0.3em 0 0.6em;"
"padding:0.2em 1em}"
"li{margin-bottom:0.5em}</style>",
"</head><body>",
f"<h1>Blogwatcher Daily — {escape(label)}</h1>",
]
if not articles:
parts.append("<p><em>无文章</em></p></body></html>")
return "\n".join(parts) + "\n"
parts.append(f"<p>共 <strong>{len(articles)}</strong> 篇文章。</p><hr>")
for channel, items in _group_by_channel(articles).items():
parts.append(f"<h2>【{escape(channel)}】</h2>\n<ul>")
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
parts.append(f' <li><a href="{escape(link)}">{escape(title)}</a>')
desc = (it["description"] or "").strip()
if desc:
parts.append(f" <blockquote>{escape(desc[:200])}</blockquote>")
parts.append(" </li>")
parts.append("</ul>")
parts.append("</body></html>")
return "\n".join(parts) + "\n"
FORMATTERS = {
"txt": format_txt,
"markdown": format_markdown,
"html": format_html,
}
def export_articles(fmt, target_date, limit):
articles = fetch_articles(target_date, limit)
if target_date:
label = target_date
elif limit:
label = f"最新 {limit} 篇"
else:
label = "全部"
print(f"📊 {len(articles)} 篇文章 → {fmt}", file=sys.stderr)
sys.stdout.write(FORMATTERS[fmt](articles, label))
# ========== CLI ==========
def main():
parser = argparse.ArgumentParser(description='Blogwatcher Daily RSS 监控脚本')
parser.add_argument('--list', '-l', action='store_true', help='列出所有订阅')
parser.add_argument('--add', nargs=2, metavar=('NAME', 'URL'), help='添加订阅')
parser.add_argument('--mark-read', action='store_true', help='标记所有为已读')
parser.add_argument('--rsshub', help='设置 RSSHub 地址')
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇')
parser.add_argument('--export', choices=['txt', 'markdown', 'html'],
help='导出文章为指定格式,写到 stdout(可搭配 --date/--limit)')
parser.add_argument('--date', help='导出的目标日期 YYYY-MM-DD(默认今天,仅 --export 有效)')
parser.add_argument('--limit', type=int,
help='导出的文章数上限;单独使用时忽略日期取全库最新 N 篇(仅 --export 有效)')
args = parser.parse_args()
global RSSHUB_BASE
if args.rsshub:
RSSHUB_BASE = args.rsshub
if args.list:
list_subscriptions()
elif args.add:
name, url = args.add
# 如果 URL 不是完整地址,尝试添加 RSSHub 前缀
if not url.startswith('http'):
url = f"{RSSHUB_BASE}/{url}"
# YouTube URL 自动转为 RSSHub 格式
stored_url = convert_to_stored_url(url)
save_subscription(name, stored_url)
elif args.mark_read:
conn = get_db()
mark_all_read(conn)
conn.close()
print("✅ 已标记所有文章为已读")
elif args.export:
if args.date:
target_date = args.date
elif args.limit:
target_date = None
else:
target_date = date.today().isoformat()
export_articles(args.export, target_date, args.limit)
else:
scan_all(force_all=args.all)
if __name__ == "__main__":
main()