#!/usr/bin/env python3 """ Blogwatcher Daily - RSS Feed 监控脚本 放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py Usage: python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库 python3 blogwatcher-daily.py --list # 列出所有订阅 python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅 python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读 python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown python3 blogwatcher-daily.py --export html --date 2026-08-28 python3 blogwatcher-daily.py --export txt --limit 20 """ import os import re import sys import sqlite3 import argparse import feedparser from datetime import date from html import escape, unescape from urllib.request import Request, urlopen from urllib.error import URLError, HTTPError import ssl # ========== 配置 ========== SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) SKILL_DIR = os.path.dirname(SCRIPT_DIR) # .../research/blogwatcher-daily DB_PATH = os.path.join(SKILL_DIR, "blogwatcher.db") SUBSCRIPTIONS_FILE = os.path.join(SKILL_DIR, "subscriptions.txt") # RSSHub 基础地址(可配置) RSSHUB_BASE = os.environ.get("RSSHUB_URL", "http://192.168.3.45:1200") # YouTube channel ID 提取正则 YOUTUBE_CHANNEL_RE = re.compile(r'youtube\.com/channel/([\w-]+)', re.IGNORECASE) YOUTUBE_FEED_RE = re.compile(r'youtube\.com/feeds/videos\.xml\?channel_id=([\w-]+)', re.IGNORECASE) YOUTUBE_USER_RE = re.compile(r'youtube\.com/user/([\w-]+)', re.IGNORECASE) # ========== 数据库 ========== def get_db(): """获取数据库连接""" os.makedirs(os.path.dirname(DB_PATH), exist_ok=True) conn = sqlite3.connect(DB_PATH) conn.execute(""" CREATE TABLE IF NOT EXISTS articles ( id INTEGER PRIMARY KEY AUTOINCREMENT, feed_url TEXT NOT NULL, channel_title TEXT, title TEXT NOT NULL, link TEXT UNIQUE NOT NULL, description TEXT, pub_date TEXT, is_read INTEGER DEFAULT 0, fetched_at TEXT DEFAULT CURRENT_TIMESTAMP ) """) conn.execute(""" CREATE TABLE IF NOT EXISTS subscriptions ( id INTEGER PRIMARY KEY AUTOINCREMENT, name TEXT NOT NULL, feed_url TEXT UNIQUE NOT NULL, enabled INTEGER DEFAULT 1, created_at TEXT DEFAULT CURRENT_TIMESTAMP ) """) return conn def is_article_new(conn, link): """检查文章是否已存在""" cursor = conn.execute("SELECT is_read FROM articles WHERE link = ?", (link,)) row = cursor.fetchone() return row is None def save_article(conn, feed_url, channel_title, title, link, description, pub_date): """保存文章到数据库""" try: conn.execute(""" INSERT OR IGNORE INTO articles (feed_url, channel_title, title, link, description, pub_date) VALUES (?, ?, ?, ?, ?, ?) """, (feed_url, channel_title, title, link, description, pub_date)) conn.commit() except Exception as e: print(f" ⚠️ 保存失败: {e}") def mark_all_read(conn, feed_url=None): """标记所有文章为已读""" if feed_url: conn.execute("UPDATE articles SET is_read = 1 WHERE feed_url = ?", (feed_url,)) else: conn.execute("UPDATE articles SET is_read = 1") conn.commit() # ========== RSS 获取 ========== def build_fetch_url(url): """根据 URL 类型决定获取方式: - YouTube channel/user/feed URL → 路由到 RSSHub - RSSHub URL → 直接使用 - 其他 → 直接访问(绕过 RSSHub /rss/ 路由不稳定) """ if url.startswith(RSSHUB_BASE): return url # YouTube channel m = YOUTUBE_CHANNEL_RE.search(url) if m: return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}" # YouTube feed m = YOUTUBE_FEED_RE.search(url) if m: return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}" # YouTube user m = YOUTUBE_USER_RE.search(url) if m: return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}" return url def convert_to_stored_url(url): """将用户输入的 YouTube URL 转换为 RSSHub 存储格式""" if url.startswith(RSSHUB_BASE): return url m = YOUTUBE_CHANNEL_RE.search(url) if m: return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}" m = YOUTUBE_FEED_RE.search(url) if m: return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}" m = YOUTUBE_USER_RE.search(url) if m: return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}" return url def fetch_rss(url, timeout=15): """获取并解析 RSS feed""" ctx = ssl.create_default_context() ctx.check_hostname = False ctx.verify_mode = ssl.CERT_NONE fetch_url = build_fetch_url(url) req = Request(fetch_url, headers={'User-Agent': 'Mozilla/5.0 (compatible; Blogwatcher/1.0)'}) try: with urlopen(req, timeout=timeout, context=ctx) as response: content = response.read().decode('utf-8', errors='replace') return parse_rss(content, fetch_url) except (URLError, HTTPError, Exception) as e: print(f" ❌ 获取失败: {e}") return None, [] def parse_rss(xml_content, fetch_url): """解析 RSS/Atom XML(支持 RSS 1.0/2.0/Atom)""" parsed = feedparser.parse(xml_content, response_headers={'fetch_url': fetch_url}) channel_title = parsed.feed.get('title', 'Unknown') if parsed.feed else 'Unknown' items = [] for entry in parsed.entries[:20]: # 最多取20篇 # 提取 link link_text = '' if hasattr(entry, 'link') and entry.link: link_text = entry.link elif hasattr(entry, 'id') and entry.id: link_text = entry.id # 提取 title title_text = entry.get('title', 'No Title') or 'No Title' # 提取 description/summary desc_text = '' if hasattr(entry, 'summary') and entry.summary: desc_text = re.sub(r'<[^>]+>', '', entry.summary) desc_text = unescape(desc_text) desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200] elif hasattr(entry, 'description') and entry.description: desc_text = re.sub(r'<[^>]+>', '', entry.description) desc_text = unescape(desc_text) desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200] # 提取 pubDate pub_text = '' if hasattr(entry, 'published') and entry.published: pub_text = entry.published elif hasattr(entry, 'updated') and entry.updated: pub_text = entry.updated items.append({ 'title': title_text, 'link': link_text, 'description': desc_text, 'pub_date': pub_text }) return channel_title, items # ========== 订阅管理 ========== def load_subscriptions(): """从文件加载订阅列表""" subs = [] if os.path.exists(SUBSCRIPTIONS_FILE): with open(SUBSCRIPTIONS_FILE, 'r') as f: for line in f: line = line.strip() if line and not line.startswith('#'): parts = line.split('|') if len(parts) >= 2: name = parts[0].strip() url = parts[1].strip() enabled = parts[2].strip() != '0' if len(parts) > 2 else True if enabled: subs.append({'name': name, 'url': url}) return subs def save_subscription(name, url): """添加订阅到文件""" # 检查是否已存在 subs = load_subscriptions() for sub in subs: if sub['url'] == url: print(f"⚠️ 订阅已存在: {name}") return False with open(SUBSCRIPTIONS_FILE, 'a') as f: f.write(f"{name}|{url}\n") print(f"✅ 已添加订阅: {name}") return True def list_subscriptions(): """列出所有订阅""" subs = load_subscriptions() if not subs: print("📭 暂无订阅") return print(f"\n📡 当前订阅 ({len(subs)} 个):\n") for i, sub in enumerate(subs, 1): print(f" [{i}] {sub['name']}") print(f" {sub['url']}\n") # ========== 主流程 ========== def scan_all(force_all=False): """扫描所有订阅 force_all: True 则忽略已读状态,每个频道强制抓10篇 """ print("=" * 50) print("Blogwatcher Daily Scan") print("=" * 50) subs = load_subscriptions() if not subs: print("\n📭 暂无订阅,请先添加:") print(f" python3 {__file__} --add \"频道名\" \"RSS URL\"") return conn = get_db() all_new_articles = [] new_count = 0 print(f"\n📡 开始扫描 {len(subs)} 个订阅...\n") for sub in subs: print(f"🔍 扫描: {sub['name']}") channel_title, items = fetch_rss(sub['url']) if items is None: print(f" ⏭️ 跳过\n") continue new_in_feed = 0 items_to_save = items[:10] if force_all else [item for item in items[:10] if is_article_new(conn, item['link'])] for item in items_to_save: save_article(conn, sub['url'], channel_title, item['title'], item['link'], item['description'], item['pub_date']) all_new_articles.append({ 'channel': channel_title, **item }) new_in_feed += 1 new_count += 1 print(f" ✅ {channel_title}: {len(items)} 篇, 新增 {new_in_feed} 篇\n") conn.close() # 生成报告 print("-" * 50) print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n") if not all_new_articles: print("📭 今日无新文章") return all_new_articles # ========== 导出 ========== def fetch_articles(target_date=None, limit=None): """从 articles 表读取;可选按本地日期过滤(date(fetched_at,'localtime'))+ 可选限制数量""" if not os.path.exists(DB_PATH): print(f"❌ 数据库不存在: {DB_PATH}", file=sys.stderr) return [] conn = sqlite3.connect(DB_PATH) conn.row_factory = sqlite3.Row where = "" params = [] if target_date: where = "WHERE date(fetched_at, 'localtime') = ?" params.append(target_date) query = f""" SELECT channel_title, title, link, description, pub_date, fetched_at FROM articles {where} ORDER BY fetched_at DESC, channel_title """ if limit: query += " LIMIT ?" params.append(limit) cursor = conn.execute(query, params) rows = [dict(r) for r in cursor.fetchall()] conn.close() return rows def _group_by_channel(articles): grouped = {} for a in articles: ch = a["channel_title"] or "Unknown" grouped.setdefault(ch, []).append(a) return grouped def format_txt(articles, label): lines = [f"Blogwatcher Daily — {label}", "=" * 50, ""] if not articles: lines.append("(无文章)") return "\n".join(lines) + "\n" lines.append(f"共 {len(articles)} 篇文章") lines.append("") for channel, items in _group_by_channel(articles).items(): lines.append(f"[{channel}]") for it in items: title = (it["title"] or "Untitled").strip() link = (it["link"] or "").strip() lines.append(f" - {title}") if link: lines.append(f" {link}") desc = (it["description"] or "").strip() if desc: lines.append(f" {desc[:200]}") lines.append("") return "\n".join(lines) + "\n" def format_markdown(articles, label): md = f"# Blogwatcher Daily — {label}\n\n" if not articles: return md + "_无文章_\n" md += f"共 **{len(articles)}** 篇文章。\n\n---\n\n" for channel, items in _group_by_channel(articles).items(): md += f"## 【{channel}】\n\n" for it in items: title = (it["title"] or "Untitled").strip() link = (it["link"] or "").strip() md += f"- [{title}]({link})\n" desc = (it["description"] or "").strip() if desc: md += f" > {desc[:200]}\n" md += "\n" return md def format_html(articles, label): parts = [ "", '
', f"无文章
") return "\n".join(parts) + "\n" parts.append(f"共 {len(articles)} 篇文章。
{escape(desc[:200])}") parts.append("