添加 blogwatcher-daily 目录内容

This commit is contained in:
2026-08-29 20:12:35 +08:00
parent 4579e36f12
commit c0679bb3e8
5 changed files with 708 additions and 0 deletions

View File

@@ -0,0 +1,436 @@
#!/usr/bin/env python3
"""
Blogwatcher Daily - RSS Feed 监控脚本
放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py
Usage:
python3 blogwatcher-daily.py # 扫描所有订阅,写入今日笔记
python3 blogwatcher-daily.py --list # 列出所有订阅
python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅
python3 blogwatcher-daily.py --scan-only # 只扫描不写入文件
python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读
"""
import os
import re
import sqlite3
import argparse
import feedparser
from datetime import datetime
from html import unescape
from urllib.request import Request, urlopen
from urllib.error import URLError, HTTPError
import ssl
# ========== 配置 ==========
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
SKILL_DIR = os.path.dirname(SCRIPT_DIR) # .../research/blogwatcher-daily
DB_PATH = os.path.join(SKILL_DIR, "blogwatcher.db")
SUBSCRIPTIONS_FILE = os.path.join(SKILL_DIR, "subscriptions.txt")
OUTPUT_DIR = os.path.expanduser("~/Workspace/nexus/ishenwei/blogwatcher")
# RSSHub 基础地址(可配置)
RSSHUB_BASE = os.environ.get("RSSHUB_URL", "http://192.168.3.45:1200")
# YouTube channel ID 提取正则
YOUTUBE_CHANNEL_RE = re.compile(r'youtube\.com/channel/([\w-]+)', re.IGNORECASE)
YOUTUBE_FEED_RE = re.compile(r'youtube\.com/feeds/videos\.xml\?channel_id=([\w-]+)', re.IGNORECASE)
YOUTUBE_USER_RE = re.compile(r'youtube\.com/user/([\w-]+)', re.IGNORECASE)
# ========== 数据库 ==========
def get_db():
"""获取数据库连接"""
os.makedirs(os.path.dirname(DB_PATH), exist_ok=True)
conn = sqlite3.connect(DB_PATH)
conn.execute("""
CREATE TABLE IF NOT EXISTS articles (
id INTEGER PRIMARY KEY AUTOINCREMENT,
feed_url TEXT NOT NULL,
channel_title TEXT,
title TEXT NOT NULL,
link TEXT UNIQUE NOT NULL,
description TEXT,
pub_date TEXT,
is_read INTEGER DEFAULT 0,
fetched_at TEXT DEFAULT CURRENT_TIMESTAMP
)
""")
conn.execute("""
CREATE TABLE IF NOT EXISTS subscriptions (
id INTEGER PRIMARY KEY AUTOINCREMENT,
name TEXT NOT NULL,
feed_url TEXT UNIQUE NOT NULL,
enabled INTEGER DEFAULT 1,
created_at TEXT DEFAULT CURRENT_TIMESTAMP
)
""")
return conn
def is_article_new(conn, link):
"""检查文章是否已存在"""
cursor = conn.execute("SELECT is_read FROM articles WHERE link = ?", (link,))
row = cursor.fetchone()
return row is None
def save_article(conn, feed_url, channel_title, title, link, description, pub_date):
"""保存文章到数据库"""
try:
conn.execute("""
INSERT OR IGNORE INTO articles
(feed_url, channel_title, title, link, description, pub_date)
VALUES (?, ?, ?, ?, ?, ?)
""", (feed_url, channel_title, title, link, description, pub_date))
conn.commit()
except Exception as e:
print(f" ⚠️ 保存失败: {e}")
def mark_all_read(conn, feed_url=None):
"""标记所有文章为已读"""
if feed_url:
conn.execute("UPDATE articles SET is_read = 1 WHERE feed_url = ?", (feed_url,))
else:
conn.execute("UPDATE articles SET is_read = 1")
conn.commit()
# ========== RSS 获取 ==========
def build_fetch_url(url):
"""根据 URL 类型决定获取方式:
- YouTube channel/user/feed URL → 路由到 RSSHub
- RSSHub URL → 直接使用
- 其他 → 直接访问(绕过 RSSHub /rss/ 路由不稳定)
"""
if url.startswith(RSSHUB_BASE):
return url
# YouTube channel
m = YOUTUBE_CHANNEL_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
# YouTube feed
m = YOUTUBE_FEED_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
# YouTube user
m = YOUTUBE_USER_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}"
return url
def convert_to_stored_url(url):
"""将用户输入的 YouTube URL 转换为 RSSHub 存储格式"""
if url.startswith(RSSHUB_BASE):
return url
m = YOUTUBE_CHANNEL_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
m = YOUTUBE_FEED_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/channel/{m.group(1)}"
m = YOUTUBE_USER_RE.search(url)
if m:
return f"{RSSHUB_BASE}/youtube/user/{m.group(1)}"
return url
def fetch_rss(url, timeout=15):
"""获取并解析 RSS feed"""
ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE
fetch_url = build_fetch_url(url)
req = Request(fetch_url, headers={'User-Agent': 'Mozilla/5.0 (compatible; Blogwatcher/1.0)'})
try:
with urlopen(req, timeout=timeout, context=ctx) as response:
content = response.read().decode('utf-8', errors='replace')
return parse_rss(content, fetch_url)
except (URLError, HTTPError, Exception) as e:
print(f" ❌ 获取失败: {e}")
return None, []
def parse_rss(xml_content, fetch_url):
"""解析 RSS/Atom XML(支持 RSS 1.0/2.0/Atom)"""
parsed = feedparser.parse(xml_content, response_headers={'fetch_url': fetch_url})
channel_title = parsed.feed.get('title', 'Unknown') if parsed.feed else 'Unknown'
items = []
for entry in parsed.entries[:20]: # 最多取20篇
# 提取 link
link_text = ''
if hasattr(entry, 'link') and entry.link:
link_text = entry.link
elif hasattr(entry, 'id') and entry.id:
link_text = entry.id
# 提取 title
title_text = entry.get('title', 'No Title') or 'No Title'
# 提取 description/summary
desc_text = ''
if hasattr(entry, 'summary') and entry.summary:
desc_text = re.sub(r'<[^>]+>', '', entry.summary)
desc_text = unescape(desc_text)
desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200]
elif hasattr(entry, 'description') and entry.description:
desc_text = re.sub(r'<[^>]+>', '', entry.description)
desc_text = unescape(desc_text)
desc_text = re.sub(r'\s+', ' ', desc_text).strip()[:200]
# 提取 pubDate
pub_text = ''
if hasattr(entry, 'published') and entry.published:
pub_text = entry.published
elif hasattr(entry, 'updated') and entry.updated:
pub_text = entry.updated
items.append({
'title': title_text,
'link': link_text,
'description': desc_text,
'pub_date': pub_text
})
return channel_title, items
# ========== 订阅管理 ==========
def load_subscriptions():
"""从文件加载订阅列表"""
subs = []
if os.path.exists(SUBSCRIPTIONS_FILE):
with open(SUBSCRIPTIONS_FILE, 'r') as f:
for line in f:
line = line.strip()
if line and not line.startswith('#'):
parts = line.split('|')
if len(parts) >= 2:
name = parts[0].strip()
url = parts[1].strip()
enabled = parts[2].strip() != '0' if len(parts) > 2 else True
if enabled:
subs.append({'name': name, 'url': url})
return subs
def save_subscription(name, url):
"""添加订阅到文件"""
# 检查是否已存在
subs = load_subscriptions()
for sub in subs:
if sub['url'] == url:
print(f"⚠️ 订阅已存在: {name}")
return False
with open(SUBSCRIPTIONS_FILE, 'a') as f:
f.write(f"{name}|{url}\n")
print(f"✅ 已添加订阅: {name}")
return True
def list_subscriptions():
"""列出所有订阅"""
subs = load_subscriptions()
if not subs:
print("📭 暂无订阅")
return
print(f"\n📡 当前订阅 ({len(subs)} 个):\n")
for i, sub in enumerate(subs, 1):
print(f" [{i}] {sub['name']}")
print(f" {sub['url']}\n")
# ========== Markdown 输出 ==========
def generate_markdown(results, date_str=None):
"""生成 Markdown 格式"""
if date_str is None:
date_str = datetime.now().strftime("%Y-%m-%d")
md = f"# Blogwatcher Daily - {date_str}\n\n"
md += f"_生成时间: {datetime.now().strftime('%H:%M:%S')}_\n\n"
md += "---\n\n"
if not results:
md += "_今日无新文章_\n"
return md
# 按频道分组
channels = {}
for item in results:
ch = item['channel']
if ch not in channels:
channels[ch] = []
channels[ch].append(item)
for channel, items in channels.items():
md += f"## 【{channel}】\n\n"
for item in items:
md += f"- [{item['title']}]({item['link']})\n"
if item['description']:
md += f" {item['description'][:150]}...\n"
md += "\n"
md += "---\n\n"
return md
def save_daily_report(new_results, date_str=None):
"""保存每日报告(追加模式,避免覆盖)"""
if date_str is None:
date_str = datetime.now().strftime("%Y-%m-%d")
os.makedirs(OUTPUT_DIR, exist_ok=True)
filepath = os.path.join(OUTPUT_DIR, f"{date_str}.md")
# 读取已有内容,提取已有链接避免重复
existing_links = set()
if os.path.exists(filepath):
with open(filepath, 'r') as f:
existing_content = f.read()
# 简单提取已有链接(用于去重),去掉末尾的 ) 等标点
import re as _re
raw_links = _re.findall(r'https?://\S+', existing_content)
existing_links = set()
for l in raw_links:
# 去掉末尾的 ) 等非 URL 字符
while l and l[-1] in '),;:':
l = l[:-1]
existing_links.add(l)
else:
existing_content = ""
# 如果没有新结果,且文件已存在,直接返回
if not new_results and existing_content:
return filepath
# 生成新内容的 header
new_md = ""
new_articles = [a for a in new_results if a['link'] not in existing_links]
if new_articles:
new_md += f"\n## 📦 新增 {len(new_articles)} 篇 ({datetime.now().strftime('%H:%M:%S')})\n\n"
channels = {}
for item in new_articles:
ch = item['channel']
if ch not in channels:
channels[ch] = []
channels[ch].append(item)
for channel, items in channels.items():
new_md += f"### 【{channel}】\n\n"
for item in items:
new_md += f"- [{item['title']}]({item['link']})\n"
if item['description']:
new_md += f" {item['description'][:150]}...\n"
new_md += "\n"
# 追加写入
with open(filepath, 'a') as f:
f.write(new_md)
return filepath
# ========== 主流程 ==========
def scan_all(force_all=False, write_file=True):
"""扫描所有订阅
force_all: True 则忽略已读状态,每个频道强制抓10篇
write_file: True 则写入文件
"""
print("=" * 50)
print("Blogwatcher Daily Scan")
print("=" * 50)
subs = load_subscriptions()
if not subs:
print("\n📭 暂无订阅,请先添加:")
print(f" python3 {__file__} --add \"频道名\" \"RSS URL\"")
return
conn = get_db()
all_new_articles = []
new_count = 0
print(f"\n📡 开始扫描 {len(subs)} 个订阅...\n")
for sub in subs:
print(f"🔍 扫描: {sub['name']}")
channel_title, items = fetch_rss(sub['url'])
if items is None:
print(f" ⏭️ 跳过\n")
continue
new_in_feed = 0
items_to_save = items[:10] if force_all else [item for item in items[:10] if is_article_new(conn, item['link'])]
for item in items_to_save:
save_article(conn, sub['url'], channel_title, item['title'], item['link'],
item['description'], item['pub_date'])
all_new_articles.append({
'channel': channel_title,
**item
})
new_in_feed += 1
new_count += 1
print(f" ✅ {channel_title}: {len(items)} 篇, 新增 {new_in_feed} 篇\n")
conn.close()
# 生成报告
print("-" * 50)
print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n")
if all_new_articles and write_file:
if force_all:
# --all 模式:写入独立文件(不追加日常报告)
md = generate_markdown(all_new_articles)
filepath = os.path.join(OUTPUT_DIR, f"all-{datetime.now().strftime('%Y-%m-%d')}.md")
with open(filepath, 'w') as f:
f.write(md)
print(f"📝 已写入(force-all 模式): {filepath}")
else:
filepath = save_daily_report(all_new_articles)
print(f"📝 已写入: {filepath}")
elif not all_new_articles:
print("📭 今日无新文章")
return all_new_articles
# ========== CLI ==========
def main():
parser = argparse.ArgumentParser(description='Blogwatcher Daily RSS 监控脚本')
parser.add_argument('--list', '-l', action='store_true', help='列出所有订阅')
parser.add_argument('--add', nargs=2, metavar=('NAME', 'URL'), help='添加订阅')
parser.add_argument('--scan-only', action='store_true', help='仅扫描不写入文件')
parser.add_argument('--mark-read', action='store_true', help='标记所有为已读')
parser.add_argument('--date', '-d', help='指定日期 (YYYY-MM-DD)')
parser.add_argument('--rsshub', help='设置 RSSHub 地址')
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇(写入 all-YYYY-MM-DD.md,不追加日常报告)')
args = parser.parse_args()
global RSSHUB_BASE
if args.rsshub:
RSSHUB_BASE = args.rsshub
if args.list:
list_subscriptions()
elif args.add:
name, url = args.add
# 如果 URL 不是完整地址,尝试添加 RSSHub 前缀
if not url.startswith('http'):
url = f"{RSSHUB_BASE}/{url}"
# YouTube URL 自动转为 RSSHub 格式
stored_url = convert_to_stored_url(url)
save_subscription(name, stored_url)
elif args.mark_read:
conn = get_db()
mark_all_read(conn)
conn.close()
print("✅ 已标记所有文章为已读")
elif args.scan_only:
scan_all(force_all=args.all, write_file=False)
else:
scan_all(force_all=args.all, write_file=True)
if __name__ == "__main__":
main()