修改blogwatcher-daily脚本

This commit is contained in:
2026-08-29 21:18:59 +08:00
parent c0679bb3e8
commit 04001f273a
4 changed files with 230 additions and 166 deletions

View File

@@ -4,20 +4,23 @@ Blogwatcher Daily - RSS Feed 监控脚本
放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py
Usage:
python3 blogwatcher-daily.py # 扫描所有订阅,写入今日笔记
python3 blogwatcher-daily.py --list # 列出所有订阅
python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库
python3 blogwatcher-daily.py --list # 列出所有订阅
python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅
python3 blogwatcher-daily.py --scan-only # 只扫描不写入文件
python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读
python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读
python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown
python3 blogwatcher-daily.py --export html --date 2026-08-28
python3 blogwatcher-daily.py --export txt --limit 20
"""
import os
import re
import sys
import sqlite3
import argparse
import feedparser
from datetime import datetime
from html import unescape
from datetime import date
from html import escape, unescape
from urllib.request import Request, urlopen
from urllib.error import URLError, HTTPError
import ssl
@@ -27,7 +30,6 @@ SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
SKILL_DIR = os.path.dirname(SCRIPT_DIR) # .../research/blogwatcher-daily
DB_PATH = os.path.join(SKILL_DIR, "blogwatcher.db")
SUBSCRIPTIONS_FILE = os.path.join(SKILL_DIR, "subscriptions.txt")
OUTPUT_DIR = os.path.expanduser("~/Workspace/nexus/ishenwei/blogwatcher")
# RSSHub 基础地址(可配置)
RSSHUB_BASE = os.environ.get("RSSHUB_URL", "http://192.168.3.45:1200")
@@ -235,99 +237,10 @@ def list_subscriptions():
print(f" [{i}] {sub['name']}")
print(f" {sub['url']}\n")
# ========== Markdown 输出 ==========
def generate_markdown(results, date_str=None):
"""生成 Markdown 格式"""
if date_str is None:
date_str = datetime.now().strftime("%Y-%m-%d")
md = f"# Blogwatcher Daily - {date_str}\n\n"
md += f"_生成时间: {datetime.now().strftime('%H:%M:%S')}_\n\n"
md += "---\n\n"
if not results:
md += "_今日无新文章_\n"
return md
# 按频道分组
channels = {}
for item in results:
ch = item['channel']
if ch not in channels:
channels[ch] = []
channels[ch].append(item)
for channel, items in channels.items():
md += f"## 【{channel}】\n\n"
for item in items:
md += f"- [{item['title']}]({item['link']})\n"
if item['description']:
md += f" {item['description'][:150]}...\n"
md += "\n"
md += "---\n\n"
return md
def save_daily_report(new_results, date_str=None):
"""保存每日报告(追加模式,避免覆盖)"""
if date_str is None:
date_str = datetime.now().strftime("%Y-%m-%d")
os.makedirs(OUTPUT_DIR, exist_ok=True)
filepath = os.path.join(OUTPUT_DIR, f"{date_str}.md")
# 读取已有内容,提取已有链接避免重复
existing_links = set()
if os.path.exists(filepath):
with open(filepath, 'r') as f:
existing_content = f.read()
# 简单提取已有链接(用于去重),去掉末尾的 ) 等标点
import re as _re
raw_links = _re.findall(r'https?://\S+', existing_content)
existing_links = set()
for l in raw_links:
# 去掉末尾的 ) 等非 URL 字符
while l and l[-1] in '),;:':
l = l[:-1]
existing_links.add(l)
else:
existing_content = ""
# 如果没有新结果,且文件已存在,直接返回
if not new_results and existing_content:
return filepath
# 生成新内容的 header
new_md = ""
new_articles = [a for a in new_results if a['link'] not in existing_links]
if new_articles:
new_md += f"\n## 📦 新增 {len(new_articles)} 篇 ({datetime.now().strftime('%H:%M:%S')})\n\n"
channels = {}
for item in new_articles:
ch = item['channel']
if ch not in channels:
channels[ch] = []
channels[ch].append(item)
for channel, items in channels.items():
new_md += f"### 【{channel}】\n\n"
for item in items:
new_md += f"- [{item['title']}]({item['link']})\n"
if item['description']:
new_md += f" {item['description'][:150]}...\n"
new_md += "\n"
# 追加写入
with open(filepath, 'a') as f:
f.write(new_md)
return filepath
# ========== 主流程 ==========
def scan_all(force_all=False, write_file=True):
def scan_all(force_all=False):
"""扫描所有订阅
force_all: True 则忽略已读状态,每个频道强制抓10篇
write_file: True 则写入文件
"""
print("=" * 50)
print("Blogwatcher Daily Scan")
@@ -375,32 +288,149 @@ def scan_all(force_all=False, write_file=True):
print("-" * 50)
print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n")
if all_new_articles and write_file:
if force_all:
# --all 模式:写入独立文件(不追加日常报告)
md = generate_markdown(all_new_articles)
filepath = os.path.join(OUTPUT_DIR, f"all-{datetime.now().strftime('%Y-%m-%d')}.md")
with open(filepath, 'w') as f:
f.write(md)
print(f"📝 已写入(force-all 模式): {filepath}")
else:
filepath = save_daily_report(all_new_articles)
print(f"📝 已写入: {filepath}")
elif not all_new_articles:
if not all_new_articles:
print("📭 今日无新文章")
return all_new_articles
# ========== 导出 ==========
def fetch_articles(target_date=None, limit=None):
"""从 articles 表读取;可选按本地日期过滤(date(fetched_at,'localtime'))+ 可选限制数量"""
if not os.path.exists(DB_PATH):
print(f"❌ 数据库不存在: {DB_PATH}", file=sys.stderr)
return []
conn = sqlite3.connect(DB_PATH)
conn.row_factory = sqlite3.Row
where = ""
params = []
if target_date:
where = "WHERE date(fetched_at, 'localtime') = ?"
params.append(target_date)
query = f"""
SELECT channel_title, title, link, description, pub_date, fetched_at
FROM articles
{where}
ORDER BY fetched_at DESC, channel_title
"""
if limit:
query += " LIMIT ?"
params.append(limit)
cursor = conn.execute(query, params)
rows = [dict(r) for r in cursor.fetchall()]
conn.close()
return rows
def _group_by_channel(articles):
grouped = {}
for a in articles:
ch = a["channel_title"] or "Unknown"
grouped.setdefault(ch, []).append(a)
return grouped
def format_txt(articles, label):
lines = [f"Blogwatcher Daily — {label}", "=" * 50, ""]
if not articles:
lines.append("(无文章)")
return "\n".join(lines) + "\n"
lines.append(f"共 {len(articles)} 篇文章")
lines.append("")
for channel, items in _group_by_channel(articles).items():
lines.append(f"[{channel}]")
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
lines.append(f" - {title}")
if link:
lines.append(f" {link}")
desc = (it["description"] or "").strip()
if desc:
lines.append(f" {desc[:200]}")
lines.append("")
return "\n".join(lines) + "\n"
def format_markdown(articles, label):
md = f"# Blogwatcher Daily — {label}\n\n"
if not articles:
return md + "_无文章_\n"
md += f"共 **{len(articles)}** 篇文章。\n\n---\n\n"
for channel, items in _group_by_channel(articles).items():
md += f"## 【{channel}】\n\n"
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
md += f"- [{title}]({link})\n"
desc = (it["description"] or "").strip()
if desc:
md += f" > {desc[:200]}\n"
md += "\n"
return md
def format_html(articles, label):
parts = [
"<!DOCTYPE html>",
'<html lang="zh-CN"><head><meta charset="utf-8">',
f"<title>Blogwatcher Daily — {escape(label)}</title>",
"<style>body{font-family:-apple-system,sans-serif;max-width:800px;"
"margin:2em auto;padding:0 1em;line-height:1.5}"
"h2{border-bottom:1px solid #ddd;padding-bottom:0.3em;margin-top:2em}"
"blockquote{color:#555;border-left:3px solid #ccc;margin:0.3em 0 0.6em;"
"padding:0.2em 1em}"
"li{margin-bottom:0.5em}</style>",
"</head><body>",
f"<h1>Blogwatcher Daily — {escape(label)}</h1>",
]
if not articles:
parts.append("<p><em>无文章</em></p></body></html>")
return "\n".join(parts) + "\n"
parts.append(f"<p>共 <strong>{len(articles)}</strong> 篇文章。</p><hr>")
for channel, items in _group_by_channel(articles).items():
parts.append(f"<h2>【{escape(channel)}】</h2>\n<ul>")
for it in items:
title = (it["title"] or "Untitled").strip()
link = (it["link"] or "").strip()
parts.append(f' <li><a href="{escape(link)}">{escape(title)}</a>')
desc = (it["description"] or "").strip()
if desc:
parts.append(f" <blockquote>{escape(desc[:200])}</blockquote>")
parts.append(" </li>")
parts.append("</ul>")
parts.append("</body></html>")
return "\n".join(parts) + "\n"
FORMATTERS = {
"txt": format_txt,
"markdown": format_markdown,
"html": format_html,
}
def export_articles(fmt, target_date, limit):
articles = fetch_articles(target_date, limit)
if target_date:
label = target_date
elif limit:
label = f"最新 {limit} 篇"
else:
label = "全部"
print(f"📊 {len(articles)} 篇文章 → {fmt}", file=sys.stderr)
sys.stdout.write(FORMATTERS[fmt](articles, label))
# ========== CLI ==========
def main():
parser = argparse.ArgumentParser(description='Blogwatcher Daily RSS 监控脚本')
parser.add_argument('--list', '-l', action='store_true', help='列出所有订阅')
parser.add_argument('--add', nargs=2, metavar=('NAME', 'URL'), help='添加订阅')
parser.add_argument('--scan-only', action='store_true', help='仅扫描不写入文件')
parser.add_argument('--mark-read', action='store_true', help='标记所有为已读')
parser.add_argument('--date', '-d', help='指定日期 (YYYY-MM-DD)')
parser.add_argument('--rsshub', help='设置 RSSHub 地址')
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇(写入 all-YYYY-MM-DD.md,不追加日常报告)')
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇')
parser.add_argument('--export', choices=['txt', 'markdown', 'html'],
help='导出文章为指定格式,写到 stdout(可搭配 --date/--limit)')
parser.add_argument('--date', help='导出的目标日期 YYYY-MM-DD(默认今天,仅 --export 有效)')
parser.add_argument('--limit', type=int,
help='导出的文章数上限;单独使用时忽略日期取全库最新 N 篇(仅 --export 有效)')
args = parser.parse_args()
@@ -426,11 +456,17 @@ def main():
conn.close()
print("✅ 已标记所有文章为已读")
elif args.scan_only:
scan_all(force_all=args.all, write_file=False)
elif args.export:
if args.date:
target_date = args.date
elif args.limit:
target_date = None
else:
target_date = date.today().isoformat()
export_articles(args.export, target_date, args.limit)
else:
scan_all(force_all=args.all, write_file=True)
scan_all(force_all=args.all)
if __name__ == "__main__":
main()