修改blogwatcher-daily脚本
This commit is contained in:
Binary file not shown.
@@ -4,20 +4,23 @@ Blogwatcher Daily - RSS Feed 监控脚本
|
||||
放在 ~/.hermes/skills/research/blogwatcher-daily/scripts/blogwatcher-daily.py
|
||||
|
||||
Usage:
|
||||
python3 blogwatcher-daily.py # 扫描所有订阅,写入今日笔记
|
||||
python3 blogwatcher-daily.py --list # 列出所有订阅
|
||||
python3 blogwatcher-daily.py # 扫描所有订阅,新增文章存入数据库
|
||||
python3 blogwatcher-daily.py --list # 列出所有订阅
|
||||
python3 blogwatcher-daily.py --add "频道名" "RSS URL" # 添加订阅
|
||||
python3 blogwatcher-daily.py --scan-only # 只扫描不写入文件
|
||||
python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读
|
||||
python3 blogwatcher-daily.py --mark-read # 标记所有文章为已读
|
||||
python3 blogwatcher-daily.py --export markdown # 把今天的文章导出为 Markdown
|
||||
python3 blogwatcher-daily.py --export html --date 2026-08-28
|
||||
python3 blogwatcher-daily.py --export txt --limit 20
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import sqlite3
|
||||
import argparse
|
||||
import feedparser
|
||||
from datetime import datetime
|
||||
from html import unescape
|
||||
from datetime import date
|
||||
from html import escape, unescape
|
||||
from urllib.request import Request, urlopen
|
||||
from urllib.error import URLError, HTTPError
|
||||
import ssl
|
||||
@@ -27,7 +30,6 @@ SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
SKILL_DIR = os.path.dirname(SCRIPT_DIR) # .../research/blogwatcher-daily
|
||||
DB_PATH = os.path.join(SKILL_DIR, "blogwatcher.db")
|
||||
SUBSCRIPTIONS_FILE = os.path.join(SKILL_DIR, "subscriptions.txt")
|
||||
OUTPUT_DIR = os.path.expanduser("~/Workspace/nexus/ishenwei/blogwatcher")
|
||||
|
||||
# RSSHub 基础地址(可配置)
|
||||
RSSHUB_BASE = os.environ.get("RSSHUB_URL", "http://192.168.3.45:1200")
|
||||
@@ -235,99 +237,10 @@ def list_subscriptions():
|
||||
print(f" [{i}] {sub['name']}")
|
||||
print(f" {sub['url']}\n")
|
||||
|
||||
# ========== Markdown 输出 ==========
|
||||
def generate_markdown(results, date_str=None):
|
||||
"""生成 Markdown 格式"""
|
||||
if date_str is None:
|
||||
date_str = datetime.now().strftime("%Y-%m-%d")
|
||||
|
||||
md = f"# Blogwatcher Daily - {date_str}\n\n"
|
||||
md += f"_生成时间: {datetime.now().strftime('%H:%M:%S')}_\n\n"
|
||||
md += "---\n\n"
|
||||
|
||||
if not results:
|
||||
md += "_今日无新文章_\n"
|
||||
return md
|
||||
|
||||
# 按频道分组
|
||||
channels = {}
|
||||
for item in results:
|
||||
ch = item['channel']
|
||||
if ch not in channels:
|
||||
channels[ch] = []
|
||||
channels[ch].append(item)
|
||||
|
||||
for channel, items in channels.items():
|
||||
md += f"## 【{channel}】\n\n"
|
||||
for item in items:
|
||||
md += f"- [{item['title']}]({item['link']})\n"
|
||||
if item['description']:
|
||||
md += f" {item['description'][:150]}...\n"
|
||||
md += "\n"
|
||||
md += "---\n\n"
|
||||
|
||||
return md
|
||||
|
||||
def save_daily_report(new_results, date_str=None):
|
||||
"""保存每日报告(追加模式,避免覆盖)"""
|
||||
if date_str is None:
|
||||
date_str = datetime.now().strftime("%Y-%m-%d")
|
||||
|
||||
os.makedirs(OUTPUT_DIR, exist_ok=True)
|
||||
filepath = os.path.join(OUTPUT_DIR, f"{date_str}.md")
|
||||
|
||||
# 读取已有内容,提取已有链接避免重复
|
||||
existing_links = set()
|
||||
if os.path.exists(filepath):
|
||||
with open(filepath, 'r') as f:
|
||||
existing_content = f.read()
|
||||
# 简单提取已有链接(用于去重),去掉末尾的 ) 等标点
|
||||
import re as _re
|
||||
raw_links = _re.findall(r'https?://\S+', existing_content)
|
||||
existing_links = set()
|
||||
for l in raw_links:
|
||||
# 去掉末尾的 ) 等非 URL 字符
|
||||
while l and l[-1] in '),;:':
|
||||
l = l[:-1]
|
||||
existing_links.add(l)
|
||||
else:
|
||||
existing_content = ""
|
||||
|
||||
# 如果没有新结果,且文件已存在,直接返回
|
||||
if not new_results and existing_content:
|
||||
return filepath
|
||||
|
||||
# 生成新内容的 header
|
||||
new_md = ""
|
||||
new_articles = [a for a in new_results if a['link'] not in existing_links]
|
||||
|
||||
if new_articles:
|
||||
new_md += f"\n## 📦 新增 {len(new_articles)} 篇 ({datetime.now().strftime('%H:%M:%S')})\n\n"
|
||||
channels = {}
|
||||
for item in new_articles:
|
||||
ch = item['channel']
|
||||
if ch not in channels:
|
||||
channels[ch] = []
|
||||
channels[ch].append(item)
|
||||
for channel, items in channels.items():
|
||||
new_md += f"### 【{channel}】\n\n"
|
||||
for item in items:
|
||||
new_md += f"- [{item['title']}]({item['link']})\n"
|
||||
if item['description']:
|
||||
new_md += f" {item['description'][:150]}...\n"
|
||||
new_md += "\n"
|
||||
|
||||
# 追加写入
|
||||
with open(filepath, 'a') as f:
|
||||
f.write(new_md)
|
||||
|
||||
return filepath
|
||||
|
||||
# ========== 主流程 ==========
|
||||
def scan_all(force_all=False, write_file=True):
|
||||
def scan_all(force_all=False):
|
||||
"""扫描所有订阅
|
||||
force_all: True 则忽略已读状态,每个频道强制抓10篇
|
||||
write_file: True 则写入文件
|
||||
"""
|
||||
print("=" * 50)
|
||||
print("Blogwatcher Daily Scan")
|
||||
@@ -375,32 +288,149 @@ def scan_all(force_all=False, write_file=True):
|
||||
print("-" * 50)
|
||||
print(f"📊 扫描完成: 共发现 {new_count} 篇新文章\n")
|
||||
|
||||
if all_new_articles and write_file:
|
||||
if force_all:
|
||||
# --all 模式:写入独立文件(不追加日常报告)
|
||||
md = generate_markdown(all_new_articles)
|
||||
filepath = os.path.join(OUTPUT_DIR, f"all-{datetime.now().strftime('%Y-%m-%d')}.md")
|
||||
with open(filepath, 'w') as f:
|
||||
f.write(md)
|
||||
print(f"📝 已写入(force-all 模式): {filepath}")
|
||||
else:
|
||||
filepath = save_daily_report(all_new_articles)
|
||||
print(f"📝 已写入: {filepath}")
|
||||
elif not all_new_articles:
|
||||
if not all_new_articles:
|
||||
print("📭 今日无新文章")
|
||||
|
||||
return all_new_articles
|
||||
|
||||
# ========== 导出 ==========
|
||||
def fetch_articles(target_date=None, limit=None):
|
||||
"""从 articles 表读取;可选按本地日期过滤(date(fetched_at,'localtime'))+ 可选限制数量"""
|
||||
if not os.path.exists(DB_PATH):
|
||||
print(f"❌ 数据库不存在: {DB_PATH}", file=sys.stderr)
|
||||
return []
|
||||
|
||||
conn = sqlite3.connect(DB_PATH)
|
||||
conn.row_factory = sqlite3.Row
|
||||
|
||||
where = ""
|
||||
params = []
|
||||
if target_date:
|
||||
where = "WHERE date(fetched_at, 'localtime') = ?"
|
||||
params.append(target_date)
|
||||
|
||||
query = f"""
|
||||
SELECT channel_title, title, link, description, pub_date, fetched_at
|
||||
FROM articles
|
||||
{where}
|
||||
ORDER BY fetched_at DESC, channel_title
|
||||
"""
|
||||
if limit:
|
||||
query += " LIMIT ?"
|
||||
params.append(limit)
|
||||
|
||||
cursor = conn.execute(query, params)
|
||||
rows = [dict(r) for r in cursor.fetchall()]
|
||||
conn.close()
|
||||
return rows
|
||||
|
||||
def _group_by_channel(articles):
|
||||
grouped = {}
|
||||
for a in articles:
|
||||
ch = a["channel_title"] or "Unknown"
|
||||
grouped.setdefault(ch, []).append(a)
|
||||
return grouped
|
||||
|
||||
def format_txt(articles, label):
|
||||
lines = [f"Blogwatcher Daily — {label}", "=" * 50, ""]
|
||||
if not articles:
|
||||
lines.append("(无文章)")
|
||||
return "\n".join(lines) + "\n"
|
||||
lines.append(f"共 {len(articles)} 篇文章")
|
||||
lines.append("")
|
||||
for channel, items in _group_by_channel(articles).items():
|
||||
lines.append(f"[{channel}]")
|
||||
for it in items:
|
||||
title = (it["title"] or "Untitled").strip()
|
||||
link = (it["link"] or "").strip()
|
||||
lines.append(f" - {title}")
|
||||
if link:
|
||||
lines.append(f" {link}")
|
||||
desc = (it["description"] or "").strip()
|
||||
if desc:
|
||||
lines.append(f" {desc[:200]}")
|
||||
lines.append("")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
def format_markdown(articles, label):
|
||||
md = f"# Blogwatcher Daily — {label}\n\n"
|
||||
if not articles:
|
||||
return md + "_无文章_\n"
|
||||
md += f"共 **{len(articles)}** 篇文章。\n\n---\n\n"
|
||||
for channel, items in _group_by_channel(articles).items():
|
||||
md += f"## 【{channel}】\n\n"
|
||||
for it in items:
|
||||
title = (it["title"] or "Untitled").strip()
|
||||
link = (it["link"] or "").strip()
|
||||
md += f"- [{title}]({link})\n"
|
||||
desc = (it["description"] or "").strip()
|
||||
if desc:
|
||||
md += f" > {desc[:200]}\n"
|
||||
md += "\n"
|
||||
return md
|
||||
|
||||
def format_html(articles, label):
|
||||
parts = [
|
||||
"<!DOCTYPE html>",
|
||||
'<html lang="zh-CN"><head><meta charset="utf-8">',
|
||||
f"<title>Blogwatcher Daily — {escape(label)}</title>",
|
||||
"<style>body{font-family:-apple-system,sans-serif;max-width:800px;"
|
||||
"margin:2em auto;padding:0 1em;line-height:1.5}"
|
||||
"h2{border-bottom:1px solid #ddd;padding-bottom:0.3em;margin-top:2em}"
|
||||
"blockquote{color:#555;border-left:3px solid #ccc;margin:0.3em 0 0.6em;"
|
||||
"padding:0.2em 1em}"
|
||||
"li{margin-bottom:0.5em}</style>",
|
||||
"</head><body>",
|
||||
f"<h1>Blogwatcher Daily — {escape(label)}</h1>",
|
||||
]
|
||||
if not articles:
|
||||
parts.append("<p><em>无文章</em></p></body></html>")
|
||||
return "\n".join(parts) + "\n"
|
||||
parts.append(f"<p>共 <strong>{len(articles)}</strong> 篇文章。</p><hr>")
|
||||
for channel, items in _group_by_channel(articles).items():
|
||||
parts.append(f"<h2>【{escape(channel)}】</h2>\n<ul>")
|
||||
for it in items:
|
||||
title = (it["title"] or "Untitled").strip()
|
||||
link = (it["link"] or "").strip()
|
||||
parts.append(f' <li><a href="{escape(link)}">{escape(title)}</a>')
|
||||
desc = (it["description"] or "").strip()
|
||||
if desc:
|
||||
parts.append(f" <blockquote>{escape(desc[:200])}</blockquote>")
|
||||
parts.append(" </li>")
|
||||
parts.append("</ul>")
|
||||
parts.append("</body></html>")
|
||||
return "\n".join(parts) + "\n"
|
||||
|
||||
FORMATTERS = {
|
||||
"txt": format_txt,
|
||||
"markdown": format_markdown,
|
||||
"html": format_html,
|
||||
}
|
||||
|
||||
def export_articles(fmt, target_date, limit):
|
||||
articles = fetch_articles(target_date, limit)
|
||||
if target_date:
|
||||
label = target_date
|
||||
elif limit:
|
||||
label = f"最新 {limit} 篇"
|
||||
else:
|
||||
label = "全部"
|
||||
print(f"📊 {len(articles)} 篇文章 → {fmt}", file=sys.stderr)
|
||||
sys.stdout.write(FORMATTERS[fmt](articles, label))
|
||||
|
||||
# ========== CLI ==========
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description='Blogwatcher Daily RSS 监控脚本')
|
||||
parser.add_argument('--list', '-l', action='store_true', help='列出所有订阅')
|
||||
parser.add_argument('--add', nargs=2, metavar=('NAME', 'URL'), help='添加订阅')
|
||||
parser.add_argument('--scan-only', action='store_true', help='仅扫描不写入文件')
|
||||
parser.add_argument('--mark-read', action='store_true', help='标记所有为已读')
|
||||
parser.add_argument('--date', '-d', help='指定日期 (YYYY-MM-DD)')
|
||||
parser.add_argument('--rsshub', help='设置 RSSHub 地址')
|
||||
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇(写入 all-YYYY-MM-DD.md,不追加日常报告)')
|
||||
parser.add_argument('--all', action='store_true', help='忽略已读状态,强制抓取每频道最新10篇')
|
||||
parser.add_argument('--export', choices=['txt', 'markdown', 'html'],
|
||||
help='导出文章为指定格式,写到 stdout(可搭配 --date/--limit)')
|
||||
parser.add_argument('--date', help='导出的目标日期 YYYY-MM-DD(默认今天,仅 --export 有效)')
|
||||
parser.add_argument('--limit', type=int,
|
||||
help='导出的文章数上限;单独使用时忽略日期取全库最新 N 篇(仅 --export 有效)')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -426,11 +456,17 @@ def main():
|
||||
conn.close()
|
||||
print("✅ 已标记所有文章为已读")
|
||||
|
||||
elif args.scan_only:
|
||||
scan_all(force_all=args.all, write_file=False)
|
||||
elif args.export:
|
||||
if args.date:
|
||||
target_date = args.date
|
||||
elif args.limit:
|
||||
target_date = None
|
||||
else:
|
||||
target_date = date.today().isoformat()
|
||||
export_articles(args.export, target_date, args.limit)
|
||||
|
||||
else:
|
||||
scan_all(force_all=args.all, write_file=True)
|
||||
scan_all(force_all=args.all)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
Reference in New Issue
Block a user