This commit is contained in:
2026-09-12 14:17:06 +08:00
parent 0c642bbb32
commit f2d9911ff4

View File

@@ -43,7 +43,7 @@ SRE Weekly 抓取与文章提取脚本
pip3 install feedparser trafilatura pip3 install feedparser trafilatura
""" """
from html.parser import HTMLParser from html.parser import HTMLParser
import argparse, html, json, os, re, sys, time, urllib.request, urllib.error import argparse, html, http.client, json, os, re, sys, time, urllib.error, urllib.parse, urllib.request
try: try:
import feedparser import feedparser
@@ -390,12 +390,16 @@ def fetch_article_html(url, cache_path, force=False):
if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0: if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0:
with open(cache_path, "r", encoding="utf-8", errors="replace") as f: with open(cache_path, "r", encoding="utf-8", errors="replace") as f:
return f.read(), None return f.read(), None
req = urllib.request.Request(url, headers={ try:
safe_url = urllib.parse.quote(url, safe=":/?#[]@!$&'()*+,;=%~")
except (UnicodeError, TypeError) as e:
return None, f"url sanitize failed: {e}"
try:
req = urllib.request.Request(safe_url, headers={
"User-Agent": DEFAULT_UA, "User-Agent": DEFAULT_UA,
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.9", "Accept-Language": "en-US,en;q=0.9",
}) })
try:
with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r: with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r:
ctype = r.headers.get("Content-Type", "") ctype = r.headers.get("Content-Type", "")
if "html" not in ctype.lower() and "xml" not in ctype.lower(): if "html" not in ctype.lower() and "xml" not in ctype.lower():
@@ -410,7 +414,7 @@ def fetch_article_html(url, cache_path, force=False):
return None, f"HTTP {e.code}" return None, f"HTTP {e.code}"
except urllib.error.URLError as e: except urllib.error.URLError as e:
return None, f"URLError: {e.reason}" return None, f"URLError: {e.reason}"
except (TimeoutError, OSError) as e: except (TimeoutError, OSError, http.client.HTTPException, ValueError, UnicodeError) as e:
return None, f"{type(e).__name__}: {e}" return None, f"{type(e).__name__}: {e}"
os.makedirs(os.path.dirname(cache_path), exist_ok=True) os.makedirs(os.path.dirname(cache_path), exist_ok=True)
with open(cache_path, "w", encoding="utf-8") as f: with open(cache_path, "w", encoding="utf-8") as f:
@@ -486,6 +490,7 @@ def extract_one(iid, entry, base_dir, force=False, fetch=True, force_fetch=False
f.write(content) f.write(content)
if fetch and fetch_log: if fetch and fetch_log:
os.makedirs(art_cache_dir, exist_ok=True)
with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f: with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f:
json.dump(fetch_log, f, ensure_ascii=False, indent=2) json.dump(fetch_log, f, ensure_ascii=False, indent=2)