This commit is contained in:
2026-09-12 14:17:06 +08:00
parent 0c642bbb32
commit f2d9911ff4

View File

@@ -43,7 +43,7 @@ SRE Weekly 抓取与文章提取脚本
pip3 install feedparser trafilatura
"""
from html.parser import HTMLParser
import argparse, html, json, os, re, sys, time, urllib.request, urllib.error
import argparse, html, http.client, json, os, re, sys, time, urllib.error, urllib.parse, urllib.request
try:
import feedparser
@@ -390,12 +390,16 @@ def fetch_article_html(url, cache_path, force=False):
if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0:
with open(cache_path, "r", encoding="utf-8", errors="replace") as f:
return f.read(), None
req = urllib.request.Request(url, headers={
try:
safe_url = urllib.parse.quote(url, safe=":/?#[]@!$&'()*+,;=%~")
except (UnicodeError, TypeError) as e:
return None, f"url sanitize failed: {e}"
try:
req = urllib.request.Request(safe_url, headers={
"User-Agent": DEFAULT_UA,
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.9",
})
try:
with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r:
ctype = r.headers.get("Content-Type", "")
if "html" not in ctype.lower() and "xml" not in ctype.lower():
@@ -410,7 +414,7 @@ def fetch_article_html(url, cache_path, force=False):
return None, f"HTTP {e.code}"
except urllib.error.URLError as e:
return None, f"URLError: {e.reason}"
except (TimeoutError, OSError) as e:
except (TimeoutError, OSError, http.client.HTTPException, ValueError, UnicodeError) as e:
return None, f"{type(e).__name__}: {e}"
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
with open(cache_path, "w", encoding="utf-8") as f:
@@ -486,6 +490,7 @@ def extract_one(iid, entry, base_dir, force=False, fetch=True, force_fetch=False
f.write(content)
if fetch and fetch_log:
os.makedirs(art_cache_dir, exist_ok=True)
with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f:
json.dump(fetch_log, f, ensure_ascii=False, indent=2)