diff --git a/sreweekly-digest/scripts/sreweekly.py b/sreweekly-digest/scripts/sreweekly.py index 5d43b12..a644058 100644 --- a/sreweekly-digest/scripts/sreweekly.py +++ b/sreweekly-digest/scripts/sreweekly.py @@ -43,7 +43,7 @@ SRE Weekly 抓取与文章提取脚本 pip3 install feedparser trafilatura """ from html.parser import HTMLParser -import argparse, html, json, os, re, sys, time, urllib.request, urllib.error +import argparse, html, http.client, json, os, re, sys, time, urllib.error, urllib.parse, urllib.request try: import feedparser @@ -390,12 +390,16 @@ def fetch_article_html(url, cache_path, force=False): if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0: with open(cache_path, "r", encoding="utf-8", errors="replace") as f: return f.read(), None - req = urllib.request.Request(url, headers={ - "User-Agent": DEFAULT_UA, - "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", - "Accept-Language": "en-US,en;q=0.9", - }) try: + safe_url = urllib.parse.quote(url, safe=":/?#[]@!$&'()*+,;=%~") + except (UnicodeError, TypeError) as e: + return None, f"url sanitize failed: {e}" + try: + req = urllib.request.Request(safe_url, headers={ + "User-Agent": DEFAULT_UA, + "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", + "Accept-Language": "en-US,en;q=0.9", + }) with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r: ctype = r.headers.get("Content-Type", "") if "html" not in ctype.lower() and "xml" not in ctype.lower(): @@ -410,7 +414,7 @@ def fetch_article_html(url, cache_path, force=False): return None, f"HTTP {e.code}" except urllib.error.URLError as e: return None, f"URLError: {e.reason}" - except (TimeoutError, OSError) as e: + except (TimeoutError, OSError, http.client.HTTPException, ValueError, UnicodeError) as e: return None, f"{type(e).__name__}: {e}" os.makedirs(os.path.dirname(cache_path), exist_ok=True) with open(cache_path, "w", encoding="utf-8") as f: @@ -486,6 +490,7 @@ def extract_one(iid, entry, base_dir, force=False, fetch=True, force_fetch=False f.write(content) if fetch and fetch_log: + os.makedirs(art_cache_dir, exist_ok=True) with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f: json.dump(fetch_log, f, ensure_ascii=False, indent=2) @@ -558,7 +563,7 @@ def cmd_extract(args, manifest, base_dir): force=args.force, fetch=fetch, force_fetch=args.force_fetch, sleep=args.sleep, ) - save_manifest(base_dir, manifest) + save_manifest(base_dir, manifest) # ---------------------------------------------------------------- 展示