bug fix
This commit is contained in:
@@ -43,7 +43,7 @@ SRE Weekly 抓取与文章提取脚本
|
|||||||
pip3 install feedparser trafilatura
|
pip3 install feedparser trafilatura
|
||||||
"""
|
"""
|
||||||
from html.parser import HTMLParser
|
from html.parser import HTMLParser
|
||||||
import argparse, html, json, os, re, sys, time, urllib.request, urllib.error
|
import argparse, html, http.client, json, os, re, sys, time, urllib.error, urllib.parse, urllib.request
|
||||||
|
|
||||||
try:
|
try:
|
||||||
import feedparser
|
import feedparser
|
||||||
@@ -390,12 +390,16 @@ def fetch_article_html(url, cache_path, force=False):
|
|||||||
if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0:
|
if not force and os.path.exists(cache_path) and os.path.getsize(cache_path) > 0:
|
||||||
with open(cache_path, "r", encoding="utf-8", errors="replace") as f:
|
with open(cache_path, "r", encoding="utf-8", errors="replace") as f:
|
||||||
return f.read(), None
|
return f.read(), None
|
||||||
req = urllib.request.Request(url, headers={
|
try:
|
||||||
|
safe_url = urllib.parse.quote(url, safe=":/?#[]@!$&'()*+,;=%~")
|
||||||
|
except (UnicodeError, TypeError) as e:
|
||||||
|
return None, f"url sanitize failed: {e}"
|
||||||
|
try:
|
||||||
|
req = urllib.request.Request(safe_url, headers={
|
||||||
"User-Agent": DEFAULT_UA,
|
"User-Agent": DEFAULT_UA,
|
||||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||||
"Accept-Language": "en-US,en;q=0.9",
|
"Accept-Language": "en-US,en;q=0.9",
|
||||||
})
|
})
|
||||||
try:
|
|
||||||
with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r:
|
with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT) as r:
|
||||||
ctype = r.headers.get("Content-Type", "")
|
ctype = r.headers.get("Content-Type", "")
|
||||||
if "html" not in ctype.lower() and "xml" not in ctype.lower():
|
if "html" not in ctype.lower() and "xml" not in ctype.lower():
|
||||||
@@ -410,7 +414,7 @@ def fetch_article_html(url, cache_path, force=False):
|
|||||||
return None, f"HTTP {e.code}"
|
return None, f"HTTP {e.code}"
|
||||||
except urllib.error.URLError as e:
|
except urllib.error.URLError as e:
|
||||||
return None, f"URLError: {e.reason}"
|
return None, f"URLError: {e.reason}"
|
||||||
except (TimeoutError, OSError) as e:
|
except (TimeoutError, OSError, http.client.HTTPException, ValueError, UnicodeError) as e:
|
||||||
return None, f"{type(e).__name__}: {e}"
|
return None, f"{type(e).__name__}: {e}"
|
||||||
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
|
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
|
||||||
with open(cache_path, "w", encoding="utf-8") as f:
|
with open(cache_path, "w", encoding="utf-8") as f:
|
||||||
@@ -486,6 +490,7 @@ def extract_one(iid, entry, base_dir, force=False, fetch=True, force_fetch=False
|
|||||||
f.write(content)
|
f.write(content)
|
||||||
|
|
||||||
if fetch and fetch_log:
|
if fetch and fetch_log:
|
||||||
|
os.makedirs(art_cache_dir, exist_ok=True)
|
||||||
with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f:
|
with open(os.path.join(art_cache_dir, "index.json"), "w", encoding="utf-8") as f:
|
||||||
json.dump(fetch_log, f, ensure_ascii=False, indent=2)
|
json.dump(fetch_log, f, ensure_ascii=False, indent=2)
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user