Files
nexus/sreweekly/analyze_translate.py

56 lines
1.6 KiB
Python

import os, re, statistics, sys, io
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8', errors='replace')
root = "/Users/weishen/Workspace/nexus/sreweekly/markdown"
files = []
for dp, dn, fn in os.walk(root):
for f in fn:
if f.endswith(".md"):
files.append(os.path.join(dp, f))
print("TOTAL_FILES", len(files))
has_body = no_body = empty_body = 0
body_words = body_chars = 0
existing_zh = 0
short_bodies = 0
per_dir = {}
sizes = []
for p in files:
base = os.path.basename(p)
if re.search(r"-zh\.md$|\.zh\.md$|zh-CN\.md$", base):
existing_zh += 1
continue
try:
txt = open(p, encoding="utf-8").read()
except Exception as e:
continue
m = re.search(r"##\s*正文\s*\n(.*)", txt, re.S)
if not m:
no_body += 1
continue
body = m.group(1).strip()
if len(body) < 50:
empty_body += 1
continue
words = len(re.findall(r"\S+", body))
has_body += 1
body_words += words
body_chars += len(body)
if words < 100:
short_bodies += 1
sizes.append(words)
d = os.path.basename(os.path.dirname(p))
per_dir[d] = per_dir.get(d, 0) + 1
print("HAS_BODY", has_body)
print("NO_BODY", no_body)
print("EMPTY_BODY", empty_body)
print("SHORT_BODY_LT100W", short_bodies)
print("BODY_WORDS", body_words)
print("BODY_CHARS", body_chars)
print("EXISTING_ZH", existing_zh)
print("ISSUE_DIRS_WITH_BODY", len(per_dir))
sizes.sort()
print("MEDIAN_BODY_WORDS", statistics.median(sizes))
print("P90_BODY_WORDS", sizes[int(len(sizes)*0.9)])
print("MAX_BODY_WORDS", sizes[-1])
print("SUM_OVER_5K", sum(1 for s in sizes if s > 5000))