56 lines
1.6 KiB
Python
56 lines
1.6 KiB
Python
import os, re, statistics, sys, io
|
|
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8', errors='replace')
|
|
root = "/Users/weishen/Workspace/nexus/sreweekly/markdown"
|
|
files = []
|
|
for dp, dn, fn in os.walk(root):
|
|
for f in fn:
|
|
if f.endswith(".md"):
|
|
files.append(os.path.join(dp, f))
|
|
print("TOTAL_FILES", len(files))
|
|
|
|
has_body = no_body = empty_body = 0
|
|
body_words = body_chars = 0
|
|
existing_zh = 0
|
|
short_bodies = 0
|
|
per_dir = {}
|
|
sizes = []
|
|
for p in files:
|
|
base = os.path.basename(p)
|
|
if re.search(r"-zh\.md$|\.zh\.md$|zh-CN\.md$", base):
|
|
existing_zh += 1
|
|
continue
|
|
try:
|
|
txt = open(p, encoding="utf-8").read()
|
|
except Exception as e:
|
|
continue
|
|
m = re.search(r"##\s*正文\s*\n(.*)", txt, re.S)
|
|
if not m:
|
|
no_body += 1
|
|
continue
|
|
body = m.group(1).strip()
|
|
if len(body) < 50:
|
|
empty_body += 1
|
|
continue
|
|
words = len(re.findall(r"\S+", body))
|
|
has_body += 1
|
|
body_words += words
|
|
body_chars += len(body)
|
|
if words < 100:
|
|
short_bodies += 1
|
|
sizes.append(words)
|
|
d = os.path.basename(os.path.dirname(p))
|
|
per_dir[d] = per_dir.get(d, 0) + 1
|
|
|
|
print("HAS_BODY", has_body)
|
|
print("NO_BODY", no_body)
|
|
print("EMPTY_BODY", empty_body)
|
|
print("SHORT_BODY_LT100W", short_bodies)
|
|
print("BODY_WORDS", body_words)
|
|
print("BODY_CHARS", body_chars)
|
|
print("EXISTING_ZH", existing_zh)
|
|
print("ISSUE_DIRS_WITH_BODY", len(per_dir))
|
|
sizes.sort()
|
|
print("MEDIAN_BODY_WORDS", statistics.median(sizes))
|
|
print("P90_BODY_WORDS", sizes[int(len(sizes)*0.9)])
|
|
print("MAX_BODY_WORDS", sizes[-1])
|
|
print("SUM_OVER_5K", sum(1 for s in sizes if s > 5000)) |