65 lines
2.3 KiB
Python
65 lines
2.3 KiB
Python
import os, re, io, sys, json
|
|
sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8', errors='replace')
|
|
root = "/Users/weishen/Workspace/nexus/sreweekly/markdown"
|
|
|
|
# collect files with 正文 >= 50 chars, excluding those with existing -zh.md counterpart
|
|
pending = []
|
|
existing_zh = set()
|
|
for dp, dn, fn in os.walk(root):
|
|
for f in fn:
|
|
if re.search(r"-zh\.md$|\.zh\.md$|zh-CN\.md$", f):
|
|
existing_zh.add(os.path.join(dp, f))
|
|
|
|
for dp, dn, fn in os.walk(root):
|
|
for f in sorted(fn):
|
|
if not f.endswith(".md"): continue
|
|
if re.search(r"-zh\.md$|\.zh\.md$|zh-CN\.md$", f): continue
|
|
p = os.path.join(dp, f)
|
|
if p + "-zh" in [e.replace(".zh.md","").replace("-zh.md","") for e in [] ]: pass
|
|
stem = f[:-3]
|
|
if os.path.exists(os.path.join(dp, stem + "-zh.md")) or os.path.exists(os.path.join(dp, stem + ".zh.md")):
|
|
continue
|
|
try:
|
|
txt = open(p, encoding="utf-8").read()
|
|
except Exception:
|
|
continue
|
|
m = re.search(r"##\s*正文\s*\n(.*)", txt, re.S)
|
|
if not m: continue
|
|
body = m.group(1).strip()
|
|
if len(body) < 50: continue
|
|
pending.append((len(body), p))
|
|
|
|
pending.sort() # smallest first
|
|
print("PENDING", len(pending))
|
|
print("TOTAL_CHARS", sum(c for c, _ in pending))
|
|
|
|
# pack into tasks of ~50K chars each
|
|
TARGET = 50000
|
|
tasks = []
|
|
cur = []
|
|
cur_chars = 0
|
|
for c, p in pending:
|
|
if cur and cur_chars + c > TARGET:
|
|
tasks.append(cur)
|
|
cur = []
|
|
cur_chars = 0
|
|
cur.append(p)
|
|
cur_chars += c
|
|
if cur: tasks.append(cur)
|
|
print("TASKS", len(tasks))
|
|
|
|
os.makedirs("/Users/weishen/Workspace/nexus/sreweekly/translation_tasks", exist_ok=True)
|
|
meta = []
|
|
for i, t in enumerate(tasks):
|
|
# split into smaller chunks of ~15 files each for subagent work units inside a task?
|
|
with open(f"/Users/weishen/Workspace/nexus/sreweekly/translation_tasks/task_{i:04d}.txt", "w") as fh:
|
|
fh.write("\n".join(t) + "\n")
|
|
total = sum(os.path.getsize(p) for p in t)
|
|
meta.append({"task": i, "files": len(t), "chars": total})
|
|
|
|
with open("/Users/weishen/Workspace/nexus/sreweekly/translation_tasks/meta.json", "w") as fh:
|
|
json.dump(meta, fh, indent=1)
|
|
print("meta.json written; chars distribution:")
|
|
import statistics
|
|
cs = [m["chars"] for m in meta]
|
|
print("max task chars", max(cs), "median", statistics.median(cs)) |