import os, re, io, sys, json sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8', errors='replace') root = "/Users/weishen/Workspace/nexus/sreweekly/markdown" # collect files with 正文 >= 50 chars, excluding those with existing -zh.md counterpart pending = [] existing_zh = set() for dp, dn, fn in os.walk(root): for f in fn: if re.search(r"-zh\.md$|\.zh\.md$|zh-CN\.md$", f): existing_zh.add(os.path.join(dp, f)) for dp, dn, fn in os.walk(root): for f in sorted(fn): if not f.endswith(".md"): continue if re.search(r"-zh\.md$|\.zh\.md$|zh-CN\.md$", f): continue p = os.path.join(dp, f) if p + "-zh" in [e.replace(".zh.md","").replace("-zh.md","") for e in [] ]: pass stem = f[:-3] if os.path.exists(os.path.join(dp, stem + "-zh.md")) or os.path.exists(os.path.join(dp, stem + ".zh.md")): continue try: txt = open(p, encoding="utf-8").read() except Exception: continue m = re.search(r"##\s*正文\s*\n(.*)", txt, re.S) if not m: continue body = m.group(1).strip() if len(body) < 50: continue pending.append((len(body), p)) pending.sort() # smallest first print("PENDING", len(pending)) print("TOTAL_CHARS", sum(c for c, _ in pending)) # pack into tasks of ~50K chars each TARGET = 50000 tasks = [] cur = [] cur_chars = 0 for c, p in pending: if cur and cur_chars + c > TARGET: tasks.append(cur) cur = [] cur_chars = 0 cur.append(p) cur_chars += c if cur: tasks.append(cur) print("TASKS", len(tasks)) os.makedirs("/Users/weishen/Workspace/nexus/sreweekly/translation_tasks", exist_ok=True) meta = [] for i, t in enumerate(tasks): # split into smaller chunks of ~15 files each for subagent work units inside a task? with open(f"/Users/weishen/Workspace/nexus/sreweekly/translation_tasks/task_{i:04d}.txt", "w") as fh: fh.write("\n".join(t) + "\n") total = sum(os.path.getsize(p) for p in t) meta.append({"task": i, "files": len(t), "chars": total}) with open("/Users/weishen/Workspace/nexus/sreweekly/translation_tasks/meta.json", "w") as fh: json.dump(meta, fh, indent=1) print("meta.json written; chars distribution:") import statistics cs = [m["chars"] for m in meta] print("max task chars", max(cs), "median", statistics.median(cs))