diff --git a/synthmind/README.md b/synthmind/README.md index 02c464d..25b8e8a 100644 --- a/synthmind/README.md +++ b/synthmind/README.md @@ -47,7 +47,7 @@ - 支持指定语言(`--language zh/en/...`)或自动检测 - 支持音频**切分转录**:超长音频切分后逐段转录再合并 - 支持多种音频格式:`.mp3/.m4a/.wav/.flac/.ogg/.aac` -- **Manifest key 含模型维度**:多模型多参数结果共存,互不覆盖 +- **Manifest 嵌套 transcriptions**:同一视频/音频记录里,多模型多参数结果按 `[:]` 为 subkey 共存,互不覆盖 --- @@ -344,55 +344,78 @@ python3 transcribe_audio.py /home/user/audio/ --retry-failed ## Manifest 文件说明 -两个脚本**共享同一个 `manifest.json`**,通过 key 前缀区分: +两个脚本**共享同一个 `manifest.json`**(schema v2)。每个视频文件在 manifest 中**只有一条记录**,`extract_audio.py` 与 `transcribe_audio.py` 分别写入其中的 `extraction` 与 `transcriptions` 两个子对象,互不覆盖。 -- **视频提取记录**:key = 视频绝对路径 -- **音频转录记录**:key = `transcribe:[:]:<音频绝对路径>` +### key 选择规则 -### key 结构详解 +| Key | 来源 | 场景 | +|-----|------|------| +| 视频绝对路径(如 `/path/to/x.mp4`)| `extract_audio.py` 产生的记录 | 完整流水线(视频 → 音频 → 转录)| +| 音频绝对路径(如 `/path/to/x.mp3`)| 直接对 mp3 跑 `transcribe_audio.py` | 独立音频转录(无对应视频)| -| Key 格式 | 场景 | -|---------|------| -| `/path/to/x.mp4` | 视频提取记录 | -| `transcribe:tiny:/path/to/x.mp3` | tiny 模型转录(无 suffix) | -| `transcribe:medium:/path/to/x.mp3` | medium 模型转录(无 suffix) | -| `transcribe:base:_v1:/path/to/x.mp3` | base 模型 + `_v1` 后缀 | -| `transcribe:base:_v2:/path/to/x.mp3` | base 模型 + `_v2` 后缀 | +**反向查找**:当 `transcribe_audio.py` 处理某个 mp3 时,会先扫描 manifest,如果发现某条视频记录的 `extraction.audio_path` 等于该 mp3 路径,就把转录结果并入该视频记录,而不是新建 key。 -多个 key 可并存于同一 manifest,互不覆盖。 - -### manifest.json 完整示例(含多模型对比) +### 完整示例(v2) ```json { - "version": 1, + "version": 2, "files": { "/path/to/lesson1.mp4": { - "audio_extracted": true, - "audio_path": "/path/to/lesson1.mp3", - "split_segments": [], - "error": null, - "updated_at": "2026-08-23T10:30:00" + "extraction": { + "extracted": true, + "audio_path": "/path/to/lesson1.mp3", + "split_segments": [], + "error": null, + "updated_at": "2026-08-23T10:30:00" + }, + "transcriptions": { + "tiny": { + "transcribed": true, + "txt_path": "/path/to/lesson1.tiny.txt", + "split_segments": [], + "model": "tiny", + "language": "zh", + "output_suffix": null, + "error": null, + "updated_at": "2026-08-23T10:35:00" + }, + "medium": { + "transcribed": true, + "txt_path": "/path/to/lesson1.medium.txt", + "split_segments": [], + "model": "medium", + "language": "zh", + "output_suffix": null, + "error": null, + "updated_at": "2026-08-23T10:50:00" + }, + "base:_v1": { + "transcribed": true, + "txt_path": "/path/to/lesson1_v1.txt", + "split_segments": [], + "model": "base", + "language": null, + "output_suffix": "_v1", + "error": null, + "updated_at": "2026-08-23T11:00:00" + } + } }, - "transcribe:tiny:/path/to/lesson1.mp3": { - "transcribed": true, - "txt_path": "/path/to/lesson1.tiny.txt", - "split_segments": [], - "model": "tiny", - "language": "zh", - "output_suffix": null, - "error": null, - "updated_at": "2026-08-23T10:35:00" - }, - "transcribe:medium:/path/to/lesson1.mp3": { - "transcribed": true, - "txt_path": "/path/to/lesson1.medium.txt", - "split_segments": [], - "model": "medium", - "language": "zh", - "output_suffix": null, - "error": null, - "updated_at": "2026-08-23T10:50:00" + "/path/to/standalone.mp3": { + "extraction": null, + "transcriptions": { + "medium": { + "transcribed": true, + "txt_path": "/path/to/standalone.medium.txt", + "split_segments": [], + "model": "medium", + "language": "en", + "output_suffix": null, + "error": null, + "updated_at": "2026-08-23T11:15:00" + } + } } } } @@ -400,14 +423,18 @@ python3 transcribe_audio.py /home/user/audio/ --retry-failed ### 字段说明 -**视频提取记录字段** -- `audio_extracted` - 是否已提取音频(布尔) +**顶层每条记录** +- `extraction` - `extract_audio.py` 的产物;`null` 表示没有走过提取(例如直接对 mp3 跑转录) +- `transcriptions` - `transcribe_audio.py` 的产物;以 `` 或 `:` 为 subkey,允许同一音频用多种模型/参数并存 + +**`extraction` 字段** +- `extracted` - 是否已提取音频(布尔);`false` + 有 `error` 表示上次失败 - `audio_path` - 生成的 MP3 路径(切分模式下为第一个片段) -- `split_segments` - 切分片段列表(非切分模式为空) +- `split_segments` - 切分片段列表(非切分模式为空 `[]`) - `error` - 失败时的错误信息,成功为 `null` - `updated_at` - 最后更新时间(ISO 8601) -**音频转录记录字段** +**`transcriptions.` 字段** - `transcribed` - 是否已转录(布尔) - `txt_path` - 生成的 TXT 路径 - `split_segments` - 切分片段列表 @@ -420,12 +447,18 @@ python3 transcribe_audio.py /home/user/audio/ --retry-failed ### 断点续传逻辑 脚本判断是否需要处理某文件时,**同时校验**: -1. manifest 中标记为已完成 +1. manifest 中标记为已完成(`extraction.extracted=true` 或 `transcriptions..transcribed=true`) 2. 目标文件实际存在且非空 **任一条件不满足则重新处理**(例如手动删除了 mp3 后重跑,会重新提取)。 -**多模型模式**:每个 (音频 × 模型 × 后缀) 组合独立跟踪。已完成的组合被跳过,未完成的继续。 +**多模型模式**:每个 (音频 × 模型 × 后缀) 组合独立跟踪,对应 `transcriptions` 里独立的 subkey。已完成的组合被跳过,未完成的继续。 + +### 从旧版 (v1) 升级 + +老版本使用扁平 schema:视频记录 key = 视频路径;转录记录 key = `transcribe:[:]:<音频路径>`。这些散落的记录在 v2 里会被合并到同一条视频记录里。 + +**升级方式**:脚本首次加载旧 manifest 时会打印警告并**以空 v2 schema 重新开始**(旧数据将被下次保存覆盖)。如果你想保留旧记录,运行升级前**先手动备份 manifest.json**。所有已提取的 mp3 和已转录的 txt 会在下次运行时被识别为"存在但 manifest 无记录"→ 重新处理并写入新 schema。 --- diff --git a/synthmind/extract_audio.py b/synthmind/extract_audio.py index 2eb338d..88c9d25 100644 --- a/synthmind/extract_audio.py +++ b/synthmind/extract_audio.py @@ -56,13 +56,32 @@ def log(msg: str, level: str = "INFO"): print(f"[{ts}] {prefix} {msg}", flush=True) -# ─── Manifest 操作 ───────────────────────────────────────────────────────────── +# ─── Manifest 操作 (v2 schema) ──────────────────────────────────────────────── +# +# v2 schema:每个条目形如 +# { +# "extraction": { "extracted": bool, "audio_path": str, "split_segments": [], +# "error": str|None, "updated_at": str }, +# "transcriptions": { +# "[:]": { "transcribed": bool, "txt_path": str, ... } +# } +# } +# transcribe_audio.py 会把转录结果并入同一条视频记录的 transcriptions 里。 +# ───────────────────────────────────────────────────────────────────────────── + +MANIFEST_VERSION = 2 + def load_manifest(manifest_path: Path) -> dict: - if manifest_path.exists(): - with open(manifest_path, encoding="utf-8") as f: - return json.load(f) - return {"version": 1, "files": {}} + if not manifest_path.exists(): + return {"version": MANIFEST_VERSION, "files": {}} + with open(manifest_path, encoding="utf-8") as f: + data = json.load(f) + # 旧版 manifest (v1) 直接废弃,从头开始(用户已明确接受) + if data.get("version", 1) < MANIFEST_VERSION: + log("检测到旧版 manifest.json (v1),将以 v2 schema 重新初始化 - 旧记录将被覆盖", "WARN") + return {"version": MANIFEST_VERSION, "files": {}} + return data def save_manifest(manifest_path: Path, manifest: dict): @@ -72,34 +91,41 @@ def save_manifest(manifest_path: Path, manifest: dict): def get_file_entry(manifest: dict, video_path: str) -> dict: - """获取或初始化某个视频文件的 manifest 条目""" + """获取或初始化某个视频文件的 manifest 条目 (v2 schema)""" if video_path not in manifest["files"]: manifest["files"][video_path] = { - "audio_extracted": False, - "audio_path": None, - "split_segments": [], # 切分后的片段列表(若有) - "error": None, - "updated_at": None, + "extraction": None, + "transcriptions": {}, } - return manifest["files"][video_path] + entry = manifest["files"][video_path] + if "transcriptions" not in entry: + entry["transcriptions"] = {} + return entry def mark_done(manifest: dict, manifest_path: Path, video_path: str, audio_path: str, segments: list = None): entry = get_file_entry(manifest, video_path) - entry["audio_extracted"] = True - entry["audio_path"] = audio_path - entry["split_segments"] = segments or [] - entry["error"] = None - entry["updated_at"] = time.strftime("%Y-%m-%dT%H:%M:%S") + entry["extraction"] = { + "extracted": True, + "audio_path": audio_path, + "split_segments": segments or [], + "error": None, + "updated_at": time.strftime("%Y-%m-%dT%H:%M:%S"), + } save_manifest(manifest_path, manifest) def mark_failed(manifest: dict, manifest_path: Path, video_path: str, error: str): entry = get_file_entry(manifest, video_path) - entry["audio_extracted"] = False - entry["error"] = error - entry["updated_at"] = time.strftime("%Y-%m-%dT%H:%M:%S") + prev = entry.get("extraction") or {} + entry["extraction"] = { + "extracted": False, + "audio_path": prev.get("audio_path"), + "split_segments": prev.get("split_segments") or [], + "error": error, + "updated_at": time.strftime("%Y-%m-%dT%H:%M:%S"), + } save_manifest(manifest_path, manifest) @@ -220,22 +246,25 @@ def show_status(manifest: dict, all_videos: list): print("\n" + "=" * 60) print("📊 提取音频进度状态") print("=" * 60) - done = sum(1 for v in all_videos - if manifest["files"].get(v, {}).get("audio_extracted")) - failed = sum(1 for v in all_videos - if manifest["files"].get(v, {}).get("error")) - pending = len(all_videos) - done + done, failed = 0, 0 + for v in all_videos: + ext = ((manifest["files"].get(v) or {}).get("extraction")) or {} + if ext.get("extracted"): + done += 1 + elif ext.get("error"): + failed += 1 + pending = len(all_videos) - done - failed print(f" 总文件数 : {len(all_videos)}") print(f" 已完成 : {done}") print(f" 失败 : {failed}") print(f" 待处理 : {pending}") print("=" * 60) for v in all_videos: - entry = manifest["files"].get(v, {}) - if entry.get("audio_extracted"): + ext = ((manifest["files"].get(v) or {}).get("extraction")) or {} + if ext.get("extracted"): status = "✅ 完成" - elif entry.get("error"): - status = f"❌ 失败: {entry['error']}" + elif ext.get("error"): + status = f"❌ 失败: {ext['error']}" else: status = "⏳ 待处理" print(f" {status} {Path(v).name}") @@ -292,9 +321,10 @@ def main(): pending = [] for v in all_videos: entry = get_file_entry(manifest, v) - already_done = entry.get("audio_extracted") and entry.get("audio_path") and \ - mp3_exists(entry["audio_path"]) - is_failed = bool(entry.get("error")) + ext = entry.get("extraction") or {} + already_done = ext.get("extracted") and ext.get("audio_path") and \ + mp3_exists(ext["audio_path"]) + is_failed = bool(ext.get("error")) if already_done: continue diff --git a/synthmind/transcribe_audio.py b/synthmind/transcribe_audio.py index 866196b..f4e821e 100644 --- a/synthmind/transcribe_audio.py +++ b/synthmind/transcribe_audio.py @@ -66,13 +66,38 @@ def log(msg: str, level: str = "INFO"): print(f"[{ts}] {prefix} {msg}", flush=True) -# ─── Manifest 操作 (key 包含 model,支持同一音频多模型共存) ──────────────────── +# ─── Manifest 操作 (v2 schema) ──────────────────────────────────────────────── +# +# v2 schema:每条记录以 视频路径 (extract_audio 流) 或 音频路径 (独立转录) 为 key, +# 包含 extraction + transcriptions 两部分。extract_audio 与 transcribe_audio +# 共写同一条记录,互不覆盖。 +# +# { +# "": { +# "extraction": { "extracted": bool, "audio_path": str, "split_segments": [], +# "error": str|None, "updated_at": str } | null, +# "transcriptions": { +# "[:]": { "transcribed": bool, "txt_path": str, "model": str, +# "language": str|None, "output_suffix": str|None, +# "split_segments": [], "error": str|None, +# "updated_at": str } +# } +# } +# } +# ───────────────────────────────────────────────────────────────────────────── + +MANIFEST_VERSION = 2 + def load_manifest(manifest_path: Path) -> dict: - if manifest_path.exists(): - with open(manifest_path, encoding="utf-8") as f: - return json.load(f) - return {"version": 1, "files": {}} + if not manifest_path.exists(): + return {"version": MANIFEST_VERSION, "files": {}} + with open(manifest_path, encoding="utf-8") as f: + data = json.load(f) + if data.get("version", 1) < MANIFEST_VERSION: + log("检测到旧版 manifest.json (v1),将以 v2 schema 重新初始化 - 旧记录将被覆盖", "WARN") + return {"version": MANIFEST_VERSION, "files": {}} + return data def save_manifest(manifest_path: Path, manifest: dict): @@ -81,31 +106,56 @@ def save_manifest(manifest_path: Path, manifest: dict): json.dump(manifest, f, ensure_ascii=False, indent=2) -def transcribe_key(model: str, audio_path: str, output_suffix: str = "") -> str: - """ - manifest key 由 model + output_suffix + 音频路径共同决定, - 因此改变 --output-suffix 会形成新的 key(不会误跳过)。 - """ +def transcription_subkey(model: str, output_suffix: str = "") -> str: + """transcriptions 字典的 subkey: 或 :""" if output_suffix: - return f"transcribe:{model}:{output_suffix}:{audio_path}" - return f"transcribe:{model}:{audio_path}" + return f"{model}:{output_suffix}" + return model + + +def find_entry_key_for_audio(manifest: dict, audio_path: str) -> str: + """ + 反向查找:audio 属于哪条 manifest 记录。 + - 若某视频记录的 extraction.audio_path == audio → 归入该视频 key + - 否则以 audio_path 自身为 key(独立音频场景) + """ + for key, entry in manifest["files"].items(): + extraction = entry.get("extraction") + if extraction and extraction.get("audio_path") == audio_path: + return key + return audio_path + + +def ensure_entry(manifest: dict, key: str) -> dict: + if key not in manifest["files"]: + manifest["files"][key] = { + "extraction": None, + "transcriptions": {}, + } + entry = manifest["files"][key] + if "transcriptions" not in entry: + entry["transcriptions"] = {} + return entry def get_transcribe_entry(manifest: dict, audio_path: str, model: str, output_suffix: str = "") -> dict: - key = transcribe_key(model, audio_path, output_suffix) - if key not in manifest["files"]: - manifest["files"][key] = { + """获取或初始化某个 (音频, 模型, suffix) 的转录记录""" + key = find_entry_key_for_audio(manifest, audio_path) + parent = ensure_entry(manifest, key) + subkey = transcription_subkey(model, output_suffix) + if subkey not in parent["transcriptions"]: + parent["transcriptions"][subkey] = { "transcribed": False, "txt_path": None, "split_segments": [], - "error": None, "model": model, "language": None, "output_suffix": output_suffix or None, + "error": None, "updated_at": None, } - return manifest["files"][key] + return parent["transcriptions"][subkey] def mark_transcribed(manifest: dict, manifest_path: Path, audio_path: str, @@ -305,31 +355,43 @@ def show_status(manifest: dict, all_audios: list): print("📊 音频转录进度状态(按 模型 x 音频 展示)") print("=" * 70) - entries = {k: v for k, v in manifest["files"].items() if k.startswith("transcribe:")} - done = sum(1 for v in entries.values() if v.get("transcribed")) - failed = sum(1 for v in entries.values() if v.get("error")) - print(f" Manifest 记录: {len(entries)} 已完成: {done} 失败: {failed}") + all_trans = [] + for parent_key, entry in manifest["files"].items(): + for subkey, tr in (entry.get("transcriptions") or {}).items(): + all_trans.append((parent_key, subkey, tr)) + + done = sum(1 for _, _, v in all_trans if v.get("transcribed")) + failed = sum(1 for _, _, v in all_trans if v.get("error")) + print(f" 转录记录: {len(all_trans)} 已完成: {done} 失败: {failed}") print("=" * 70) for a in all_audios: print(f"\n 📄 {Path(a).name}") - matching = [(k, v) for k, v in entries.items() if k.endswith(f":{a}")] + matching = [] + for parent_key, entry in manifest["files"].items(): + extraction = entry.get("extraction") or {} + is_match = extraction.get("audio_path") == a or parent_key == a + if not is_match: + continue + for subkey, tr in (entry.get("transcriptions") or {}).items(): + matching.append((subkey, tr)) + if not matching: print(f" ⏳ 无记录(未转录)") continue - for key, entry in matching: - m = entry.get("model", "?") - suffix = entry.get("output_suffix") or "" + for subkey, tr in matching: + m = tr.get("model", "?") + suffix = tr.get("output_suffix") or "" suffix_tag = f" suffix={suffix}" if suffix else "" - if entry.get("transcribed"): + if tr.get("transcribed"): size = "" - if entry.get("txt_path") and os.path.exists(entry["txt_path"]): - size = f" {os.path.getsize(entry['txt_path']) // 1024}KB" - lang = entry.get("language") or "auto" - txt_name = Path(entry['txt_path']).name if entry.get('txt_path') else '?' + if tr.get("txt_path") and os.path.exists(tr["txt_path"]): + size = f" {os.path.getsize(tr['txt_path']) // 1024}KB" + lang = tr.get("language") or "auto" + txt_name = Path(tr['txt_path']).name if tr.get('txt_path') else '?' print(f" ✅ [{m:6s}] lang={lang}{suffix_tag}{size} → {txt_name}") - elif entry.get("error"): - print(f" ❌ [{m:6s}]{suffix_tag} {entry['error'][:60]}") + elif tr.get("error"): + print(f" ❌ [{m:6s}]{suffix_tag} {tr['error'][:60]}") print()