manifest.json里extract audio, transcribe audio 共用一条记录

This commit is contained in:
2026-09-17 08:49:16 +08:00
parent 922c3b4071
commit 44110a0ccf
3 changed files with 235 additions and 110 deletions

View File

@@ -66,13 +66,38 @@ def log(msg: str, level: str = "INFO"):
print(f"[{ts}] {prefix} {msg}", flush=True)
# ─── Manifest 操作 (key 包含 model,支持同一音频多模型共存) ────────────────────
# ─── Manifest 操作 (v2 schema) ────────────────────────────────────────────────
#
# v2 schema:每条记录以 视频路径 (extract_audio 流) 或 音频路径 (独立转录) 为 key,
# 包含 extraction + transcriptions 两部分。extract_audio 与 transcribe_audio
# 共写同一条记录,互不覆盖。
#
# {
# "<key>": {
# "extraction": { "extracted": bool, "audio_path": str, "split_segments": [],
# "error": str|None, "updated_at": str } | null,
# "transcriptions": {
# "<model>[:<suffix>]": { "transcribed": bool, "txt_path": str, "model": str,
# "language": str|None, "output_suffix": str|None,
# "split_segments": [], "error": str|None,
# "updated_at": str }
# }
# }
# }
# ─────────────────────────────────────────────────────────────────────────────
MANIFEST_VERSION = 2
def load_manifest(manifest_path: Path) -> dict:
if manifest_path.exists():
with open(manifest_path, encoding="utf-8") as f:
return json.load(f)
return {"version": 1, "files": {}}
if not manifest_path.exists():
return {"version": MANIFEST_VERSION, "files": {}}
with open(manifest_path, encoding="utf-8") as f:
data = json.load(f)
if data.get("version", 1) < MANIFEST_VERSION:
log("检测到旧版 manifest.json (v1),将以 v2 schema 重新初始化 - 旧记录将被覆盖", "WARN")
return {"version": MANIFEST_VERSION, "files": {}}
return data
def save_manifest(manifest_path: Path, manifest: dict):
@@ -81,31 +106,56 @@ def save_manifest(manifest_path: Path, manifest: dict):
json.dump(manifest, f, ensure_ascii=False, indent=2)
def transcribe_key(model: str, audio_path: str, output_suffix: str = "") -> str:
"""
manifest key 由 model + output_suffix + 音频路径共同决定,
因此改变 --output-suffix 会形成新的 key(不会误跳过)。
"""
def transcription_subkey(model: str, output_suffix: str = "") -> str:
"""transcriptions 字典的 subkey:<model> 或 <model>:<suffix>"""
if output_suffix:
return f"transcribe:{model}:{output_suffix}:{audio_path}"
return f"transcribe:{model}:{audio_path}"
return f"{model}:{output_suffix}"
return model
def find_entry_key_for_audio(manifest: dict, audio_path: str) -> str:
"""
反向查找:audio 属于哪条 manifest 记录。
- 若某视频记录的 extraction.audio_path == audio → 归入该视频 key
- 否则以 audio_path 自身为 key(独立音频场景)
"""
for key, entry in manifest["files"].items():
extraction = entry.get("extraction")
if extraction and extraction.get("audio_path") == audio_path:
return key
return audio_path
def ensure_entry(manifest: dict, key: str) -> dict:
if key not in manifest["files"]:
manifest["files"][key] = {
"extraction": None,
"transcriptions": {},
}
entry = manifest["files"][key]
if "transcriptions" not in entry:
entry["transcriptions"] = {}
return entry
def get_transcribe_entry(manifest: dict, audio_path: str, model: str,
output_suffix: str = "") -> dict:
key = transcribe_key(model, audio_path, output_suffix)
if key not in manifest["files"]:
manifest["files"][key] = {
"""获取或初始化某个 (音频, 模型, suffix) 的转录记录"""
key = find_entry_key_for_audio(manifest, audio_path)
parent = ensure_entry(manifest, key)
subkey = transcription_subkey(model, output_suffix)
if subkey not in parent["transcriptions"]:
parent["transcriptions"][subkey] = {
"transcribed": False,
"txt_path": None,
"split_segments": [],
"error": None,
"model": model,
"language": None,
"output_suffix": output_suffix or None,
"error": None,
"updated_at": None,
}
return manifest["files"][key]
return parent["transcriptions"][subkey]
def mark_transcribed(manifest: dict, manifest_path: Path, audio_path: str,
@@ -305,31 +355,43 @@ def show_status(manifest: dict, all_audios: list):
print("📊 音频转录进度状态(按 模型 x 音频 展示)")
print("=" * 70)
entries = {k: v for k, v in manifest["files"].items() if k.startswith("transcribe:")}
done = sum(1 for v in entries.values() if v.get("transcribed"))
failed = sum(1 for v in entries.values() if v.get("error"))
print(f" Manifest 记录: {len(entries)} 已完成: {done} 失败: {failed}")
all_trans = []
for parent_key, entry in manifest["files"].items():
for subkey, tr in (entry.get("transcriptions") or {}).items():
all_trans.append((parent_key, subkey, tr))
done = sum(1 for _, _, v in all_trans if v.get("transcribed"))
failed = sum(1 for _, _, v in all_trans if v.get("error"))
print(f" 转录记录: {len(all_trans)} 已完成: {done} 失败: {failed}")
print("=" * 70)
for a in all_audios:
print(f"\n 📄 {Path(a).name}")
matching = [(k, v) for k, v in entries.items() if k.endswith(f":{a}")]
matching = []
for parent_key, entry in manifest["files"].items():
extraction = entry.get("extraction") or {}
is_match = extraction.get("audio_path") == a or parent_key == a
if not is_match:
continue
for subkey, tr in (entry.get("transcriptions") or {}).items():
matching.append((subkey, tr))
if not matching:
print(f" ⏳ 无记录(未转录)")
continue
for key, entry in matching:
m = entry.get("model", "?")
suffix = entry.get("output_suffix") or ""
for subkey, tr in matching:
m = tr.get("model", "?")
suffix = tr.get("output_suffix") or ""
suffix_tag = f" suffix={suffix}" if suffix else ""
if entry.get("transcribed"):
if tr.get("transcribed"):
size = ""
if entry.get("txt_path") and os.path.exists(entry["txt_path"]):
size = f" {os.path.getsize(entry['txt_path']) // 1024}KB"
lang = entry.get("language") or "auto"
txt_name = Path(entry['txt_path']).name if entry.get('txt_path') else '?'
if tr.get("txt_path") and os.path.exists(tr["txt_path"]):
size = f" {os.path.getsize(tr['txt_path']) // 1024}KB"
lang = tr.get("language") or "auto"
txt_name = Path(tr['txt_path']).name if tr.get('txt_path') else '?'
print(f" ✅ [{m:6s}] lang={lang}{suffix_tag}{size} → {txt_name}")
elif entry.get("error"):
print(f" ❌ [{m:6s}]{suffix_tag} {entry['error'][:60]}")
elif tr.get("error"):
print(f" ❌ [{m:6s}]{suffix_tag} {tr['error'][:60]}")
print()