transcode_music source code + doc
This commit is contained in:
933
transcode_music/transcode_music.py
Normal file
933
transcode_music/transcode_music.py
Normal file
@@ -0,0 +1,933 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
transcode_music.py - 批量转码无损音乐 (FLAC/WAV/DSF/APE) 为 MP3 320k CBR.
|
||||
|
||||
CLI tool. Walks source dir, transcodes lossless audio to MP3, copies covers/txt/cue,
|
||||
tracks state in a JSON manifest for resumable runs. Detects CUE+whole-disc albums
|
||||
that need manual splitting and reports them without attempting transcode.
|
||||
|
||||
Dependencies: Python 3.10+, ffmpeg, ffprobe.
|
||||
|
||||
Usage:
|
||||
# Analyze only (recommended first run to catch CUE issues)
|
||||
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test --analyze-only
|
||||
|
||||
# Full run (auto-resumes, skips completed)
|
||||
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test
|
||||
|
||||
# Force re-run (ignore existing manifest)
|
||||
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test --force
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import datetime as dt
|
||||
import errno
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
from dataclasses import asdict, dataclass
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
__version__ = "1.0.0"
|
||||
|
||||
MANIFEST_SCHEMA_VERSION = 1
|
||||
MANIFEST_NAME = ".transcode_manifest.json"
|
||||
|
||||
LOSSLESS_EXTS = {".flac", ".wav", ".dsf", ".dff", ".ape"}
|
||||
LOSSY_EXTS = {".mp3", ".m4a", ".aac", ".ogg", ".opus", ".wma", ".mp2"}
|
||||
IMAGE_EXTS = {".jpg", ".jpeg", ".png", ".bmp", ".gif", ".webp", ".tif", ".tiff"}
|
||||
DOCUMENT_EXTS = {".txt", ".log", ".pdf", ".md", ".nfo", ".doc", ".docx", ".rtf"}
|
||||
CUE_EXTS = {".cue"}
|
||||
|
||||
IGNORE_NAMES = {".ds_store", "thumbs.db", "desktop.ini", ".directory"}
|
||||
IGNORE_PREFIXES = ("._",) # macOS resource forks — not user content
|
||||
|
||||
# LAME/MP3 spec: only these sample rates are legal for MP3 output.
|
||||
# Inputs outside this set (high-res 96/192k, DSD 352.8/705.6k) MUST be resampled.
|
||||
MP3_SAMPLE_RATES = {8000, 11025, 12000, 16000, 22050, 24000, 32000, 44100, 48000}
|
||||
|
||||
|
||||
class Status(str, Enum):
|
||||
PENDING = "pending"
|
||||
ANALYZED = "analyzed"
|
||||
PROCESSING = "processing"
|
||||
COMPLETED = "completed"
|
||||
FAILED = "failed"
|
||||
NEEDS_CUE_SPLIT = "needs_cue_split"
|
||||
SKIPPED = "skipped"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Config:
|
||||
source_root: Path
|
||||
target_root: Path
|
||||
bitrate: str = "320k"
|
||||
id3v2_version: str = "3"
|
||||
output_sample_rate_when_high: int = 44100
|
||||
keep_source_mtime: bool = True
|
||||
force: bool = False
|
||||
analyze_only: bool = False
|
||||
ffmpeg: str = "ffmpeg"
|
||||
ffprobe: str = "ffprobe"
|
||||
ffmpeg_timeout_sec: int = 3600
|
||||
|
||||
|
||||
@dataclass
|
||||
class AudioFile:
|
||||
type: str
|
||||
size: int
|
||||
sample_rate: Optional[int] = None
|
||||
channels: Optional[int] = None
|
||||
duration_sec: Optional[float] = None
|
||||
bits_per_sample: Optional[int] = None
|
||||
codec: Optional[str] = None
|
||||
status: str = Status.PENDING.value
|
||||
target_rel: str = ""
|
||||
error: Optional[str] = None
|
||||
converted_at: Optional[str] = None
|
||||
|
||||
|
||||
@dataclass
|
||||
class AssetFile:
|
||||
type: str
|
||||
size: int
|
||||
status: str = Status.PENDING.value
|
||||
target_rel: str = ""
|
||||
error: Optional[str] = None
|
||||
copied_at: Optional[str] = None
|
||||
|
||||
|
||||
def _now_iso() -> str:
|
||||
return dt.datetime.now(dt.timezone.utc).astimezone().isoformat(timespec="seconds")
|
||||
|
||||
|
||||
def _to_int(x) -> Optional[int]:
|
||||
if x in (None, "", "N/A"):
|
||||
return None
|
||||
try:
|
||||
return int(x)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _to_float(x) -> Optional[float]:
|
||||
if x in (None, "", "N/A"):
|
||||
return None
|
||||
try:
|
||||
return float(x)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def atomic_write_json(path: Path, data: dict) -> None:
|
||||
payload = json.dumps(data, ensure_ascii=False, indent=2)
|
||||
tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.tmp")
|
||||
|
||||
fd = os.open(str(tmp), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644)
|
||||
try:
|
||||
os.write(fd, payload.encode("utf-8"))
|
||||
try:
|
||||
os.fsync(fd)
|
||||
except OSError:
|
||||
pass # network mounts (CIFS/NFS) may not implement fsync
|
||||
finally:
|
||||
os.close(fd)
|
||||
|
||||
os.replace(str(tmp), str(path))
|
||||
|
||||
try:
|
||||
dfd = os.open(str(path.parent), os.O_RDONLY)
|
||||
try:
|
||||
os.fsync(dfd) # persist the rename itself (durability guarantee)
|
||||
except OSError:
|
||||
pass
|
||||
finally:
|
||||
os.close(dfd)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def quarantine_corrupt(path: Path, reason: str) -> Path:
|
||||
ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S")
|
||||
q = path.with_name(f"{path.name}.corrupt-{ts}")
|
||||
try:
|
||||
os.replace(str(path), str(q))
|
||||
logging.error(f"Manifest 损坏, 已隔离到 {q}: {reason}")
|
||||
except OSError as e:
|
||||
logging.error(f"隔离失败 {path}: {e}")
|
||||
return q
|
||||
|
||||
|
||||
def ffprobe_audio(path: Path, cfg: Config) -> dict:
|
||||
try:
|
||||
result = subprocess.run(
|
||||
[
|
||||
cfg.ffprobe, "-v", "error",
|
||||
"-select_streams", "a:0",
|
||||
"-show_entries", "stream=codec_name,sample_rate,channels,bits_per_raw_sample",
|
||||
"-show_entries", "format=duration",
|
||||
"-of", "json",
|
||||
str(path),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=60,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
return {"error": (result.stderr or "ffprobe failed").strip()[-200:]}
|
||||
data = json.loads(result.stdout or "{}")
|
||||
stream = (data.get("streams") or [{}])[0]
|
||||
fmt = data.get("format") or {}
|
||||
return {
|
||||
"codec": stream.get("codec_name"),
|
||||
"sample_rate": _to_int(stream.get("sample_rate")),
|
||||
"channels": _to_int(stream.get("channels")),
|
||||
"duration_sec": _to_float(fmt.get("duration")),
|
||||
"bits_per_sample": _to_int(stream.get("bits_per_raw_sample")),
|
||||
}
|
||||
except subprocess.TimeoutExpired:
|
||||
return {"error": "ffprobe timeout"}
|
||||
except (json.JSONDecodeError, OSError) as e:
|
||||
return {"error": str(e)}
|
||||
|
||||
|
||||
def choose_output_sample_rate(input_rate: Optional[int], prefer_when_high: int) -> Optional[int]:
|
||||
if input_rate is None:
|
||||
return prefer_when_high
|
||||
if input_rate in MP3_SAMPLE_RATES:
|
||||
return None
|
||||
return prefer_when_high
|
||||
|
||||
|
||||
def transcode_to_mp3(
|
||||
src: Path,
|
||||
dst: Path,
|
||||
input_sample_rate: Optional[int],
|
||||
cfg: Config,
|
||||
) -> tuple[bool, Optional[str]]:
|
||||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||||
tmp = dst.with_name(dst.name + ".tmp")
|
||||
if tmp.exists():
|
||||
try:
|
||||
tmp.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
out_rate = choose_output_sample_rate(input_sample_rate, cfg.output_sample_rate_when_high)
|
||||
|
||||
cmd = [
|
||||
cfg.ffmpeg, "-nostdin", "-hide_banner", "-loglevel", "error", "-y",
|
||||
"-i", str(src),
|
||||
"-c:a", "libmp3lame",
|
||||
"-b:a", cfg.bitrate,
|
||||
]
|
||||
if out_rate is not None:
|
||||
cmd += ["-ar", str(out_rate)]
|
||||
cmd += [
|
||||
"-map_metadata", "0",
|
||||
"-id3v2_version", cfg.id3v2_version,
|
||||
"-map", "0:a",
|
||||
"-map", "0:v?", # optional embedded cover art
|
||||
"-c:v", "copy",
|
||||
"-f", "mp3", # explicit muxer: tmp filename has .tmp suffix, cannot auto-detect
|
||||
str(tmp),
|
||||
]
|
||||
|
||||
orig_mtime: Optional[float] = None
|
||||
if cfg.keep_source_mtime:
|
||||
try:
|
||||
orig_mtime = src.stat().st_mtime
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
result = subprocess.run(
|
||||
cmd, capture_output=True, text=True, timeout=cfg.ffmpeg_timeout_sec,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
stderr = (result.stderr or "").strip()
|
||||
err_msg = stderr.splitlines()[-1] if stderr else f"ffmpeg exit {result.returncode}"
|
||||
_try_unlink(tmp)
|
||||
return False, err_msg[-400:]
|
||||
|
||||
if orig_mtime is not None:
|
||||
try:
|
||||
os.utime(str(tmp), (orig_mtime, orig_mtime))
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
try:
|
||||
os.replace(str(tmp), str(dst)) # same target-drive rename → atomic
|
||||
except OSError as e:
|
||||
if e.errno == errno.EXDEV:
|
||||
shutil.move(str(tmp), str(dst)) # cross-device fallback (rare)
|
||||
else:
|
||||
raise
|
||||
return True, None
|
||||
|
||||
except subprocess.TimeoutExpired:
|
||||
_try_unlink(tmp)
|
||||
return False, f"ffmpeg 超时 ({cfg.ffmpeg_timeout_sec}s)"
|
||||
except (OSError, subprocess.SubprocessError) as e:
|
||||
_try_unlink(tmp)
|
||||
return False, str(e)
|
||||
|
||||
|
||||
def _try_unlink(p: Path) -> None:
|
||||
try:
|
||||
if p.exists():
|
||||
p.unlink()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def copy_asset(src: Path, dst: Path, _cfg: Config) -> tuple[bool, Optional[str]]:
|
||||
try:
|
||||
dst.parent.mkdir(parents=True, exist_ok=True)
|
||||
src_size = src.stat().st_size
|
||||
if dst.exists():
|
||||
try:
|
||||
if dst.stat().st_size == src_size:
|
||||
return True, None # size-only idempotency (9p mounts can't preserve mtime)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
shutil.copyfile(str(src), str(dst))
|
||||
try:
|
||||
src_mtime = src.stat().st_mtime
|
||||
os.utime(str(dst), (src_mtime, src_mtime))
|
||||
except OSError:
|
||||
pass # 9p / SMB mounts often reject chmod/utime; content copy still succeeded
|
||||
return True, None
|
||||
except OSError as e:
|
||||
return False, str(e)
|
||||
|
||||
|
||||
CUE_FILE_RE = re.compile(r'^\s*FILE\s+"([^"]+)"\s+(\w+)', re.IGNORECASE)
|
||||
CUE_TRACK_RE = re.compile(r'^\s*TRACK\s+(\d+)\s+(\w+)', re.IGNORECASE)
|
||||
|
||||
# CUE files from Chinese/Japanese/Korean releases often use legacy encodings.
|
||||
CUE_ENCODINGS = ("utf-8-sig", "utf-8", "gbk", "big5", "shift_jis", "cp936", "latin1")
|
||||
|
||||
|
||||
def read_cue_text(path: Path) -> Optional[str]:
|
||||
try:
|
||||
raw = path.read_bytes()
|
||||
except OSError:
|
||||
return None
|
||||
for enc in CUE_ENCODINGS:
|
||||
try:
|
||||
return raw.decode(enc)
|
||||
except UnicodeDecodeError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
def parse_cue(path: Path) -> Optional[dict]:
|
||||
text = read_cue_text(path)
|
||||
if text is None:
|
||||
return None
|
||||
files: list[dict] = []
|
||||
tracks: list[dict] = []
|
||||
counter = 0
|
||||
for line in text.splitlines():
|
||||
m = CUE_FILE_RE.match(line)
|
||||
if m:
|
||||
counter += 1
|
||||
files.append({"name": m.group(1), "format": m.group(2), "counter": counter})
|
||||
continue
|
||||
m = CUE_TRACK_RE.match(line)
|
||||
if m:
|
||||
tracks.append({"num": int(m.group(1)), "file_counter": counter})
|
||||
return {"files": files, "tracks": tracks}
|
||||
|
||||
|
||||
def analyze_cue_situation(
|
||||
cue_files: list[Path],
|
||||
audio_files_by_name: dict[str, Path],
|
||||
) -> dict:
|
||||
"""
|
||||
Classify CUE + audio combinations. Returns:
|
||||
{
|
||||
"needs_split": bool, # True => album must be manually CUE-split before transcode
|
||||
"skip_audio": set[str], # audio filenames to skip (redundant whole-disc file)
|
||||
"issues": list[str], # human-readable notes/warnings
|
||||
}
|
||||
|
||||
Whipper counter-pattern logic:
|
||||
- CUE with 1 FILE and that file is the only audio → whole-disc → needs_split
|
||||
- CUE with 1 FILE but per-track audio also present → skip the whole-disc file
|
||||
- CUE with multiple FILEs → per-track index, transcode all
|
||||
- CUE references missing files → note but continue
|
||||
"""
|
||||
result = {"needs_split": False, "skip_audio": set(), "issues": []}
|
||||
|
||||
for cue in cue_files:
|
||||
parsed = parse_cue(cue)
|
||||
if parsed is None:
|
||||
result["issues"].append(f"{cue.name}: 无法解析 (编码问题)")
|
||||
continue
|
||||
|
||||
refs = parsed["files"]
|
||||
num_files = len(refs)
|
||||
num_tracks = len(parsed["tracks"])
|
||||
|
||||
if num_files == 0:
|
||||
result["issues"].append(f"{cue.name}: 无 FILE entry")
|
||||
continue
|
||||
|
||||
refs_present = [r["name"] for r in refs if r["name"] in audio_files_by_name]
|
||||
refs_missing = [r["name"] for r in refs if r["name"] not in audio_files_by_name]
|
||||
|
||||
if num_files == 1:
|
||||
ref_name = refs[0]["name"]
|
||||
if ref_name in audio_files_by_name:
|
||||
other_audio = [n for n in audio_files_by_name if n != ref_name]
|
||||
if not other_audio:
|
||||
result["needs_split"] = True
|
||||
result["issues"].append(
|
||||
f"{cue.name}: 整轨型 (单文件 '{ref_name}', {num_tracks} 轨) "
|
||||
f"— 需先手动 CUE split 再转码"
|
||||
)
|
||||
else:
|
||||
result["skip_audio"].add(ref_name)
|
||||
result["issues"].append(
|
||||
f"{cue.name}: 引用 '{ref_name}' 但存在 {len(other_audio)} 个分轨 "
|
||||
f"→ 跳过 '{ref_name}' 只转分轨"
|
||||
)
|
||||
else:
|
||||
result["issues"].append(
|
||||
f"{cue.name}: 引用 '{ref_name}' 但文件不存在"
|
||||
)
|
||||
else:
|
||||
if refs_missing:
|
||||
result["issues"].append(
|
||||
f"{cue.name}: {len(refs_missing)}/{num_files} 个引用文件缺失 "
|
||||
f"(例: {refs_missing[:2]})"
|
||||
)
|
||||
if len(refs_present) < 2:
|
||||
result["issues"].append(
|
||||
f"{cue.name}: 多 FILE 引用但仅 {len(refs_present)} 个存在, 请人工核查"
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def classify_file(path: Path) -> str:
|
||||
name = path.name.lower()
|
||||
if name in IGNORE_NAMES:
|
||||
return "ignore"
|
||||
for pref in IGNORE_PREFIXES:
|
||||
if name.startswith(pref):
|
||||
return "ignore"
|
||||
if MANIFEST_NAME.lower() in name:
|
||||
return "ignore"
|
||||
|
||||
ext = path.suffix.lower()
|
||||
if ext in LOSSLESS_EXTS:
|
||||
return "audio-lossless"
|
||||
if ext in LOSSY_EXTS:
|
||||
return "audio-lossy"
|
||||
if ext in IMAGE_EXTS:
|
||||
return "image"
|
||||
if ext in DOCUMENT_EXTS:
|
||||
return "document"
|
||||
if ext in CUE_EXTS:
|
||||
return "cue"
|
||||
return "other"
|
||||
|
||||
|
||||
def find_all_dirs(source_root: Path) -> list[Path]:
|
||||
result = []
|
||||
for dirpath, dirnames, _ in os.walk(source_root):
|
||||
dirnames.sort()
|
||||
for d in dirnames:
|
||||
result.append(Path(dirpath) / d)
|
||||
return sorted(result)
|
||||
|
||||
|
||||
def analyze_directory(dir_path: Path, cfg: Config) -> dict:
|
||||
entry = {
|
||||
"status": Status.PENDING.value,
|
||||
"issues": [],
|
||||
"audio_files": {},
|
||||
"asset_files": {},
|
||||
"ignored_files": [],
|
||||
"has_audio": False,
|
||||
"started_at": None,
|
||||
"completed_at": None,
|
||||
"error": None,
|
||||
}
|
||||
|
||||
try:
|
||||
items = sorted(dir_path.iterdir())
|
||||
except OSError as e:
|
||||
entry["status"] = Status.FAILED.value
|
||||
entry["error"] = f"无法读取目录: {e}"
|
||||
return entry
|
||||
|
||||
file_paths = [p for p in items if p.is_file()]
|
||||
|
||||
audio_paths: list[Path] = []
|
||||
cue_paths: list[Path] = []
|
||||
|
||||
for f in file_paths:
|
||||
cat = classify_file(f)
|
||||
try:
|
||||
stat = f.stat()
|
||||
except OSError as e:
|
||||
entry["issues"].append(f"stat '{f.name}' 失败: {e}")
|
||||
continue
|
||||
|
||||
rel = f.name
|
||||
|
||||
if cat == "ignore":
|
||||
entry["ignored_files"].append(rel)
|
||||
continue
|
||||
|
||||
if cat == "audio-lossless":
|
||||
info = ffprobe_audio(f, cfg)
|
||||
af = AudioFile(
|
||||
type=f.suffix.lower().lstrip("."),
|
||||
size=stat.st_size,
|
||||
sample_rate=info.get("sample_rate"),
|
||||
channels=info.get("channels"),
|
||||
duration_sec=info.get("duration_sec"),
|
||||
bits_per_sample=info.get("bits_per_sample"),
|
||||
codec=info.get("codec"),
|
||||
target_rel=str(Path(rel).with_suffix(".mp3")),
|
||||
)
|
||||
if "error" in info:
|
||||
af.error = info["error"]
|
||||
entry["issues"].append(f"ffprobe '{rel}': {info['error']}")
|
||||
entry["audio_files"][rel] = asdict(af)
|
||||
audio_paths.append(f)
|
||||
elif cat == "cue":
|
||||
entry["asset_files"][rel] = asdict(
|
||||
AssetFile(type="cue", size=stat.st_size, target_rel=rel)
|
||||
)
|
||||
cue_paths.append(f)
|
||||
elif cat in ("image", "document"):
|
||||
entry["asset_files"][rel] = asdict(
|
||||
AssetFile(type=cat, size=stat.st_size, target_rel=rel)
|
||||
)
|
||||
elif cat == "audio-lossy":
|
||||
entry["asset_files"][rel] = asdict(
|
||||
AssetFile(type=f.suffix.lower().lstrip("."), size=stat.st_size, target_rel=rel)
|
||||
)
|
||||
else:
|
||||
entry["asset_files"][rel] = asdict(
|
||||
AssetFile(type="other", size=stat.st_size, target_rel=rel)
|
||||
)
|
||||
|
||||
entry["has_audio"] = bool(audio_paths)
|
||||
|
||||
if cue_paths and audio_paths:
|
||||
audio_by_name = {p.name: p for p in audio_paths}
|
||||
cue_result = analyze_cue_situation(cue_paths, audio_by_name)
|
||||
entry["issues"].extend(cue_result["issues"])
|
||||
if cue_result["needs_split"]:
|
||||
entry["status"] = Status.NEEDS_CUE_SPLIT.value
|
||||
else:
|
||||
for skip_name in cue_result["skip_audio"]:
|
||||
if skip_name in entry["audio_files"]:
|
||||
entry["audio_files"][skip_name]["status"] = Status.SKIPPED.value
|
||||
entry["audio_files"][skip_name]["error"] = "CUE 引用的整轨, 已有分轨故跳过"
|
||||
entry["status"] = Status.ANALYZED.value
|
||||
elif audio_paths:
|
||||
entry["status"] = Status.ANALYZED.value
|
||||
else:
|
||||
if entry["asset_files"]:
|
||||
entry["status"] = Status.ANALYZED.value
|
||||
else:
|
||||
entry["status"] = Status.SKIPPED.value
|
||||
|
||||
return entry
|
||||
|
||||
|
||||
def merge_entry(existing: dict, fresh: dict) -> dict:
|
||||
for name, af in fresh.get("audio_files", {}).items():
|
||||
old = (existing.get("audio_files") or {}).get(name)
|
||||
if old and old.get("status") == Status.COMPLETED.value:
|
||||
if old.get("size") == af.get("size"):
|
||||
af["status"] = Status.COMPLETED.value
|
||||
af["converted_at"] = old.get("converted_at")
|
||||
af["error"] = None
|
||||
|
||||
for name, ae in fresh.get("asset_files", {}).items():
|
||||
old = (existing.get("asset_files") or {}).get(name)
|
||||
if old and old.get("status") == Status.COMPLETED.value:
|
||||
if old.get("size") == ae.get("size"):
|
||||
ae["status"] = Status.COMPLETED.value
|
||||
ae["copied_at"] = old.get("copied_at")
|
||||
ae["error"] = None
|
||||
|
||||
if existing.get("started_at"):
|
||||
fresh["started_at"] = existing["started_at"]
|
||||
|
||||
_refresh_album_status(fresh)
|
||||
return fresh
|
||||
|
||||
|
||||
def _refresh_album_status(entry: dict) -> None:
|
||||
if entry["status"] in (Status.NEEDS_CUE_SPLIT.value, Status.SKIPPED.value):
|
||||
return
|
||||
file_statuses = []
|
||||
file_statuses.extend(af.get("status") for af in entry.get("audio_files", {}).values())
|
||||
file_statuses.extend(ae.get("status") for ae in entry.get("asset_files", {}).values())
|
||||
if not file_statuses:
|
||||
return
|
||||
unique = set(file_statuses)
|
||||
if unique.issubset({Status.COMPLETED.value, Status.SKIPPED.value}) \
|
||||
and Status.COMPLETED.value in unique:
|
||||
entry["status"] = Status.COMPLETED.value
|
||||
if not entry.get("completed_at"):
|
||||
entry["completed_at"] = _now_iso()
|
||||
elif Status.FAILED.value in unique:
|
||||
entry["status"] = Status.FAILED.value
|
||||
|
||||
|
||||
class Manifest:
|
||||
def __init__(self, path: Path, cfg: Config):
|
||||
self.path = path
|
||||
self.cfg = cfg
|
||||
self.data: dict = {}
|
||||
|
||||
def load_or_init(self) -> None:
|
||||
if self.path.exists() and not self.cfg.force:
|
||||
try:
|
||||
text = self.path.read_text(encoding="utf-8")
|
||||
data = json.loads(text)
|
||||
if not isinstance(data, dict) or not isinstance(data.get("albums"), dict):
|
||||
raise ValueError("bad structure")
|
||||
self.data = data
|
||||
logging.info(f"读取 manifest: {len(self.data['albums'])} 个专辑记录")
|
||||
return
|
||||
except (json.JSONDecodeError, ValueError, OSError) as e:
|
||||
quarantine_corrupt(self.path, str(e))
|
||||
self.data = self._new()
|
||||
|
||||
def _new(self) -> dict:
|
||||
now = _now_iso()
|
||||
return {
|
||||
"manifest_version": MANIFEST_SCHEMA_VERSION,
|
||||
"created_at": now,
|
||||
"updated_at": now,
|
||||
"source_root": str(self.cfg.source_root),
|
||||
"target_root": str(self.cfg.target_root),
|
||||
"config": {
|
||||
"bitrate": self.cfg.bitrate,
|
||||
"id3v2_version": self.cfg.id3v2_version,
|
||||
"output_sample_rate_when_high": self.cfg.output_sample_rate_when_high,
|
||||
},
|
||||
"stats": {},
|
||||
"albums": {},
|
||||
}
|
||||
|
||||
def save(self) -> None:
|
||||
self.data["updated_at"] = _now_iso()
|
||||
self.data["stats"] = self._compute_stats()
|
||||
atomic_write_json(self.path, self.data)
|
||||
|
||||
def _compute_stats(self) -> dict:
|
||||
by_album: dict[str, int] = {}
|
||||
by_audio: dict[str, int] = {}
|
||||
by_asset: dict[str, int] = {}
|
||||
total_audio = 0
|
||||
total_asset = 0
|
||||
for album in self.data.get("albums", {}).values():
|
||||
s = album.get("status", "unknown")
|
||||
by_album[s] = by_album.get(s, 0) + 1
|
||||
for af in album.get("audio_files", {}).values():
|
||||
fs = af.get("status", "unknown")
|
||||
by_audio[fs] = by_audio.get(fs, 0) + 1
|
||||
total_audio += 1
|
||||
for ae in album.get("asset_files", {}).values():
|
||||
fs = ae.get("status", "unknown")
|
||||
by_asset[fs] = by_asset.get(fs, 0) + 1
|
||||
total_asset += 1
|
||||
return {
|
||||
"total_albums": len(self.data.get("albums", {})),
|
||||
"by_album_status": by_album,
|
||||
"by_audio_status": by_audio,
|
||||
"by_asset_status": by_asset,
|
||||
"total_audio_files": total_audio,
|
||||
"total_asset_files": total_asset,
|
||||
}
|
||||
|
||||
|
||||
def analyze_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
|
||||
log.info(f"扫描 {cfg.source_root} ...")
|
||||
all_dirs = find_all_dirs(cfg.source_root)
|
||||
log.info(f"发现 {len(all_dirs)} 个子目录")
|
||||
|
||||
for d in all_dirs:
|
||||
rel = str(d.relative_to(cfg.source_root))
|
||||
existing = manifest.data["albums"].get(rel)
|
||||
if existing and existing.get("status") == Status.COMPLETED.value and not cfg.force:
|
||||
log.debug(f"[已完成 skip] {rel}")
|
||||
continue
|
||||
log.info(f"[分析] {rel}")
|
||||
fresh = analyze_directory(d, cfg)
|
||||
if existing:
|
||||
fresh = merge_entry(existing, fresh)
|
||||
manifest.data["albums"][rel] = fresh
|
||||
|
||||
manifest.save()
|
||||
|
||||
|
||||
def process_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
|
||||
albums = manifest.data["albums"]
|
||||
keys = sorted(albums.keys())
|
||||
total = len(keys)
|
||||
|
||||
for i, rel in enumerate(keys, 1):
|
||||
album = albums[rel]
|
||||
status = album.get("status")
|
||||
|
||||
if status == Status.COMPLETED.value and not cfg.force:
|
||||
log.debug(f"[{i}/{total}] SKIP (已完成): {rel}")
|
||||
continue
|
||||
if status == Status.NEEDS_CUE_SPLIT.value:
|
||||
log.warning(f"[{i}/{total}] SKIP (需 CUE split): {rel}")
|
||||
continue
|
||||
if status == Status.SKIPPED.value:
|
||||
log.debug(f"[{i}/{total}] SKIP (空目录): {rel}")
|
||||
continue
|
||||
|
||||
log.info(f"[{i}/{total}] 处理: {rel}")
|
||||
album["status"] = Status.PROCESSING.value
|
||||
if album.get("started_at") is None:
|
||||
album["started_at"] = _now_iso()
|
||||
manifest.save()
|
||||
|
||||
source_album = cfg.source_root / rel
|
||||
target_album = cfg.target_root / rel
|
||||
|
||||
album_ok = True
|
||||
|
||||
for name, af in album.get("audio_files", {}).items():
|
||||
if af.get("status") == Status.COMPLETED.value and not cfg.force:
|
||||
continue
|
||||
if af.get("status") == Status.SKIPPED.value:
|
||||
continue
|
||||
|
||||
src = source_album / name
|
||||
dst = target_album / af["target_rel"]
|
||||
log.info(f" 转码: {name} ({af.get('type', '?')}, "
|
||||
f"{af.get('sample_rate') or '?'}Hz)")
|
||||
ok, err = transcode_to_mp3(src, dst, af.get("sample_rate"), cfg)
|
||||
if ok:
|
||||
af["status"] = Status.COMPLETED.value
|
||||
af["error"] = None
|
||||
af["converted_at"] = _now_iso()
|
||||
else:
|
||||
af["status"] = Status.FAILED.value
|
||||
af["error"] = err
|
||||
album_ok = False
|
||||
log.error(f" ✗ 失败: {err}")
|
||||
manifest.save()
|
||||
|
||||
for name, ae in album.get("asset_files", {}).items():
|
||||
if ae.get("status") == Status.COMPLETED.value and not cfg.force:
|
||||
continue
|
||||
src = source_album / name
|
||||
dst = target_album / ae["target_rel"]
|
||||
log.info(f" 复制: {name} ({ae.get('type', '?')})")
|
||||
ok, err = copy_asset(src, dst, cfg)
|
||||
if ok:
|
||||
ae["status"] = Status.COMPLETED.value
|
||||
ae["error"] = None
|
||||
ae["copied_at"] = _now_iso()
|
||||
else:
|
||||
ae["status"] = Status.FAILED.value
|
||||
ae["error"] = err
|
||||
album_ok = False
|
||||
log.error(f" ✗ 失败: {err}")
|
||||
manifest.save()
|
||||
|
||||
album["status"] = Status.COMPLETED.value if album_ok else Status.FAILED.value
|
||||
if album_ok:
|
||||
album["completed_at"] = _now_iso()
|
||||
manifest.save()
|
||||
|
||||
|
||||
def print_report(manifest: Manifest, log: logging.Logger) -> None:
|
||||
albums = manifest.data.get("albums", {})
|
||||
stats = manifest.data.get("stats") or {}
|
||||
|
||||
log.info("")
|
||||
log.info("=" * 70)
|
||||
log.info("摘要")
|
||||
log.info("=" * 70)
|
||||
log.info(f"总专辑数: {stats.get('total_albums', 0)}")
|
||||
for k, v in sorted((stats.get("by_album_status") or {}).items()):
|
||||
log.info(f" · album {k}: {v}")
|
||||
log.info(f"音频文件: {stats.get('total_audio_files', 0)}")
|
||||
for k, v in sorted((stats.get("by_audio_status") or {}).items()):
|
||||
log.info(f" · audio {k}: {v}")
|
||||
log.info(f"资源文件: {stats.get('total_asset_files', 0)}")
|
||||
for k, v in sorted((stats.get("by_asset_status") or {}).items()):
|
||||
log.info(f" · asset {k}: {v}")
|
||||
|
||||
needs_split = [(r, a) for r, a in albums.items()
|
||||
if a.get("status") == Status.NEEDS_CUE_SPLIT.value]
|
||||
if needs_split:
|
||||
log.warning("")
|
||||
log.warning(f"[!] {len(needs_split)} 个专辑需先手动 CUE split (已跳过转码):")
|
||||
for r, a in needs_split:
|
||||
log.warning(f" • {r}")
|
||||
for issue in a.get("issues", []):
|
||||
log.warning(f" {issue}")
|
||||
|
||||
failed = [(r, a) for r, a in albums.items()
|
||||
if a.get("status") == Status.FAILED.value]
|
||||
if failed:
|
||||
log.error("")
|
||||
log.error(f"[x] {len(failed)} 个专辑失败 (再次运行会重试):")
|
||||
for r, a in failed:
|
||||
log.error(f" • {r}")
|
||||
audio_errs = [(n, af.get("error"))
|
||||
for n, af in a.get("audio_files", {}).items()
|
||||
if af.get("status") == Status.FAILED.value]
|
||||
asset_errs = [(n, ae.get("error"))
|
||||
for n, ae in a.get("asset_files", {}).items()
|
||||
if ae.get("status") == Status.FAILED.value]
|
||||
for n, e in audio_errs[:3]:
|
||||
log.error(f" ✗ audio {n}: {e}")
|
||||
for n, e in asset_errs[:2]:
|
||||
log.error(f" ✗ asset {n}: {e}")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
ap = argparse.ArgumentParser(
|
||||
description="批量转码无损音乐 (FLAC/WAV/DSF/APE) 为 MP3 320k, 带 manifest 状态追踪与断点续跑.",
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog="""示例:
|
||||
# 最简用法 (位置参数)
|
||||
%(prog)s /mnt/z/test /mnt/x/music/test
|
||||
|
||||
# 仅分析, 生成 manifest (推荐首次运行时用来排查 CUE 整轨等问题)
|
||||
%(prog)s /mnt/z/test /mnt/x/music/test --analyze-only
|
||||
|
||||
# 使用命名参数形式 (与位置参数等价)
|
||||
%(prog)s -s /mnt/z/test -t /mnt/x/music/test
|
||||
|
||||
# 强制重跑 (忽略现有 manifest)
|
||||
%(prog)s /mnt/z/test /mnt/x/music/test --force
|
||||
|
||||
# 使用 256k 比特率
|
||||
%(prog)s /mnt/z/test /mnt/x/music/test -b 256k
|
||||
""",
|
||||
)
|
||||
ap.add_argument("source_pos", nargs="?", metavar="SOURCE",
|
||||
help="源目录 (位置参数, 如 /mnt/z/test). 也可用 --source/-s")
|
||||
ap.add_argument("target_pos", nargs="?", metavar="TARGET",
|
||||
help="目标目录 (位置参数, 如 /mnt/x/music/test). 也可用 --target/-t")
|
||||
ap.add_argument("--source", "-s", dest="source_flag",
|
||||
help="源目录 (等价于第 1 个位置参数)")
|
||||
ap.add_argument("--target", "-t", dest="target_flag",
|
||||
help="目标目录 (等价于第 2 个位置参数)")
|
||||
ap.add_argument("--bitrate", "-b", default="320k",
|
||||
help="MP3 比特率 (默认 320k). 可用 256k/192k 等")
|
||||
ap.add_argument("--analyze-only", action="store_true",
|
||||
help="只分析生成 manifest, 不实际转码")
|
||||
ap.add_argument("--force", action="store_true",
|
||||
help="忽略现有 manifest, 强制重新分析并转码全部")
|
||||
ap.add_argument("--verbose", "-v", action="store_true", help="详细日志")
|
||||
ap.add_argument("--ffmpeg", default="ffmpeg", help="ffmpeg 路径 (默认从 PATH 查)")
|
||||
ap.add_argument("--ffprobe", default="ffprobe", help="ffprobe 路径")
|
||||
ap.add_argument("--timeout", type=int, default=3600,
|
||||
help="单文件转码超时秒数 (默认 3600)")
|
||||
ap.add_argument("--sample-rate-when-high", type=int, default=44100,
|
||||
choices=[44100, 48000],
|
||||
help="输入采样率>48kHz时的输出采样率, 默认 44100 (CD 标准)")
|
||||
args = ap.parse_args()
|
||||
|
||||
source_arg = args.source_pos or args.source_flag
|
||||
target_arg = args.target_pos or args.target_flag
|
||||
if not source_arg or not target_arg:
|
||||
ap.error("必须提供源目录和目标目录 (位置参数 SOURCE TARGET, 或 --source/--target)")
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.DEBUG if args.verbose else logging.INFO,
|
||||
format="%(asctime)s %(levelname)-5s %(message)s",
|
||||
datefmt="%H:%M:%S",
|
||||
)
|
||||
log = logging.getLogger("transcode")
|
||||
|
||||
source = Path(source_arg).resolve()
|
||||
target = Path(target_arg).resolve()
|
||||
|
||||
if not source.exists():
|
||||
log.error(f"源目录不存在: {source}")
|
||||
return 2
|
||||
if not source.is_dir():
|
||||
log.error(f"源路径不是目录: {source}")
|
||||
return 2
|
||||
|
||||
try:
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
except OSError as e:
|
||||
log.error(f"无法创建目标目录 {target}: {e}")
|
||||
return 2
|
||||
|
||||
cfg = Config(
|
||||
source_root=source,
|
||||
target_root=target,
|
||||
bitrate=args.bitrate,
|
||||
force=args.force,
|
||||
analyze_only=args.analyze_only,
|
||||
ffmpeg=args.ffmpeg,
|
||||
ffprobe=args.ffprobe,
|
||||
ffmpeg_timeout_sec=args.timeout,
|
||||
output_sample_rate_when_high=args.sample_rate_when_high,
|
||||
)
|
||||
|
||||
manifest_path = source / MANIFEST_NAME
|
||||
manifest = Manifest(manifest_path, cfg)
|
||||
manifest.load_or_init()
|
||||
|
||||
log.info(f"源目录: {source}")
|
||||
log.info(f"目标目录: {target}")
|
||||
log.info(f"Manifest: {manifest_path}")
|
||||
log.info(f"比特率: {cfg.bitrate}")
|
||||
log.info(f"高采样率降到: {cfg.output_sample_rate_when_high} Hz")
|
||||
log.info(f"模式: {'仅分析' if cfg.analyze_only else '分析 + 转码'}")
|
||||
log.info(f"Resume: {'关闭 (--force)' if cfg.force else '开启 (跳过已完成)'}")
|
||||
log.info("")
|
||||
|
||||
try:
|
||||
analyze_all(cfg, manifest, log)
|
||||
|
||||
if cfg.analyze_only:
|
||||
print_report(manifest, log)
|
||||
log.info("")
|
||||
log.info("仅分析模式完成. 请查看 manifest, 处理 needs_cue_split 的专辑后再跑一次.")
|
||||
return 0
|
||||
|
||||
process_all(cfg, manifest, log)
|
||||
print_report(manifest, log)
|
||||
|
||||
except KeyboardInterrupt:
|
||||
log.warning("")
|
||||
log.warning("用户中断 (Ctrl+C) - 保存 manifest 后退出")
|
||||
try:
|
||||
manifest.save()
|
||||
except OSError as e:
|
||||
log.error(f"保存 manifest 失败: {e}")
|
||||
return 130
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user