transcode_music source code + doc

This commit is contained in:
2026-08-16 15:45:55 +08:00
commit 9286f0b5b2
2 changed files with 1818 additions and 0 deletions

View File

@@ -0,0 +1,933 @@
#!/usr/bin/env python3
"""
transcode_music.py - 批量转码无损音乐 (FLAC/WAV/DSF/APE) 为 MP3 320k CBR.
CLI tool. Walks source dir, transcodes lossless audio to MP3, copies covers/txt/cue,
tracks state in a JSON manifest for resumable runs. Detects CUE+whole-disc albums
that need manual splitting and reports them without attempting transcode.
Dependencies: Python 3.10+, ffmpeg, ffprobe.
Usage:
# Analyze only (recommended first run to catch CUE issues)
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test --analyze-only
# Full run (auto-resumes, skips completed)
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test
# Force re-run (ignore existing manifest)
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test --force
"""
from __future__ import annotations
import argparse
import datetime as dt
import errno
import json
import logging
import os
import re
import shutil
import subprocess
import sys
import threading
from dataclasses import asdict, dataclass
from enum import Enum
from pathlib import Path
from typing import Optional
__version__ = "1.0.0"
MANIFEST_SCHEMA_VERSION = 1
MANIFEST_NAME = ".transcode_manifest.json"
LOSSLESS_EXTS = {".flac", ".wav", ".dsf", ".dff", ".ape"}
LOSSY_EXTS = {".mp3", ".m4a", ".aac", ".ogg", ".opus", ".wma", ".mp2"}
IMAGE_EXTS = {".jpg", ".jpeg", ".png", ".bmp", ".gif", ".webp", ".tif", ".tiff"}
DOCUMENT_EXTS = {".txt", ".log", ".pdf", ".md", ".nfo", ".doc", ".docx", ".rtf"}
CUE_EXTS = {".cue"}
IGNORE_NAMES = {".ds_store", "thumbs.db", "desktop.ini", ".directory"}
IGNORE_PREFIXES = ("._",) # macOS resource forks — not user content
# LAME/MP3 spec: only these sample rates are legal for MP3 output.
# Inputs outside this set (high-res 96/192k, DSD 352.8/705.6k) MUST be resampled.
MP3_SAMPLE_RATES = {8000, 11025, 12000, 16000, 22050, 24000, 32000, 44100, 48000}
class Status(str, Enum):
PENDING = "pending"
ANALYZED = "analyzed"
PROCESSING = "processing"
COMPLETED = "completed"
FAILED = "failed"
NEEDS_CUE_SPLIT = "needs_cue_split"
SKIPPED = "skipped"
@dataclass
class Config:
source_root: Path
target_root: Path
bitrate: str = "320k"
id3v2_version: str = "3"
output_sample_rate_when_high: int = 44100
keep_source_mtime: bool = True
force: bool = False
analyze_only: bool = False
ffmpeg: str = "ffmpeg"
ffprobe: str = "ffprobe"
ffmpeg_timeout_sec: int = 3600
@dataclass
class AudioFile:
type: str
size: int
sample_rate: Optional[int] = None
channels: Optional[int] = None
duration_sec: Optional[float] = None
bits_per_sample: Optional[int] = None
codec: Optional[str] = None
status: str = Status.PENDING.value
target_rel: str = ""
error: Optional[str] = None
converted_at: Optional[str] = None
@dataclass
class AssetFile:
type: str
size: int
status: str = Status.PENDING.value
target_rel: str = ""
error: Optional[str] = None
copied_at: Optional[str] = None
def _now_iso() -> str:
return dt.datetime.now(dt.timezone.utc).astimezone().isoformat(timespec="seconds")
def _to_int(x) -> Optional[int]:
if x in (None, "", "N/A"):
return None
try:
return int(x)
except (TypeError, ValueError):
return None
def _to_float(x) -> Optional[float]:
if x in (None, "", "N/A"):
return None
try:
return float(x)
except (TypeError, ValueError):
return None
def atomic_write_json(path: Path, data: dict) -> None:
payload = json.dumps(data, ensure_ascii=False, indent=2)
tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.tmp")
fd = os.open(str(tmp), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644)
try:
os.write(fd, payload.encode("utf-8"))
try:
os.fsync(fd)
except OSError:
pass # network mounts (CIFS/NFS) may not implement fsync
finally:
os.close(fd)
os.replace(str(tmp), str(path))
try:
dfd = os.open(str(path.parent), os.O_RDONLY)
try:
os.fsync(dfd) # persist the rename itself (durability guarantee)
except OSError:
pass
finally:
os.close(dfd)
except OSError:
pass
def quarantine_corrupt(path: Path, reason: str) -> Path:
ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S")
q = path.with_name(f"{path.name}.corrupt-{ts}")
try:
os.replace(str(path), str(q))
logging.error(f"Manifest 损坏, 已隔离到 {q}: {reason}")
except OSError as e:
logging.error(f"隔离失败 {path}: {e}")
return q
def ffprobe_audio(path: Path, cfg: Config) -> dict:
try:
result = subprocess.run(
[
cfg.ffprobe, "-v", "error",
"-select_streams", "a:0",
"-show_entries", "stream=codec_name,sample_rate,channels,bits_per_raw_sample",
"-show_entries", "format=duration",
"-of", "json",
str(path),
],
capture_output=True,
text=True,
timeout=60,
)
if result.returncode != 0:
return {"error": (result.stderr or "ffprobe failed").strip()[-200:]}
data = json.loads(result.stdout or "{}")
stream = (data.get("streams") or [{}])[0]
fmt = data.get("format") or {}
return {
"codec": stream.get("codec_name"),
"sample_rate": _to_int(stream.get("sample_rate")),
"channels": _to_int(stream.get("channels")),
"duration_sec": _to_float(fmt.get("duration")),
"bits_per_sample": _to_int(stream.get("bits_per_raw_sample")),
}
except subprocess.TimeoutExpired:
return {"error": "ffprobe timeout"}
except (json.JSONDecodeError, OSError) as e:
return {"error": str(e)}
def choose_output_sample_rate(input_rate: Optional[int], prefer_when_high: int) -> Optional[int]:
if input_rate is None:
return prefer_when_high
if input_rate in MP3_SAMPLE_RATES:
return None
return prefer_when_high
def transcode_to_mp3(
src: Path,
dst: Path,
input_sample_rate: Optional[int],
cfg: Config,
) -> tuple[bool, Optional[str]]:
dst.parent.mkdir(parents=True, exist_ok=True)
tmp = dst.with_name(dst.name + ".tmp")
if tmp.exists():
try:
tmp.unlink()
except OSError:
pass
out_rate = choose_output_sample_rate(input_sample_rate, cfg.output_sample_rate_when_high)
cmd = [
cfg.ffmpeg, "-nostdin", "-hide_banner", "-loglevel", "error", "-y",
"-i", str(src),
"-c:a", "libmp3lame",
"-b:a", cfg.bitrate,
]
if out_rate is not None:
cmd += ["-ar", str(out_rate)]
cmd += [
"-map_metadata", "0",
"-id3v2_version", cfg.id3v2_version,
"-map", "0:a",
"-map", "0:v?", # optional embedded cover art
"-c:v", "copy",
"-f", "mp3", # explicit muxer: tmp filename has .tmp suffix, cannot auto-detect
str(tmp),
]
orig_mtime: Optional[float] = None
if cfg.keep_source_mtime:
try:
orig_mtime = src.stat().st_mtime
except OSError:
pass
try:
result = subprocess.run(
cmd, capture_output=True, text=True, timeout=cfg.ffmpeg_timeout_sec,
)
if result.returncode != 0:
stderr = (result.stderr or "").strip()
err_msg = stderr.splitlines()[-1] if stderr else f"ffmpeg exit {result.returncode}"
_try_unlink(tmp)
return False, err_msg[-400:]
if orig_mtime is not None:
try:
os.utime(str(tmp), (orig_mtime, orig_mtime))
except OSError:
pass
try:
os.replace(str(tmp), str(dst)) # same target-drive rename → atomic
except OSError as e:
if e.errno == errno.EXDEV:
shutil.move(str(tmp), str(dst)) # cross-device fallback (rare)
else:
raise
return True, None
except subprocess.TimeoutExpired:
_try_unlink(tmp)
return False, f"ffmpeg 超时 ({cfg.ffmpeg_timeout_sec}s)"
except (OSError, subprocess.SubprocessError) as e:
_try_unlink(tmp)
return False, str(e)
def _try_unlink(p: Path) -> None:
try:
if p.exists():
p.unlink()
except OSError:
pass
def copy_asset(src: Path, dst: Path, _cfg: Config) -> tuple[bool, Optional[str]]:
try:
dst.parent.mkdir(parents=True, exist_ok=True)
src_size = src.stat().st_size
if dst.exists():
try:
if dst.stat().st_size == src_size:
return True, None # size-only idempotency (9p mounts can't preserve mtime)
except OSError:
pass
shutil.copyfile(str(src), str(dst))
try:
src_mtime = src.stat().st_mtime
os.utime(str(dst), (src_mtime, src_mtime))
except OSError:
pass # 9p / SMB mounts often reject chmod/utime; content copy still succeeded
return True, None
except OSError as e:
return False, str(e)
CUE_FILE_RE = re.compile(r'^\s*FILE\s+"([^"]+)"\s+(\w+)', re.IGNORECASE)
CUE_TRACK_RE = re.compile(r'^\s*TRACK\s+(\d+)\s+(\w+)', re.IGNORECASE)
# CUE files from Chinese/Japanese/Korean releases often use legacy encodings.
CUE_ENCODINGS = ("utf-8-sig", "utf-8", "gbk", "big5", "shift_jis", "cp936", "latin1")
def read_cue_text(path: Path) -> Optional[str]:
try:
raw = path.read_bytes()
except OSError:
return None
for enc in CUE_ENCODINGS:
try:
return raw.decode(enc)
except UnicodeDecodeError:
continue
return None
def parse_cue(path: Path) -> Optional[dict]:
text = read_cue_text(path)
if text is None:
return None
files: list[dict] = []
tracks: list[dict] = []
counter = 0
for line in text.splitlines():
m = CUE_FILE_RE.match(line)
if m:
counter += 1
files.append({"name": m.group(1), "format": m.group(2), "counter": counter})
continue
m = CUE_TRACK_RE.match(line)
if m:
tracks.append({"num": int(m.group(1)), "file_counter": counter})
return {"files": files, "tracks": tracks}
def analyze_cue_situation(
cue_files: list[Path],
audio_files_by_name: dict[str, Path],
) -> dict:
"""
Classify CUE + audio combinations. Returns:
{
"needs_split": bool, # True => album must be manually CUE-split before transcode
"skip_audio": set[str], # audio filenames to skip (redundant whole-disc file)
"issues": list[str], # human-readable notes/warnings
}
Whipper counter-pattern logic:
- CUE with 1 FILE and that file is the only audio → whole-disc → needs_split
- CUE with 1 FILE but per-track audio also present → skip the whole-disc file
- CUE with multiple FILEs → per-track index, transcode all
- CUE references missing files → note but continue
"""
result = {"needs_split": False, "skip_audio": set(), "issues": []}
for cue in cue_files:
parsed = parse_cue(cue)
if parsed is None:
result["issues"].append(f"{cue.name}: 无法解析 (编码问题)")
continue
refs = parsed["files"]
num_files = len(refs)
num_tracks = len(parsed["tracks"])
if num_files == 0:
result["issues"].append(f"{cue.name}: 无 FILE entry")
continue
refs_present = [r["name"] for r in refs if r["name"] in audio_files_by_name]
refs_missing = [r["name"] for r in refs if r["name"] not in audio_files_by_name]
if num_files == 1:
ref_name = refs[0]["name"]
if ref_name in audio_files_by_name:
other_audio = [n for n in audio_files_by_name if n != ref_name]
if not other_audio:
result["needs_split"] = True
result["issues"].append(
f"{cue.name}: 整轨型 (单文件 '{ref_name}', {num_tracks} 轨) "
f"— 需先手动 CUE split 再转码"
)
else:
result["skip_audio"].add(ref_name)
result["issues"].append(
f"{cue.name}: 引用 '{ref_name}' 但存在 {len(other_audio)} 个分轨 "
f"→ 跳过 '{ref_name}' 只转分轨"
)
else:
result["issues"].append(
f"{cue.name}: 引用 '{ref_name}' 但文件不存在"
)
else:
if refs_missing:
result["issues"].append(
f"{cue.name}: {len(refs_missing)}/{num_files} 个引用文件缺失 "
f"(例: {refs_missing[:2]})"
)
if len(refs_present) < 2:
result["issues"].append(
f"{cue.name}: 多 FILE 引用但仅 {len(refs_present)} 个存在, 请人工核查"
)
return result
def classify_file(path: Path) -> str:
name = path.name.lower()
if name in IGNORE_NAMES:
return "ignore"
for pref in IGNORE_PREFIXES:
if name.startswith(pref):
return "ignore"
if MANIFEST_NAME.lower() in name:
return "ignore"
ext = path.suffix.lower()
if ext in LOSSLESS_EXTS:
return "audio-lossless"
if ext in LOSSY_EXTS:
return "audio-lossy"
if ext in IMAGE_EXTS:
return "image"
if ext in DOCUMENT_EXTS:
return "document"
if ext in CUE_EXTS:
return "cue"
return "other"
def find_all_dirs(source_root: Path) -> list[Path]:
result = []
for dirpath, dirnames, _ in os.walk(source_root):
dirnames.sort()
for d in dirnames:
result.append(Path(dirpath) / d)
return sorted(result)
def analyze_directory(dir_path: Path, cfg: Config) -> dict:
entry = {
"status": Status.PENDING.value,
"issues": [],
"audio_files": {},
"asset_files": {},
"ignored_files": [],
"has_audio": False,
"started_at": None,
"completed_at": None,
"error": None,
}
try:
items = sorted(dir_path.iterdir())
except OSError as e:
entry["status"] = Status.FAILED.value
entry["error"] = f"无法读取目录: {e}"
return entry
file_paths = [p for p in items if p.is_file()]
audio_paths: list[Path] = []
cue_paths: list[Path] = []
for f in file_paths:
cat = classify_file(f)
try:
stat = f.stat()
except OSError as e:
entry["issues"].append(f"stat '{f.name}' 失败: {e}")
continue
rel = f.name
if cat == "ignore":
entry["ignored_files"].append(rel)
continue
if cat == "audio-lossless":
info = ffprobe_audio(f, cfg)
af = AudioFile(
type=f.suffix.lower().lstrip("."),
size=stat.st_size,
sample_rate=info.get("sample_rate"),
channels=info.get("channels"),
duration_sec=info.get("duration_sec"),
bits_per_sample=info.get("bits_per_sample"),
codec=info.get("codec"),
target_rel=str(Path(rel).with_suffix(".mp3")),
)
if "error" in info:
af.error = info["error"]
entry["issues"].append(f"ffprobe '{rel}': {info['error']}")
entry["audio_files"][rel] = asdict(af)
audio_paths.append(f)
elif cat == "cue":
entry["asset_files"][rel] = asdict(
AssetFile(type="cue", size=stat.st_size, target_rel=rel)
)
cue_paths.append(f)
elif cat in ("image", "document"):
entry["asset_files"][rel] = asdict(
AssetFile(type=cat, size=stat.st_size, target_rel=rel)
)
elif cat == "audio-lossy":
entry["asset_files"][rel] = asdict(
AssetFile(type=f.suffix.lower().lstrip("."), size=stat.st_size, target_rel=rel)
)
else:
entry["asset_files"][rel] = asdict(
AssetFile(type="other", size=stat.st_size, target_rel=rel)
)
entry["has_audio"] = bool(audio_paths)
if cue_paths and audio_paths:
audio_by_name = {p.name: p for p in audio_paths}
cue_result = analyze_cue_situation(cue_paths, audio_by_name)
entry["issues"].extend(cue_result["issues"])
if cue_result["needs_split"]:
entry["status"] = Status.NEEDS_CUE_SPLIT.value
else:
for skip_name in cue_result["skip_audio"]:
if skip_name in entry["audio_files"]:
entry["audio_files"][skip_name]["status"] = Status.SKIPPED.value
entry["audio_files"][skip_name]["error"] = "CUE 引用的整轨, 已有分轨故跳过"
entry["status"] = Status.ANALYZED.value
elif audio_paths:
entry["status"] = Status.ANALYZED.value
else:
if entry["asset_files"]:
entry["status"] = Status.ANALYZED.value
else:
entry["status"] = Status.SKIPPED.value
return entry
def merge_entry(existing: dict, fresh: dict) -> dict:
for name, af in fresh.get("audio_files", {}).items():
old = (existing.get("audio_files") or {}).get(name)
if old and old.get("status") == Status.COMPLETED.value:
if old.get("size") == af.get("size"):
af["status"] = Status.COMPLETED.value
af["converted_at"] = old.get("converted_at")
af["error"] = None
for name, ae in fresh.get("asset_files", {}).items():
old = (existing.get("asset_files") or {}).get(name)
if old and old.get("status") == Status.COMPLETED.value:
if old.get("size") == ae.get("size"):
ae["status"] = Status.COMPLETED.value
ae["copied_at"] = old.get("copied_at")
ae["error"] = None
if existing.get("started_at"):
fresh["started_at"] = existing["started_at"]
_refresh_album_status(fresh)
return fresh
def _refresh_album_status(entry: dict) -> None:
if entry["status"] in (Status.NEEDS_CUE_SPLIT.value, Status.SKIPPED.value):
return
file_statuses = []
file_statuses.extend(af.get("status") for af in entry.get("audio_files", {}).values())
file_statuses.extend(ae.get("status") for ae in entry.get("asset_files", {}).values())
if not file_statuses:
return
unique = set(file_statuses)
if unique.issubset({Status.COMPLETED.value, Status.SKIPPED.value}) \
and Status.COMPLETED.value in unique:
entry["status"] = Status.COMPLETED.value
if not entry.get("completed_at"):
entry["completed_at"] = _now_iso()
elif Status.FAILED.value in unique:
entry["status"] = Status.FAILED.value
class Manifest:
def __init__(self, path: Path, cfg: Config):
self.path = path
self.cfg = cfg
self.data: dict = {}
def load_or_init(self) -> None:
if self.path.exists() and not self.cfg.force:
try:
text = self.path.read_text(encoding="utf-8")
data = json.loads(text)
if not isinstance(data, dict) or not isinstance(data.get("albums"), dict):
raise ValueError("bad structure")
self.data = data
logging.info(f"读取 manifest: {len(self.data['albums'])} 个专辑记录")
return
except (json.JSONDecodeError, ValueError, OSError) as e:
quarantine_corrupt(self.path, str(e))
self.data = self._new()
def _new(self) -> dict:
now = _now_iso()
return {
"manifest_version": MANIFEST_SCHEMA_VERSION,
"created_at": now,
"updated_at": now,
"source_root": str(self.cfg.source_root),
"target_root": str(self.cfg.target_root),
"config": {
"bitrate": self.cfg.bitrate,
"id3v2_version": self.cfg.id3v2_version,
"output_sample_rate_when_high": self.cfg.output_sample_rate_when_high,
},
"stats": {},
"albums": {},
}
def save(self) -> None:
self.data["updated_at"] = _now_iso()
self.data["stats"] = self._compute_stats()
atomic_write_json(self.path, self.data)
def _compute_stats(self) -> dict:
by_album: dict[str, int] = {}
by_audio: dict[str, int] = {}
by_asset: dict[str, int] = {}
total_audio = 0
total_asset = 0
for album in self.data.get("albums", {}).values():
s = album.get("status", "unknown")
by_album[s] = by_album.get(s, 0) + 1
for af in album.get("audio_files", {}).values():
fs = af.get("status", "unknown")
by_audio[fs] = by_audio.get(fs, 0) + 1
total_audio += 1
for ae in album.get("asset_files", {}).values():
fs = ae.get("status", "unknown")
by_asset[fs] = by_asset.get(fs, 0) + 1
total_asset += 1
return {
"total_albums": len(self.data.get("albums", {})),
"by_album_status": by_album,
"by_audio_status": by_audio,
"by_asset_status": by_asset,
"total_audio_files": total_audio,
"total_asset_files": total_asset,
}
def analyze_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
log.info(f"扫描 {cfg.source_root} ...")
all_dirs = find_all_dirs(cfg.source_root)
log.info(f"发现 {len(all_dirs)} 个子目录")
for d in all_dirs:
rel = str(d.relative_to(cfg.source_root))
existing = manifest.data["albums"].get(rel)
if existing and existing.get("status") == Status.COMPLETED.value and not cfg.force:
log.debug(f"[已完成 skip] {rel}")
continue
log.info(f"[分析] {rel}")
fresh = analyze_directory(d, cfg)
if existing:
fresh = merge_entry(existing, fresh)
manifest.data["albums"][rel] = fresh
manifest.save()
def process_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
albums = manifest.data["albums"]
keys = sorted(albums.keys())
total = len(keys)
for i, rel in enumerate(keys, 1):
album = albums[rel]
status = album.get("status")
if status == Status.COMPLETED.value and not cfg.force:
log.debug(f"[{i}/{total}] SKIP (已完成): {rel}")
continue
if status == Status.NEEDS_CUE_SPLIT.value:
log.warning(f"[{i}/{total}] SKIP (需 CUE split): {rel}")
continue
if status == Status.SKIPPED.value:
log.debug(f"[{i}/{total}] SKIP (空目录): {rel}")
continue
log.info(f"[{i}/{total}] 处理: {rel}")
album["status"] = Status.PROCESSING.value
if album.get("started_at") is None:
album["started_at"] = _now_iso()
manifest.save()
source_album = cfg.source_root / rel
target_album = cfg.target_root / rel
album_ok = True
for name, af in album.get("audio_files", {}).items():
if af.get("status") == Status.COMPLETED.value and not cfg.force:
continue
if af.get("status") == Status.SKIPPED.value:
continue
src = source_album / name
dst = target_album / af["target_rel"]
log.info(f" 转码: {name} ({af.get('type', '?')}, "
f"{af.get('sample_rate') or '?'}Hz)")
ok, err = transcode_to_mp3(src, dst, af.get("sample_rate"), cfg)
if ok:
af["status"] = Status.COMPLETED.value
af["error"] = None
af["converted_at"] = _now_iso()
else:
af["status"] = Status.FAILED.value
af["error"] = err
album_ok = False
log.error(f" ✗ 失败: {err}")
manifest.save()
for name, ae in album.get("asset_files", {}).items():
if ae.get("status") == Status.COMPLETED.value and not cfg.force:
continue
src = source_album / name
dst = target_album / ae["target_rel"]
log.info(f" 复制: {name} ({ae.get('type', '?')})")
ok, err = copy_asset(src, dst, cfg)
if ok:
ae["status"] = Status.COMPLETED.value
ae["error"] = None
ae["copied_at"] = _now_iso()
else:
ae["status"] = Status.FAILED.value
ae["error"] = err
album_ok = False
log.error(f" ✗ 失败: {err}")
manifest.save()
album["status"] = Status.COMPLETED.value if album_ok else Status.FAILED.value
if album_ok:
album["completed_at"] = _now_iso()
manifest.save()
def print_report(manifest: Manifest, log: logging.Logger) -> None:
albums = manifest.data.get("albums", {})
stats = manifest.data.get("stats") or {}
log.info("")
log.info("=" * 70)
log.info("摘要")
log.info("=" * 70)
log.info(f"总专辑数: {stats.get('total_albums', 0)}")
for k, v in sorted((stats.get("by_album_status") or {}).items()):
log.info(f" · album {k}: {v}")
log.info(f"音频文件: {stats.get('total_audio_files', 0)}")
for k, v in sorted((stats.get("by_audio_status") or {}).items()):
log.info(f" · audio {k}: {v}")
log.info(f"资源文件: {stats.get('total_asset_files', 0)}")
for k, v in sorted((stats.get("by_asset_status") or {}).items()):
log.info(f" · asset {k}: {v}")
needs_split = [(r, a) for r, a in albums.items()
if a.get("status") == Status.NEEDS_CUE_SPLIT.value]
if needs_split:
log.warning("")
log.warning(f"[!] {len(needs_split)} 个专辑需先手动 CUE split (已跳过转码):")
for r, a in needs_split:
log.warning(f" • {r}")
for issue in a.get("issues", []):
log.warning(f" {issue}")
failed = [(r, a) for r, a in albums.items()
if a.get("status") == Status.FAILED.value]
if failed:
log.error("")
log.error(f"[x] {len(failed)} 个专辑失败 (再次运行会重试):")
for r, a in failed:
log.error(f" • {r}")
audio_errs = [(n, af.get("error"))
for n, af in a.get("audio_files", {}).items()
if af.get("status") == Status.FAILED.value]
asset_errs = [(n, ae.get("error"))
for n, ae in a.get("asset_files", {}).items()
if ae.get("status") == Status.FAILED.value]
for n, e in audio_errs[:3]:
log.error(f" ✗ audio {n}: {e}")
for n, e in asset_errs[:2]:
log.error(f" ✗ asset {n}: {e}")
def main() -> int:
ap = argparse.ArgumentParser(
description="批量转码无损音乐 (FLAC/WAV/DSF/APE) 为 MP3 320k, 带 manifest 状态追踪与断点续跑.",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""示例:
# 最简用法 (位置参数)
%(prog)s /mnt/z/test /mnt/x/music/test
# 仅分析, 生成 manifest (推荐首次运行时用来排查 CUE 整轨等问题)
%(prog)s /mnt/z/test /mnt/x/music/test --analyze-only
# 使用命名参数形式 (与位置参数等价)
%(prog)s -s /mnt/z/test -t /mnt/x/music/test
# 强制重跑 (忽略现有 manifest)
%(prog)s /mnt/z/test /mnt/x/music/test --force
# 使用 256k 比特率
%(prog)s /mnt/z/test /mnt/x/music/test -b 256k
""",
)
ap.add_argument("source_pos", nargs="?", metavar="SOURCE",
help="源目录 (位置参数, 如 /mnt/z/test). 也可用 --source/-s")
ap.add_argument("target_pos", nargs="?", metavar="TARGET",
help="目标目录 (位置参数, 如 /mnt/x/music/test). 也可用 --target/-t")
ap.add_argument("--source", "-s", dest="source_flag",
help="源目录 (等价于第 1 个位置参数)")
ap.add_argument("--target", "-t", dest="target_flag",
help="目标目录 (等价于第 2 个位置参数)")
ap.add_argument("--bitrate", "-b", default="320k",
help="MP3 比特率 (默认 320k). 可用 256k/192k 等")
ap.add_argument("--analyze-only", action="store_true",
help="只分析生成 manifest, 不实际转码")
ap.add_argument("--force", action="store_true",
help="忽略现有 manifest, 强制重新分析并转码全部")
ap.add_argument("--verbose", "-v", action="store_true", help="详细日志")
ap.add_argument("--ffmpeg", default="ffmpeg", help="ffmpeg 路径 (默认从 PATH 查)")
ap.add_argument("--ffprobe", default="ffprobe", help="ffprobe 路径")
ap.add_argument("--timeout", type=int, default=3600,
help="单文件转码超时秒数 (默认 3600)")
ap.add_argument("--sample-rate-when-high", type=int, default=44100,
choices=[44100, 48000],
help="输入采样率>48kHz时的输出采样率, 默认 44100 (CD 标准)")
args = ap.parse_args()
source_arg = args.source_pos or args.source_flag
target_arg = args.target_pos or args.target_flag
if not source_arg or not target_arg:
ap.error("必须提供源目录和目标目录 (位置参数 SOURCE TARGET, 或 --source/--target)")
logging.basicConfig(
level=logging.DEBUG if args.verbose else logging.INFO,
format="%(asctime)s %(levelname)-5s %(message)s",
datefmt="%H:%M:%S",
)
log = logging.getLogger("transcode")
source = Path(source_arg).resolve()
target = Path(target_arg).resolve()
if not source.exists():
log.error(f"源目录不存在: {source}")
return 2
if not source.is_dir():
log.error(f"源路径不是目录: {source}")
return 2
try:
target.mkdir(parents=True, exist_ok=True)
except OSError as e:
log.error(f"无法创建目标目录 {target}: {e}")
return 2
cfg = Config(
source_root=source,
target_root=target,
bitrate=args.bitrate,
force=args.force,
analyze_only=args.analyze_only,
ffmpeg=args.ffmpeg,
ffprobe=args.ffprobe,
ffmpeg_timeout_sec=args.timeout,
output_sample_rate_when_high=args.sample_rate_when_high,
)
manifest_path = source / MANIFEST_NAME
manifest = Manifest(manifest_path, cfg)
manifest.load_or_init()
log.info(f"源目录: {source}")
log.info(f"目标目录: {target}")
log.info(f"Manifest: {manifest_path}")
log.info(f"比特率: {cfg.bitrate}")
log.info(f"高采样率降到: {cfg.output_sample_rate_when_high} Hz")
log.info(f"模式: {'仅分析' if cfg.analyze_only else '分析 + 转码'}")
log.info(f"Resume: {'关闭 (--force)' if cfg.force else '开启 (跳过已完成)'}")
log.info("")
try:
analyze_all(cfg, manifest, log)
if cfg.analyze_only:
print_report(manifest, log)
log.info("")
log.info("仅分析模式完成. 请查看 manifest, 处理 needs_cue_split 的专辑后再跑一次.")
return 0
process_all(cfg, manifest, log)
print_report(manifest, log)
except KeyboardInterrupt:
log.warning("")
log.warning("用户中断 (Ctrl+C) - 保存 manifest 后退出")
try:
manifest.save()
except OSError as e:
log.error(f"保存 manifest 失败: {e}")
return 130
return 0
if __name__ == "__main__":
sys.exit(main())