934 lines
32 KiB
Python
934 lines
32 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
transcode_music.py - 批量转码无损音乐 (FLAC/WAV/DSF/APE) 为 MP3 320k CBR.
|
|
|
|
CLI tool. Walks source dir, transcodes lossless audio to MP3, copies covers/txt/cue,
|
|
tracks state in a JSON manifest for resumable runs. Detects CUE+whole-disc albums
|
|
that need manual splitting and reports them without attempting transcode.
|
|
|
|
Dependencies: Python 3.10+, ffmpeg, ffprobe.
|
|
|
|
Usage:
|
|
# Analyze only (recommended first run to catch CUE issues)
|
|
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test --analyze-only
|
|
|
|
# Full run (auto-resumes, skips completed)
|
|
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test
|
|
|
|
# Force re-run (ignore existing manifest)
|
|
python3 transcode_music.py -s /mnt/z/test -t /mnt/x/music/test --force
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import datetime as dt
|
|
import errno
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import threading
|
|
from dataclasses import asdict, dataclass
|
|
from enum import Enum
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
__version__ = "1.0.0"
|
|
|
|
MANIFEST_SCHEMA_VERSION = 1
|
|
MANIFEST_NAME = ".transcode_manifest.json"
|
|
|
|
LOSSLESS_EXTS = {".flac", ".wav", ".dsf", ".dff", ".ape"}
|
|
LOSSY_EXTS = {".mp3", ".m4a", ".aac", ".ogg", ".opus", ".wma", ".mp2"}
|
|
IMAGE_EXTS = {".jpg", ".jpeg", ".png", ".bmp", ".gif", ".webp", ".tif", ".tiff"}
|
|
DOCUMENT_EXTS = {".txt", ".log", ".pdf", ".md", ".nfo", ".doc", ".docx", ".rtf"}
|
|
CUE_EXTS = {".cue"}
|
|
|
|
IGNORE_NAMES = {".ds_store", "thumbs.db", "desktop.ini", ".directory"}
|
|
IGNORE_PREFIXES = ("._",) # macOS resource forks — not user content
|
|
|
|
# LAME/MP3 spec: only these sample rates are legal for MP3 output.
|
|
# Inputs outside this set (high-res 96/192k, DSD 352.8/705.6k) MUST be resampled.
|
|
MP3_SAMPLE_RATES = {8000, 11025, 12000, 16000, 22050, 24000, 32000, 44100, 48000}
|
|
|
|
|
|
class Status(str, Enum):
|
|
PENDING = "pending"
|
|
ANALYZED = "analyzed"
|
|
PROCESSING = "processing"
|
|
COMPLETED = "completed"
|
|
FAILED = "failed"
|
|
NEEDS_CUE_SPLIT = "needs_cue_split"
|
|
SKIPPED = "skipped"
|
|
|
|
|
|
@dataclass
|
|
class Config:
|
|
source_root: Path
|
|
target_root: Path
|
|
bitrate: str = "320k"
|
|
id3v2_version: str = "3"
|
|
output_sample_rate_when_high: int = 44100
|
|
keep_source_mtime: bool = True
|
|
force: bool = False
|
|
analyze_only: bool = False
|
|
ffmpeg: str = "ffmpeg"
|
|
ffprobe: str = "ffprobe"
|
|
ffmpeg_timeout_sec: int = 3600
|
|
|
|
|
|
@dataclass
|
|
class AudioFile:
|
|
type: str
|
|
size: int
|
|
sample_rate: Optional[int] = None
|
|
channels: Optional[int] = None
|
|
duration_sec: Optional[float] = None
|
|
bits_per_sample: Optional[int] = None
|
|
codec: Optional[str] = None
|
|
status: str = Status.PENDING.value
|
|
target_rel: str = ""
|
|
error: Optional[str] = None
|
|
converted_at: Optional[str] = None
|
|
|
|
|
|
@dataclass
|
|
class AssetFile:
|
|
type: str
|
|
size: int
|
|
status: str = Status.PENDING.value
|
|
target_rel: str = ""
|
|
error: Optional[str] = None
|
|
copied_at: Optional[str] = None
|
|
|
|
|
|
def _now_iso() -> str:
|
|
return dt.datetime.now(dt.timezone.utc).astimezone().isoformat(timespec="seconds")
|
|
|
|
|
|
def _to_int(x) -> Optional[int]:
|
|
if x in (None, "", "N/A"):
|
|
return None
|
|
try:
|
|
return int(x)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
|
|
def _to_float(x) -> Optional[float]:
|
|
if x in (None, "", "N/A"):
|
|
return None
|
|
try:
|
|
return float(x)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
|
|
def atomic_write_json(path: Path, data: dict) -> None:
|
|
payload = json.dumps(data, ensure_ascii=False, indent=2)
|
|
tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.tmp")
|
|
|
|
fd = os.open(str(tmp), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644)
|
|
try:
|
|
os.write(fd, payload.encode("utf-8"))
|
|
try:
|
|
os.fsync(fd)
|
|
except OSError:
|
|
pass # network mounts (CIFS/NFS) may not implement fsync
|
|
finally:
|
|
os.close(fd)
|
|
|
|
os.replace(str(tmp), str(path))
|
|
|
|
try:
|
|
dfd = os.open(str(path.parent), os.O_RDONLY)
|
|
try:
|
|
os.fsync(dfd) # persist the rename itself (durability guarantee)
|
|
except OSError:
|
|
pass
|
|
finally:
|
|
os.close(dfd)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def quarantine_corrupt(path: Path, reason: str) -> Path:
|
|
ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S")
|
|
q = path.with_name(f"{path.name}.corrupt-{ts}")
|
|
try:
|
|
os.replace(str(path), str(q))
|
|
logging.error(f"Manifest 损坏, 已隔离到 {q}: {reason}")
|
|
except OSError as e:
|
|
logging.error(f"隔离失败 {path}: {e}")
|
|
return q
|
|
|
|
|
|
def ffprobe_audio(path: Path, cfg: Config) -> dict:
|
|
try:
|
|
result = subprocess.run(
|
|
[
|
|
cfg.ffprobe, "-v", "error",
|
|
"-select_streams", "a:0",
|
|
"-show_entries", "stream=codec_name,sample_rate,channels,bits_per_raw_sample",
|
|
"-show_entries", "format=duration",
|
|
"-of", "json",
|
|
str(path),
|
|
],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=60,
|
|
)
|
|
if result.returncode != 0:
|
|
return {"error": (result.stderr or "ffprobe failed").strip()[-200:]}
|
|
data = json.loads(result.stdout or "{}")
|
|
stream = (data.get("streams") or [{}])[0]
|
|
fmt = data.get("format") or {}
|
|
return {
|
|
"codec": stream.get("codec_name"),
|
|
"sample_rate": _to_int(stream.get("sample_rate")),
|
|
"channels": _to_int(stream.get("channels")),
|
|
"duration_sec": _to_float(fmt.get("duration")),
|
|
"bits_per_sample": _to_int(stream.get("bits_per_raw_sample")),
|
|
}
|
|
except subprocess.TimeoutExpired:
|
|
return {"error": "ffprobe timeout"}
|
|
except (json.JSONDecodeError, OSError) as e:
|
|
return {"error": str(e)}
|
|
|
|
|
|
def choose_output_sample_rate(input_rate: Optional[int], prefer_when_high: int) -> Optional[int]:
|
|
if input_rate is None:
|
|
return prefer_when_high
|
|
if input_rate in MP3_SAMPLE_RATES:
|
|
return None
|
|
return prefer_when_high
|
|
|
|
|
|
def transcode_to_mp3(
|
|
src: Path,
|
|
dst: Path,
|
|
input_sample_rate: Optional[int],
|
|
cfg: Config,
|
|
) -> tuple[bool, Optional[str]]:
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
tmp = dst.with_name(dst.name + ".tmp")
|
|
if tmp.exists():
|
|
try:
|
|
tmp.unlink()
|
|
except OSError:
|
|
pass
|
|
|
|
out_rate = choose_output_sample_rate(input_sample_rate, cfg.output_sample_rate_when_high)
|
|
|
|
cmd = [
|
|
cfg.ffmpeg, "-nostdin", "-hide_banner", "-loglevel", "error", "-y",
|
|
"-i", str(src),
|
|
"-c:a", "libmp3lame",
|
|
"-b:a", cfg.bitrate,
|
|
]
|
|
if out_rate is not None:
|
|
cmd += ["-ar", str(out_rate)]
|
|
cmd += [
|
|
"-map_metadata", "0",
|
|
"-id3v2_version", cfg.id3v2_version,
|
|
"-map", "0:a",
|
|
"-map", "0:v?", # optional embedded cover art
|
|
"-c:v", "copy",
|
|
"-f", "mp3", # explicit muxer: tmp filename has .tmp suffix, cannot auto-detect
|
|
str(tmp),
|
|
]
|
|
|
|
orig_mtime: Optional[float] = None
|
|
if cfg.keep_source_mtime:
|
|
try:
|
|
orig_mtime = src.stat().st_mtime
|
|
except OSError:
|
|
pass
|
|
|
|
try:
|
|
result = subprocess.run(
|
|
cmd, capture_output=True, text=True, timeout=cfg.ffmpeg_timeout_sec,
|
|
)
|
|
if result.returncode != 0:
|
|
stderr = (result.stderr or "").strip()
|
|
err_msg = stderr.splitlines()[-1] if stderr else f"ffmpeg exit {result.returncode}"
|
|
_try_unlink(tmp)
|
|
return False, err_msg[-400:]
|
|
|
|
if orig_mtime is not None:
|
|
try:
|
|
os.utime(str(tmp), (orig_mtime, orig_mtime))
|
|
except OSError:
|
|
pass
|
|
|
|
try:
|
|
os.replace(str(tmp), str(dst)) # same target-drive rename → atomic
|
|
except OSError as e:
|
|
if e.errno == errno.EXDEV:
|
|
shutil.move(str(tmp), str(dst)) # cross-device fallback (rare)
|
|
else:
|
|
raise
|
|
return True, None
|
|
|
|
except subprocess.TimeoutExpired:
|
|
_try_unlink(tmp)
|
|
return False, f"ffmpeg 超时 ({cfg.ffmpeg_timeout_sec}s)"
|
|
except (OSError, subprocess.SubprocessError) as e:
|
|
_try_unlink(tmp)
|
|
return False, str(e)
|
|
|
|
|
|
def _try_unlink(p: Path) -> None:
|
|
try:
|
|
if p.exists():
|
|
p.unlink()
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def copy_asset(src: Path, dst: Path, _cfg: Config) -> tuple[bool, Optional[str]]:
|
|
try:
|
|
dst.parent.mkdir(parents=True, exist_ok=True)
|
|
src_size = src.stat().st_size
|
|
if dst.exists():
|
|
try:
|
|
if dst.stat().st_size == src_size:
|
|
return True, None # size-only idempotency (9p mounts can't preserve mtime)
|
|
except OSError:
|
|
pass
|
|
|
|
shutil.copyfile(str(src), str(dst))
|
|
try:
|
|
src_mtime = src.stat().st_mtime
|
|
os.utime(str(dst), (src_mtime, src_mtime))
|
|
except OSError:
|
|
pass # 9p / SMB mounts often reject chmod/utime; content copy still succeeded
|
|
return True, None
|
|
except OSError as e:
|
|
return False, str(e)
|
|
|
|
|
|
CUE_FILE_RE = re.compile(r'^\s*FILE\s+"([^"]+)"\s+(\w+)', re.IGNORECASE)
|
|
CUE_TRACK_RE = re.compile(r'^\s*TRACK\s+(\d+)\s+(\w+)', re.IGNORECASE)
|
|
|
|
# CUE files from Chinese/Japanese/Korean releases often use legacy encodings.
|
|
CUE_ENCODINGS = ("utf-8-sig", "utf-8", "gbk", "big5", "shift_jis", "cp936", "latin1")
|
|
|
|
|
|
def read_cue_text(path: Path) -> Optional[str]:
|
|
try:
|
|
raw = path.read_bytes()
|
|
except OSError:
|
|
return None
|
|
for enc in CUE_ENCODINGS:
|
|
try:
|
|
return raw.decode(enc)
|
|
except UnicodeDecodeError:
|
|
continue
|
|
return None
|
|
|
|
|
|
def parse_cue(path: Path) -> Optional[dict]:
|
|
text = read_cue_text(path)
|
|
if text is None:
|
|
return None
|
|
files: list[dict] = []
|
|
tracks: list[dict] = []
|
|
counter = 0
|
|
for line in text.splitlines():
|
|
m = CUE_FILE_RE.match(line)
|
|
if m:
|
|
counter += 1
|
|
files.append({"name": m.group(1), "format": m.group(2), "counter": counter})
|
|
continue
|
|
m = CUE_TRACK_RE.match(line)
|
|
if m:
|
|
tracks.append({"num": int(m.group(1)), "file_counter": counter})
|
|
return {"files": files, "tracks": tracks}
|
|
|
|
|
|
def analyze_cue_situation(
|
|
cue_files: list[Path],
|
|
audio_files_by_name: dict[str, Path],
|
|
) -> dict:
|
|
"""
|
|
Classify CUE + audio combinations. Returns:
|
|
{
|
|
"needs_split": bool, # True => album must be manually CUE-split before transcode
|
|
"skip_audio": set[str], # audio filenames to skip (redundant whole-disc file)
|
|
"issues": list[str], # human-readable notes/warnings
|
|
}
|
|
|
|
Whipper counter-pattern logic:
|
|
- CUE with 1 FILE and that file is the only audio → whole-disc → needs_split
|
|
- CUE with 1 FILE but per-track audio also present → skip the whole-disc file
|
|
- CUE with multiple FILEs → per-track index, transcode all
|
|
- CUE references missing files → note but continue
|
|
"""
|
|
result = {"needs_split": False, "skip_audio": set(), "issues": []}
|
|
|
|
for cue in cue_files:
|
|
parsed = parse_cue(cue)
|
|
if parsed is None:
|
|
result["issues"].append(f"{cue.name}: 无法解析 (编码问题)")
|
|
continue
|
|
|
|
refs = parsed["files"]
|
|
num_files = len(refs)
|
|
num_tracks = len(parsed["tracks"])
|
|
|
|
if num_files == 0:
|
|
result["issues"].append(f"{cue.name}: 无 FILE entry")
|
|
continue
|
|
|
|
refs_present = [r["name"] for r in refs if r["name"] in audio_files_by_name]
|
|
refs_missing = [r["name"] for r in refs if r["name"] not in audio_files_by_name]
|
|
|
|
if num_files == 1:
|
|
ref_name = refs[0]["name"]
|
|
if ref_name in audio_files_by_name:
|
|
other_audio = [n for n in audio_files_by_name if n != ref_name]
|
|
if not other_audio:
|
|
result["needs_split"] = True
|
|
result["issues"].append(
|
|
f"{cue.name}: 整轨型 (单文件 '{ref_name}', {num_tracks} 轨) "
|
|
f"— 需先手动 CUE split 再转码"
|
|
)
|
|
else:
|
|
result["skip_audio"].add(ref_name)
|
|
result["issues"].append(
|
|
f"{cue.name}: 引用 '{ref_name}' 但存在 {len(other_audio)} 个分轨 "
|
|
f"→ 跳过 '{ref_name}' 只转分轨"
|
|
)
|
|
else:
|
|
result["issues"].append(
|
|
f"{cue.name}: 引用 '{ref_name}' 但文件不存在"
|
|
)
|
|
else:
|
|
if refs_missing:
|
|
result["issues"].append(
|
|
f"{cue.name}: {len(refs_missing)}/{num_files} 个引用文件缺失 "
|
|
f"(例: {refs_missing[:2]})"
|
|
)
|
|
if len(refs_present) < 2:
|
|
result["issues"].append(
|
|
f"{cue.name}: 多 FILE 引用但仅 {len(refs_present)} 个存在, 请人工核查"
|
|
)
|
|
|
|
return result
|
|
|
|
|
|
def classify_file(path: Path) -> str:
|
|
name = path.name.lower()
|
|
if name in IGNORE_NAMES:
|
|
return "ignore"
|
|
for pref in IGNORE_PREFIXES:
|
|
if name.startswith(pref):
|
|
return "ignore"
|
|
if MANIFEST_NAME.lower() in name:
|
|
return "ignore"
|
|
|
|
ext = path.suffix.lower()
|
|
if ext in LOSSLESS_EXTS:
|
|
return "audio-lossless"
|
|
if ext in LOSSY_EXTS:
|
|
return "audio-lossy"
|
|
if ext in IMAGE_EXTS:
|
|
return "image"
|
|
if ext in DOCUMENT_EXTS:
|
|
return "document"
|
|
if ext in CUE_EXTS:
|
|
return "cue"
|
|
return "other"
|
|
|
|
|
|
def find_all_dirs(source_root: Path) -> list[Path]:
|
|
result = []
|
|
for dirpath, dirnames, _ in os.walk(source_root):
|
|
dirnames.sort()
|
|
for d in dirnames:
|
|
result.append(Path(dirpath) / d)
|
|
return sorted(result)
|
|
|
|
|
|
def analyze_directory(dir_path: Path, cfg: Config) -> dict:
|
|
entry = {
|
|
"status": Status.PENDING.value,
|
|
"issues": [],
|
|
"audio_files": {},
|
|
"asset_files": {},
|
|
"ignored_files": [],
|
|
"has_audio": False,
|
|
"started_at": None,
|
|
"completed_at": None,
|
|
"error": None,
|
|
}
|
|
|
|
try:
|
|
items = sorted(dir_path.iterdir())
|
|
except OSError as e:
|
|
entry["status"] = Status.FAILED.value
|
|
entry["error"] = f"无法读取目录: {e}"
|
|
return entry
|
|
|
|
file_paths = [p for p in items if p.is_file()]
|
|
|
|
audio_paths: list[Path] = []
|
|
cue_paths: list[Path] = []
|
|
|
|
for f in file_paths:
|
|
cat = classify_file(f)
|
|
try:
|
|
stat = f.stat()
|
|
except OSError as e:
|
|
entry["issues"].append(f"stat '{f.name}' 失败: {e}")
|
|
continue
|
|
|
|
rel = f.name
|
|
|
|
if cat == "ignore":
|
|
entry["ignored_files"].append(rel)
|
|
continue
|
|
|
|
if cat == "audio-lossless":
|
|
info = ffprobe_audio(f, cfg)
|
|
af = AudioFile(
|
|
type=f.suffix.lower().lstrip("."),
|
|
size=stat.st_size,
|
|
sample_rate=info.get("sample_rate"),
|
|
channels=info.get("channels"),
|
|
duration_sec=info.get("duration_sec"),
|
|
bits_per_sample=info.get("bits_per_sample"),
|
|
codec=info.get("codec"),
|
|
target_rel=str(Path(rel).with_suffix(".mp3")),
|
|
)
|
|
if "error" in info:
|
|
af.error = info["error"]
|
|
entry["issues"].append(f"ffprobe '{rel}': {info['error']}")
|
|
entry["audio_files"][rel] = asdict(af)
|
|
audio_paths.append(f)
|
|
elif cat == "cue":
|
|
entry["asset_files"][rel] = asdict(
|
|
AssetFile(type="cue", size=stat.st_size, target_rel=rel)
|
|
)
|
|
cue_paths.append(f)
|
|
elif cat in ("image", "document"):
|
|
entry["asset_files"][rel] = asdict(
|
|
AssetFile(type=cat, size=stat.st_size, target_rel=rel)
|
|
)
|
|
elif cat == "audio-lossy":
|
|
entry["asset_files"][rel] = asdict(
|
|
AssetFile(type=f.suffix.lower().lstrip("."), size=stat.st_size, target_rel=rel)
|
|
)
|
|
else:
|
|
entry["asset_files"][rel] = asdict(
|
|
AssetFile(type="other", size=stat.st_size, target_rel=rel)
|
|
)
|
|
|
|
entry["has_audio"] = bool(audio_paths)
|
|
|
|
if cue_paths and audio_paths:
|
|
audio_by_name = {p.name: p for p in audio_paths}
|
|
cue_result = analyze_cue_situation(cue_paths, audio_by_name)
|
|
entry["issues"].extend(cue_result["issues"])
|
|
if cue_result["needs_split"]:
|
|
entry["status"] = Status.NEEDS_CUE_SPLIT.value
|
|
else:
|
|
for skip_name in cue_result["skip_audio"]:
|
|
if skip_name in entry["audio_files"]:
|
|
entry["audio_files"][skip_name]["status"] = Status.SKIPPED.value
|
|
entry["audio_files"][skip_name]["error"] = "CUE 引用的整轨, 已有分轨故跳过"
|
|
entry["status"] = Status.ANALYZED.value
|
|
elif audio_paths:
|
|
entry["status"] = Status.ANALYZED.value
|
|
else:
|
|
if entry["asset_files"]:
|
|
entry["status"] = Status.ANALYZED.value
|
|
else:
|
|
entry["status"] = Status.SKIPPED.value
|
|
|
|
return entry
|
|
|
|
|
|
def merge_entry(existing: dict, fresh: dict) -> dict:
|
|
for name, af in fresh.get("audio_files", {}).items():
|
|
old = (existing.get("audio_files") or {}).get(name)
|
|
if old and old.get("status") == Status.COMPLETED.value:
|
|
if old.get("size") == af.get("size"):
|
|
af["status"] = Status.COMPLETED.value
|
|
af["converted_at"] = old.get("converted_at")
|
|
af["error"] = None
|
|
|
|
for name, ae in fresh.get("asset_files", {}).items():
|
|
old = (existing.get("asset_files") or {}).get(name)
|
|
if old and old.get("status") == Status.COMPLETED.value:
|
|
if old.get("size") == ae.get("size"):
|
|
ae["status"] = Status.COMPLETED.value
|
|
ae["copied_at"] = old.get("copied_at")
|
|
ae["error"] = None
|
|
|
|
if existing.get("started_at"):
|
|
fresh["started_at"] = existing["started_at"]
|
|
|
|
_refresh_album_status(fresh)
|
|
return fresh
|
|
|
|
|
|
def _refresh_album_status(entry: dict) -> None:
|
|
if entry["status"] in (Status.NEEDS_CUE_SPLIT.value, Status.SKIPPED.value):
|
|
return
|
|
file_statuses = []
|
|
file_statuses.extend(af.get("status") for af in entry.get("audio_files", {}).values())
|
|
file_statuses.extend(ae.get("status") for ae in entry.get("asset_files", {}).values())
|
|
if not file_statuses:
|
|
return
|
|
unique = set(file_statuses)
|
|
if unique.issubset({Status.COMPLETED.value, Status.SKIPPED.value}) \
|
|
and Status.COMPLETED.value in unique:
|
|
entry["status"] = Status.COMPLETED.value
|
|
if not entry.get("completed_at"):
|
|
entry["completed_at"] = _now_iso()
|
|
elif Status.FAILED.value in unique:
|
|
entry["status"] = Status.FAILED.value
|
|
|
|
|
|
class Manifest:
|
|
def __init__(self, path: Path, cfg: Config):
|
|
self.path = path
|
|
self.cfg = cfg
|
|
self.data: dict = {}
|
|
|
|
def load_or_init(self) -> None:
|
|
if self.path.exists() and not self.cfg.force:
|
|
try:
|
|
text = self.path.read_text(encoding="utf-8")
|
|
data = json.loads(text)
|
|
if not isinstance(data, dict) or not isinstance(data.get("albums"), dict):
|
|
raise ValueError("bad structure")
|
|
self.data = data
|
|
logging.info(f"读取 manifest: {len(self.data['albums'])} 个专辑记录")
|
|
return
|
|
except (json.JSONDecodeError, ValueError, OSError) as e:
|
|
quarantine_corrupt(self.path, str(e))
|
|
self.data = self._new()
|
|
|
|
def _new(self) -> dict:
|
|
now = _now_iso()
|
|
return {
|
|
"manifest_version": MANIFEST_SCHEMA_VERSION,
|
|
"created_at": now,
|
|
"updated_at": now,
|
|
"source_root": str(self.cfg.source_root),
|
|
"target_root": str(self.cfg.target_root),
|
|
"config": {
|
|
"bitrate": self.cfg.bitrate,
|
|
"id3v2_version": self.cfg.id3v2_version,
|
|
"output_sample_rate_when_high": self.cfg.output_sample_rate_when_high,
|
|
},
|
|
"stats": {},
|
|
"albums": {},
|
|
}
|
|
|
|
def save(self) -> None:
|
|
self.data["updated_at"] = _now_iso()
|
|
self.data["stats"] = self._compute_stats()
|
|
atomic_write_json(self.path, self.data)
|
|
|
|
def _compute_stats(self) -> dict:
|
|
by_album: dict[str, int] = {}
|
|
by_audio: dict[str, int] = {}
|
|
by_asset: dict[str, int] = {}
|
|
total_audio = 0
|
|
total_asset = 0
|
|
for album in self.data.get("albums", {}).values():
|
|
s = album.get("status", "unknown")
|
|
by_album[s] = by_album.get(s, 0) + 1
|
|
for af in album.get("audio_files", {}).values():
|
|
fs = af.get("status", "unknown")
|
|
by_audio[fs] = by_audio.get(fs, 0) + 1
|
|
total_audio += 1
|
|
for ae in album.get("asset_files", {}).values():
|
|
fs = ae.get("status", "unknown")
|
|
by_asset[fs] = by_asset.get(fs, 0) + 1
|
|
total_asset += 1
|
|
return {
|
|
"total_albums": len(self.data.get("albums", {})),
|
|
"by_album_status": by_album,
|
|
"by_audio_status": by_audio,
|
|
"by_asset_status": by_asset,
|
|
"total_audio_files": total_audio,
|
|
"total_asset_files": total_asset,
|
|
}
|
|
|
|
|
|
def analyze_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
|
|
log.info(f"扫描 {cfg.source_root} ...")
|
|
all_dirs = find_all_dirs(cfg.source_root)
|
|
log.info(f"发现 {len(all_dirs)} 个子目录")
|
|
|
|
for d in all_dirs:
|
|
rel = str(d.relative_to(cfg.source_root))
|
|
existing = manifest.data["albums"].get(rel)
|
|
if existing and existing.get("status") == Status.COMPLETED.value and not cfg.force:
|
|
log.debug(f"[已完成 skip] {rel}")
|
|
continue
|
|
log.info(f"[分析] {rel}")
|
|
fresh = analyze_directory(d, cfg)
|
|
if existing:
|
|
fresh = merge_entry(existing, fresh)
|
|
manifest.data["albums"][rel] = fresh
|
|
|
|
manifest.save()
|
|
|
|
|
|
def process_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
|
|
albums = manifest.data["albums"]
|
|
keys = sorted(albums.keys())
|
|
total = len(keys)
|
|
|
|
for i, rel in enumerate(keys, 1):
|
|
album = albums[rel]
|
|
status = album.get("status")
|
|
|
|
if status == Status.COMPLETED.value and not cfg.force:
|
|
log.debug(f"[{i}/{total}] SKIP (已完成): {rel}")
|
|
continue
|
|
if status == Status.NEEDS_CUE_SPLIT.value:
|
|
log.warning(f"[{i}/{total}] SKIP (需 CUE split): {rel}")
|
|
continue
|
|
if status == Status.SKIPPED.value:
|
|
log.debug(f"[{i}/{total}] SKIP (空目录): {rel}")
|
|
continue
|
|
|
|
log.info(f"[{i}/{total}] 处理: {rel}")
|
|
album["status"] = Status.PROCESSING.value
|
|
if album.get("started_at") is None:
|
|
album["started_at"] = _now_iso()
|
|
manifest.save()
|
|
|
|
source_album = cfg.source_root / rel
|
|
target_album = cfg.target_root / rel
|
|
|
|
album_ok = True
|
|
|
|
for name, af in album.get("audio_files", {}).items():
|
|
if af.get("status") == Status.COMPLETED.value and not cfg.force:
|
|
continue
|
|
if af.get("status") == Status.SKIPPED.value:
|
|
continue
|
|
|
|
src = source_album / name
|
|
dst = target_album / af["target_rel"]
|
|
log.info(f" 转码: {name} ({af.get('type', '?')}, "
|
|
f"{af.get('sample_rate') or '?'}Hz)")
|
|
ok, err = transcode_to_mp3(src, dst, af.get("sample_rate"), cfg)
|
|
if ok:
|
|
af["status"] = Status.COMPLETED.value
|
|
af["error"] = None
|
|
af["converted_at"] = _now_iso()
|
|
else:
|
|
af["status"] = Status.FAILED.value
|
|
af["error"] = err
|
|
album_ok = False
|
|
log.error(f" ✗ 失败: {err}")
|
|
manifest.save()
|
|
|
|
for name, ae in album.get("asset_files", {}).items():
|
|
if ae.get("status") == Status.COMPLETED.value and not cfg.force:
|
|
continue
|
|
src = source_album / name
|
|
dst = target_album / ae["target_rel"]
|
|
log.info(f" 复制: {name} ({ae.get('type', '?')})")
|
|
ok, err = copy_asset(src, dst, cfg)
|
|
if ok:
|
|
ae["status"] = Status.COMPLETED.value
|
|
ae["error"] = None
|
|
ae["copied_at"] = _now_iso()
|
|
else:
|
|
ae["status"] = Status.FAILED.value
|
|
ae["error"] = err
|
|
album_ok = False
|
|
log.error(f" ✗ 失败: {err}")
|
|
manifest.save()
|
|
|
|
album["status"] = Status.COMPLETED.value if album_ok else Status.FAILED.value
|
|
if album_ok:
|
|
album["completed_at"] = _now_iso()
|
|
manifest.save()
|
|
|
|
|
|
def print_report(manifest: Manifest, log: logging.Logger) -> None:
|
|
albums = manifest.data.get("albums", {})
|
|
stats = manifest.data.get("stats") or {}
|
|
|
|
log.info("")
|
|
log.info("=" * 70)
|
|
log.info("摘要")
|
|
log.info("=" * 70)
|
|
log.info(f"总专辑数: {stats.get('total_albums', 0)}")
|
|
for k, v in sorted((stats.get("by_album_status") or {}).items()):
|
|
log.info(f" · album {k}: {v}")
|
|
log.info(f"音频文件: {stats.get('total_audio_files', 0)}")
|
|
for k, v in sorted((stats.get("by_audio_status") or {}).items()):
|
|
log.info(f" · audio {k}: {v}")
|
|
log.info(f"资源文件: {stats.get('total_asset_files', 0)}")
|
|
for k, v in sorted((stats.get("by_asset_status") or {}).items()):
|
|
log.info(f" · asset {k}: {v}")
|
|
|
|
needs_split = [(r, a) for r, a in albums.items()
|
|
if a.get("status") == Status.NEEDS_CUE_SPLIT.value]
|
|
if needs_split:
|
|
log.warning("")
|
|
log.warning(f"[!] {len(needs_split)} 个专辑需先手动 CUE split (已跳过转码):")
|
|
for r, a in needs_split:
|
|
log.warning(f" • {r}")
|
|
for issue in a.get("issues", []):
|
|
log.warning(f" {issue}")
|
|
|
|
failed = [(r, a) for r, a in albums.items()
|
|
if a.get("status") == Status.FAILED.value]
|
|
if failed:
|
|
log.error("")
|
|
log.error(f"[x] {len(failed)} 个专辑失败 (再次运行会重试):")
|
|
for r, a in failed:
|
|
log.error(f" • {r}")
|
|
audio_errs = [(n, af.get("error"))
|
|
for n, af in a.get("audio_files", {}).items()
|
|
if af.get("status") == Status.FAILED.value]
|
|
asset_errs = [(n, ae.get("error"))
|
|
for n, ae in a.get("asset_files", {}).items()
|
|
if ae.get("status") == Status.FAILED.value]
|
|
for n, e in audio_errs[:3]:
|
|
log.error(f" ✗ audio {n}: {e}")
|
|
for n, e in asset_errs[:2]:
|
|
log.error(f" ✗ asset {n}: {e}")
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser(
|
|
description="批量转码无损音乐 (FLAC/WAV/DSF/APE) 为 MP3 320k, 带 manifest 状态追踪与断点续跑.",
|
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
epilog="""示例:
|
|
# 最简用法 (位置参数)
|
|
%(prog)s /mnt/z/test /mnt/x/music/test
|
|
|
|
# 仅分析, 生成 manifest (推荐首次运行时用来排查 CUE 整轨等问题)
|
|
%(prog)s /mnt/z/test /mnt/x/music/test --analyze-only
|
|
|
|
# 使用命名参数形式 (与位置参数等价)
|
|
%(prog)s -s /mnt/z/test -t /mnt/x/music/test
|
|
|
|
# 强制重跑 (忽略现有 manifest)
|
|
%(prog)s /mnt/z/test /mnt/x/music/test --force
|
|
|
|
# 使用 256k 比特率
|
|
%(prog)s /mnt/z/test /mnt/x/music/test -b 256k
|
|
""",
|
|
)
|
|
ap.add_argument("source_pos", nargs="?", metavar="SOURCE",
|
|
help="源目录 (位置参数, 如 /mnt/z/test). 也可用 --source/-s")
|
|
ap.add_argument("target_pos", nargs="?", metavar="TARGET",
|
|
help="目标目录 (位置参数, 如 /mnt/x/music/test). 也可用 --target/-t")
|
|
ap.add_argument("--source", "-s", dest="source_flag",
|
|
help="源目录 (等价于第 1 个位置参数)")
|
|
ap.add_argument("--target", "-t", dest="target_flag",
|
|
help="目标目录 (等价于第 2 个位置参数)")
|
|
ap.add_argument("--bitrate", "-b", default="320k",
|
|
help="MP3 比特率 (默认 320k). 可用 256k/192k 等")
|
|
ap.add_argument("--analyze-only", action="store_true",
|
|
help="只分析生成 manifest, 不实际转码")
|
|
ap.add_argument("--force", action="store_true",
|
|
help="忽略现有 manifest, 强制重新分析并转码全部")
|
|
ap.add_argument("--verbose", "-v", action="store_true", help="详细日志")
|
|
ap.add_argument("--ffmpeg", default="ffmpeg", help="ffmpeg 路径 (默认从 PATH 查)")
|
|
ap.add_argument("--ffprobe", default="ffprobe", help="ffprobe 路径")
|
|
ap.add_argument("--timeout", type=int, default=3600,
|
|
help="单文件转码超时秒数 (默认 3600)")
|
|
ap.add_argument("--sample-rate-when-high", type=int, default=44100,
|
|
choices=[44100, 48000],
|
|
help="输入采样率>48kHz时的输出采样率, 默认 44100 (CD 标准)")
|
|
args = ap.parse_args()
|
|
|
|
source_arg = args.source_pos or args.source_flag
|
|
target_arg = args.target_pos or args.target_flag
|
|
if not source_arg or not target_arg:
|
|
ap.error("必须提供源目录和目标目录 (位置参数 SOURCE TARGET, 或 --source/--target)")
|
|
|
|
logging.basicConfig(
|
|
level=logging.DEBUG if args.verbose else logging.INFO,
|
|
format="%(asctime)s %(levelname)-5s %(message)s",
|
|
datefmt="%H:%M:%S",
|
|
)
|
|
log = logging.getLogger("transcode")
|
|
|
|
source = Path(source_arg).resolve()
|
|
target = Path(target_arg).resolve()
|
|
|
|
if not source.exists():
|
|
log.error(f"源目录不存在: {source}")
|
|
return 2
|
|
if not source.is_dir():
|
|
log.error(f"源路径不是目录: {source}")
|
|
return 2
|
|
|
|
try:
|
|
target.mkdir(parents=True, exist_ok=True)
|
|
except OSError as e:
|
|
log.error(f"无法创建目标目录 {target}: {e}")
|
|
return 2
|
|
|
|
cfg = Config(
|
|
source_root=source,
|
|
target_root=target,
|
|
bitrate=args.bitrate,
|
|
force=args.force,
|
|
analyze_only=args.analyze_only,
|
|
ffmpeg=args.ffmpeg,
|
|
ffprobe=args.ffprobe,
|
|
ffmpeg_timeout_sec=args.timeout,
|
|
output_sample_rate_when_high=args.sample_rate_when_high,
|
|
)
|
|
|
|
manifest_path = source / MANIFEST_NAME
|
|
manifest = Manifest(manifest_path, cfg)
|
|
manifest.load_or_init()
|
|
|
|
log.info(f"源目录: {source}")
|
|
log.info(f"目标目录: {target}")
|
|
log.info(f"Manifest: {manifest_path}")
|
|
log.info(f"比特率: {cfg.bitrate}")
|
|
log.info(f"高采样率降到: {cfg.output_sample_rate_when_high} Hz")
|
|
log.info(f"模式: {'仅分析' if cfg.analyze_only else '分析 + 转码'}")
|
|
log.info(f"Resume: {'关闭 (--force)' if cfg.force else '开启 (跳过已完成)'}")
|
|
log.info("")
|
|
|
|
try:
|
|
analyze_all(cfg, manifest, log)
|
|
|
|
if cfg.analyze_only:
|
|
print_report(manifest, log)
|
|
log.info("")
|
|
log.info("仅分析模式完成. 请查看 manifest, 处理 needs_cue_split 的专辑后再跑一次.")
|
|
return 0
|
|
|
|
process_all(cfg, manifest, log)
|
|
print_report(manifest, log)
|
|
|
|
except KeyboardInterrupt:
|
|
log.warning("")
|
|
log.warning("用户中断 (Ctrl+C) - 保存 manifest 后退出")
|
|
try:
|
|
manifest.save()
|
|
except OSError as e:
|
|
log.error(f"保存 manifest 失败: {e}")
|
|
return 130
|
|
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|