Files
atlas/transcode_music/split_cue.py
2026-09-08 06:47:00 +08:00

1094 lines
39 KiB
Python

#!/usr/bin/env python3
"""
split_cue.py - 批量用 CUE 文件切割整轨无损音乐为分轨.
Companion to transcode_music.py. Walks source dir, detects "needs_cue_split"
albums (single CUE + single whole-disc lossless file), splits them into
per-track files IN PLACE using shntool (shnsplit) + cuetag. Preserves source
format when possible:
FLAC → FLAC (via shnsplit -o flac, needs `flac` binary)
WAV → WAV (via shnsplit -o wav, native)
APE → FLAC (needs `mac` decoder; UNSUPPORTED if missing)
DSF/DFF → UNSUPPORTED (shntool doesn't handle DSD)
Whole-disc file is NEVER modified. transcode_music.py's CUE analyzer will
subsequently see it as "situation C" (CUE + whole-disc + per-track files)
and skip the whole-disc file, transcoding only the per-track outputs.
Split pipeline per album:
1. shnsplit -f cue -o <fmt> -t "%n" -O always -d <tmpdir> <src>
produces <tmpdir>/01.<ext>, 02.<ext>, ...
2. cuetag <cue> <tmpdir>/01.<ext> ... applies CUE tags positionally
3. os.replace each tmp file → dir/<NN - Title>.<ext> (sanitized name)
Dependencies: Python 3.10+, shntool, cuetools (cuetag), ffprobe (for source
metadata display only). `flac` for FLAC output. `mac` for APE input.
Usage:
# Analyze only (recommended first run — inspect split plan before splitting)
python3 split_cue.py /mnt/z/music --analyze-only
# Split all detected whole-disc albums in-place (auto-resumes)
python3 split_cue.py /mnt/z/music
# Force re-analyze + overwrite existing per-track files
python3 split_cue.py /mnt/z/music --force
"""
from __future__ import annotations
import argparse
import datetime as dt
import errno
import json
import logging
import os
import re
import shutil
import subprocess
import sys
import threading
from dataclasses import dataclass
from enum import Enum
from pathlib import Path
from typing import Optional
__version__ = "1.0.0"
MANIFEST_SCHEMA_VERSION = 1
MANIFEST_NAME = ".split_cue_manifest.json"
LOSSLESS_EXTS = {".flac", ".wav", ".dsf", ".dff", ".ape"}
CUE_EXTS = {".cue"}
IGNORE_NAMES = {".ds_store", "thumbs.db", "desktop.ini", ".directory"}
IGNORE_PREFIXES = ("._",) # macOS resource forks — not user content
# CUE files from Chinese/Japanese/Korean releases often use legacy encodings.
CUE_ENCODINGS = ("utf-8-sig", "utf-8", "gbk", "big5", "shift_jis", "cp936", "latin1")
# Filesystem-illegal characters (Windows-superset, safe on Linux/macOS too).
_ILLEGAL_CHARS_RE = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
class Status(str, Enum):
PENDING = "pending"
ANALYZED = "analyzed"
PROCESSING = "processing"
COMPLETED = "completed"
FAILED = "failed"
SKIPPED = "skipped"
UNSUPPORTED = "unsupported"
@dataclass
class Config:
source_root: Path
force: bool = False
analyze_only: bool = False
shnsplit: str = "shnsplit"
cuetag: str = "cuetag"
ffprobe: str = "ffprobe"
split_timeout_sec: int = 1800
tag_timeout_sec: int = 60
# ────────────────────────── helpers ──────────────────────────
def _now_iso() -> str:
return dt.datetime.now(dt.timezone.utc).astimezone().isoformat(timespec="seconds")
def _to_int(x) -> Optional[int]:
if x in (None, "", "N/A"):
return None
try:
return int(x)
except (TypeError, ValueError):
return None
def _to_float(x) -> Optional[float]:
if x in (None, "", "N/A"):
return None
try:
return float(x)
except (TypeError, ValueError):
return None
def _try_unlink(p: Path) -> None:
try:
if p.exists():
p.unlink()
except OSError:
pass
def _remove_dir(path: Path) -> None:
try:
shutil.rmtree(str(path))
except OSError:
pass
def _is_ignored(p: Path) -> bool:
name = p.name.lower()
if name in IGNORE_NAMES:
return True
for pref in IGNORE_PREFIXES:
if name.startswith(pref):
return True
if MANIFEST_NAME.lower() in name:
return True
return False
def atomic_write_json(path: Path, data: dict) -> None:
payload = json.dumps(data, ensure_ascii=False, indent=2)
tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.tmp")
fd = os.open(str(tmp), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644)
try:
os.write(fd, payload.encode("utf-8"))
try:
os.fsync(fd)
except OSError:
pass # network mounts (CIFS/NFS) may not implement fsync
finally:
os.close(fd)
os.replace(str(tmp), str(path))
try:
dfd = os.open(str(path.parent), os.O_RDONLY)
try:
os.fsync(dfd) # persist the rename itself (durability guarantee)
except OSError:
pass
finally:
os.close(dfd)
except OSError:
pass
def quarantine_corrupt(path: Path, reason: str) -> Path:
ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S")
q = path.with_name(f"{path.name}.corrupt-{ts}")
try:
os.replace(str(path), str(q))
logging.error(f"Manifest 损坏, 已隔离到 {q}: {reason}")
except OSError as e:
logging.error(f"隔离失败 {path}: {e}")
return q
def sanitize_filename(name: str, max_len: int = 200) -> str:
s = _ILLEGAL_CHARS_RE.sub("_", name)
s = re.sub(r"\s+", " ", s).strip()
s = s.rstrip(". ") # Windows dislikes trailing dots/spaces
if len(s) > max_len:
s = s[:max_len].rstrip()
return s or "untitled"
# ────────────────────────── ffprobe ──────────────────────────
def ffprobe_audio(path: Path, cfg: Config) -> dict:
try:
result = subprocess.run(
[
cfg.ffprobe, "-v", "error",
"-select_streams", "a:0",
"-show_entries", "stream=codec_name,sample_rate,channels,bits_per_raw_sample",
"-show_entries", "format=duration",
"-of", "json",
str(path),
],
capture_output=True, text=True, errors="replace", timeout=60,
)
if result.returncode != 0:
return {"error": (result.stderr or "ffprobe failed").strip()[-200:]}
data = json.loads(result.stdout or "{}")
stream = (data.get("streams") or [{}])[0]
fmt = data.get("format") or {}
return {
"codec": stream.get("codec_name"),
"sample_rate": _to_int(stream.get("sample_rate")),
"channels": _to_int(stream.get("channels")),
"duration_sec": _to_float(fmt.get("duration")),
"bits_per_sample": _to_int(stream.get("bits_per_raw_sample")),
}
except subprocess.TimeoutExpired:
return {"error": "ffprobe timeout"}
except (json.JSONDecodeError, OSError, UnicodeDecodeError) as e:
return {"error": str(e)}
# ────────────────────────── CUE parser ──────────────────────────
def read_cue_text(path: Path) -> Optional[str]:
try:
raw = path.read_bytes()
except OSError:
return None
for enc in CUE_ENCODINGS:
try:
return raw.decode(enc)
except UnicodeDecodeError:
continue
return None
def _unquote(s: str) -> str:
s = s.strip()
if len(s) >= 2 and s[0] == '"' and s[-1] == '"':
return s[1:-1]
return s
def parse_cue_full(path: Path) -> Optional[dict]:
"""
Parse a CUE file into structured form.
CUE spec semantics:
TITLE / PERFORMER before any TRACK = album-level
TITLE / PERFORMER inside a TRACK block = track-level
REM DATE / REM GENRE at album level = album-level metadata
"""
text = read_cue_text(path)
if text is None:
return None
album_title: Optional[str] = None
album_performer: Optional[str] = None
album_date: Optional[str] = None
album_genre: Optional[str] = None
files: list[dict] = []
tracks: list[dict] = []
current_track: Optional[dict] = None
for raw_line in text.splitlines():
line = raw_line.strip()
if not line:
continue
parts = line.split(None, 1)
if not parts:
continue
keyword = parts[0].upper()
rest = parts[1] if len(parts) > 1 else ""
if keyword == "FILE":
m = re.match(r'"([^"]*)"\s+(\S+)', rest) or re.match(r"(\S+)\s+(\S+)", rest)
if m:
files.append({"name": m.group(1), "format": m.group(2).upper()})
elif keyword == "TRACK":
m = re.match(r"(\d+)\s+(\S+)", rest)
if m:
current_track = {
"num": int(m.group(1)),
"type": m.group(2).upper(),
"title": None,
"performer": None,
}
tracks.append(current_track)
elif keyword == "TITLE":
val = _unquote(rest)
if current_track is None:
if album_title is None:
album_title = val
else:
current_track["title"] = val
elif keyword == "PERFORMER":
val = _unquote(rest)
if current_track is None:
if album_performer is None:
album_performer = val
else:
current_track["performer"] = val
elif keyword == "REM":
rem_parts = rest.split(None, 1)
if len(rem_parts) >= 2 and current_track is None:
rem_kw = rem_parts[0].upper()
rem_val = _unquote(rem_parts[1])
if rem_kw == "DATE" and album_date is None:
album_date = rem_val
elif rem_kw == "GENRE" and album_genre is None:
album_genre = rem_val
return {
"album_title": album_title,
"album_performer": album_performer,
"album_date": album_date,
"album_genre": album_genre,
"files": files,
"tracks": tracks,
}
def match_cue_ref(ref_name: str, audio_paths: list[Path]) -> Optional[Path]:
"""
Match a CUE FILE reference to an actual audio file, with fallbacks for
case differences and extension mismatch (e.g., FILE "album.wav" WAVE
where the actual file is album.flac).
"""
for p in audio_paths:
if p.name == ref_name:
return p
ref_lower = ref_name.lower()
for p in audio_paths:
if p.name.lower() == ref_lower:
return p
ref_stem = Path(ref_name).stem.lower()
matches = [p for p in audio_paths if p.stem.lower() == ref_stem]
if len(matches) == 1:
return matches[0]
return None
# ────────────────────────── dependency probing ──────────────────────────
@dataclass
class Deps:
shnsplit: bool
cuetag: bool
ffprobe: bool
flac: bool
mac: bool
def supported_source_exts(self) -> set[str]:
supported = {".wav"} # shntool native
if self.flac:
supported.add(".flac")
if self.mac and self.flac:
supported.add(".ape")
return supported
def unsupported_reason(self, ext: str) -> str:
if ext == ".ape":
missing = []
if not self.mac:
missing.append("mac (APE 解码器)")
if not self.flac:
missing.append("flac")
return f"缺少依赖: {', '.join(missing)}"
if ext in (".dsf", ".dff"):
return "shntool 不支持 DSD, 请先手动转 FLAC"
if ext == ".flac" and not self.flac:
return "缺少依赖: flac"
return f"未知源格式: {ext}"
def probe_deps(cfg: Config) -> Deps:
return Deps(
shnsplit=shutil.which(cfg.shnsplit) is not None,
cuetag=shutil.which(cfg.cuetag) is not None,
ffprobe=shutil.which(cfg.ffprobe) is not None,
flac=shutil.which("flac") is not None,
mac=shutil.which("mac") is not None,
)
def _output_format_for(src_ext: str) -> tuple[Optional[str], Optional[str]]:
if src_ext == ".flac":
return "flac", ".flac"
if src_ext == ".wav":
return "wav", ".wav"
if src_ext == ".ape":
return "flac", ".flac" # ape → flac (mac decodes, flac encodes)
return None, None
# ────────────────────────── analysis ──────────────────────────
def analyze_directory(dir_path: Path, cfg: Config, deps: Deps) -> dict:
"""
Analyze one directory. Detects the "whole-disc CUE + big file" pattern
and builds a per-track split plan. Non-matching dirs get SKIPPED.
Sources whose decoder is missing get UNSUPPORTED with reason.
"""
entry: dict = {
"status": Status.PENDING.value,
"source_audio": None,
"cue": None,
"album_title": None,
"album_performer": None,
"album_date": None,
"album_genre": None,
"source_codec": None,
"source_sample_rate": None,
"source_channels": None,
"duration_sec": None,
"output_format": None,
"tracks": [],
"issues": [],
"started_at": None,
"completed_at": None,
}
try:
files = sorted(p for p in dir_path.iterdir() if p.is_file())
except OSError as e:
entry["status"] = Status.FAILED.value
entry["issues"].append(f"无法读取目录: {e}")
return entry
audio_files = [p for p in files
if p.suffix.lower() in LOSSLESS_EXTS and not _is_ignored(p)]
cue_files = [p for p in files
if p.suffix.lower() in CUE_EXTS and not _is_ignored(p)]
if not cue_files or not audio_files:
entry["status"] = Status.SKIPPED.value
return entry
supported = deps.supported_source_exts()
for cue in cue_files:
parsed = parse_cue_full(cue)
if parsed is None:
entry["issues"].append(f"{cue.name}: 无法解析 (编码问题)")
continue
refs = parsed["files"]
if len(refs) == 0:
entry["issues"].append(f"{cue.name}: 无 FILE entry")
continue
if len(refs) > 1:
continue # multi-FILE CUE = already per-track index
ref_name = refs[0]["name"]
matched = match_cue_ref(ref_name, audio_files)
if matched is None:
entry["issues"].append(f"{cue.name}: 引用 '{ref_name}' 但目录内不存在")
continue
others = [p for p in audio_files if p != matched]
if others:
continue # situation C — already split
# ── Whole-disc pattern confirmed ──
cue_tracks = parsed["tracks"]
if not cue_tracks:
entry["issues"].append(f"{cue.name}: 无 TRACK 条目")
continue
src_ext = matched.suffix.lower()
entry["cue"] = cue.name
entry["source_audio"] = matched.name
entry["album_title"] = parsed["album_title"]
entry["album_performer"] = parsed["album_performer"]
entry["album_date"] = parsed["album_date"]
entry["album_genre"] = parsed["album_genre"]
if src_ext not in supported:
entry["status"] = Status.UNSUPPORTED.value
entry["issues"].append(deps.unsupported_reason(src_ext))
return entry
out_format, out_ext = _output_format_for(src_ext)
entry["output_format"] = out_format
probe = ffprobe_audio(matched, cfg)
if "error" in probe:
entry["issues"].append(f"ffprobe '{matched.name}': {probe['error']}")
else:
entry["source_codec"] = probe.get("codec")
entry["source_sample_rate"] = probe.get("sample_rate")
entry["source_channels"] = probe.get("channels")
entry["duration_sec"] = probe.get("duration_sec")
num_digits = max(2, len(str(len(cue_tracks))))
total_tracks = len(cue_tracks)
target_names: set[str] = set()
for i, cue_track in enumerate(cue_tracks, start=1):
title = cue_track.get("title") or f"Track {cue_track['num']:02d}"
num_str = str(cue_track["num"]).zfill(num_digits)
base = sanitize_filename(f"{num_str} - {title}")
target = base + out_ext
n = 2
while target in target_names:
target = f"{base} ({n}){out_ext}"
n += 1
target_names.add(target)
entry["tracks"].append({
"num": cue_track["num"],
"shnsplit_index": i, # 1-based, matches shnsplit's -t "%n" counter
"title": title,
"performer": cue_track.get("performer"),
"target_name": target,
"total_tracks": total_tracks,
"status": Status.PENDING.value,
"error": None,
"converted_at": None,
})
entry["status"] = Status.ANALYZED.value
return entry
entry["status"] = Status.SKIPPED.value
return entry
def merge_entry(existing: dict, fresh: dict) -> dict:
"""
Preserve completion state across re-analysis: if a previously-completed
track has the same target_name, mark it completed in the fresh entry.
"""
old_by_name = {t.get("target_name"): t for t in existing.get("tracks") or []}
for t in fresh.get("tracks", []):
old = old_by_name.get(t["target_name"])
if old and old.get("status") == Status.COMPLETED.value:
t["status"] = Status.COMPLETED.value
t["converted_at"] = old.get("converted_at")
t["error"] = None
if existing.get("started_at"):
fresh["started_at"] = existing["started_at"]
return fresh
# ────────────────────────── album splitter (shntool + cuetag) ──────────────────────────
def _mark_all_failed(album_entry: dict, err: str) -> None:
for t in album_entry["tracks"]:
if t["status"] != Status.COMPLETED.value:
t["status"] = Status.FAILED.value
t["error"] = err
# INDEX / PREGAP / POSTGAP with 1-digit fractional (e.g. "57:12.0" or "57:12:0")
# — non-standard, both shnsplit AND cuetag's internal cueprint reject them.
# Normalize to canonical CUE "MM:SS:FF" (2-digit CD frame count, 0-74) by
# interpreting the 1 digit as raw frame count and left-padding to 2 digits.
# For the overwhelmingly common ".0" case this is exact (0 frames); other
# values may drift by up to 9 frames (~120ms) which is unavoidable given
# the original CUE's ambiguity.
_BAD_CUE_TIME_RE = re.compile(
r"^([ \t]*(?:INDEX\s+\d+|PREGAP|POSTGAP)\s+)(\d+):(\d+)[.:](\d)[ \t]*$",
re.IGNORECASE | re.MULTILINE,
)
def normalize_cue_for_shntool(cue_text: str) -> str:
cue_text = cue_text.replace("\r\n", "\n").replace("\r", "\n")
normalized = _BAD_CUE_TIME_RE.sub(
lambda m: f"{m.group(1)}{m.group(2)}:{m.group(3)}:0{m.group(4)}",
cue_text,
)
# cueprint (used internally by cuetag) rejects files without a final \n.
if not normalized.endswith("\n"):
normalized += "\n"
return normalized
def split_album(
dir_path: Path,
album_entry: dict,
cfg: Config,
log: logging.Logger,
) -> bool:
"""
Split all pending tracks via shnsplit + cuetag + rename.
Returns True on full success (all tracks COMPLETED).
Design: shnsplit produces `<tmpdir>/01.<ext>`, `02.<ext>`, ... (zero-padded
per `-n` format). cuetag writes CUE metadata into each file positionally.
We then os.replace() each temp file to its sanitized final name in dir_path.
"""
src = dir_path / album_entry["source_audio"]
cue = dir_path / album_entry["cue"]
tracks = album_entry["tracks"]
total = len(tracks)
if not src.exists():
log.error(f" 源文件消失: {src}")
album_entry["issues"].append(f"处理时源文件不存在: {src.name}")
_mark_all_failed(album_entry, "源文件不存在")
return False
if not cue.exists():
log.error(f" CUE 消失: {cue}")
album_entry["issues"].append(f"处理时 CUE 不存在: {cue.name}")
_mark_all_failed(album_entry, "CUE 不存在")
return False
out_format, out_ext = _output_format_for(src.suffix.lower())
if out_format is None:
_mark_all_failed(album_entry, f"源格式 {src.suffix} 不支持")
return False
# Skip fast-path: all tracks already completed and files exist on disk.
if not cfg.force:
all_done = all(
t["status"] == Status.COMPLETED.value and (dir_path / t["target_name"]).exists()
for t in tracks
)
if all_done:
log.debug(f" 所有轨已完成且文件存在, skip")
return True
tmpdir = dir_path / f".split_cue_tmp_{os.getpid()}"
if tmpdir.exists():
_remove_dir(tmpdir)
try:
tmpdir.mkdir()
except OSError as e:
log.error(f" 无法创建临时目录 {tmpdir}: {e}")
_mark_all_failed(album_entry, f"临时目录创建失败: {e}")
return False
try:
num_digits = max(2, len(str(total)))
n_format = f"%0{num_digits}d"
# ── Step 0: prepare a UTF-8 + time-normalized CUE for both shnsplit and cuetag ──
# Feeding original CUE to shnsplit trips on non-standard MM:SS.n times.
# Feeding original CUE to cuetag corrupts CJK tags because Vorbis Comment
# requires UTF-8. One prepared CUE solves both.
prepared_cue: Optional[Path] = None
cue_text = read_cue_text(cue)
if cue_text is None:
log.error(f" CUE 无法解码: {cue.name}")
_mark_all_failed(album_entry, "CUE 无法解码")
return False
cue_text = normalize_cue_for_shntool(cue_text)
prepared_cue = tmpdir / "_prepared.cue"
try:
prepared_cue.write_text(cue_text, encoding="utf-8")
except OSError as e:
log.error(f" 临时 CUE 写入失败: {e}")
_mark_all_failed(album_entry, f"临时 CUE 写入失败: {e}")
return False
# ── Step 1: shnsplit ──
cmd = [
cfg.shnsplit,
"-f", str(prepared_cue),
"-o", out_format,
"-t", "%n",
"-n", n_format,
"-O", "always",
"-d", str(tmpdir),
str(src),
]
log.info(f" shnsplit -o {out_format} → {total} 轨")
try:
result = subprocess.run(
cmd, capture_output=True, text=True, errors="replace",
timeout=cfg.split_timeout_sec,
)
except subprocess.TimeoutExpired:
log.error(f" ✗ shnsplit 超时")
_mark_all_failed(album_entry, f"shnsplit 超时 ({cfg.split_timeout_sec}s)")
return False
except (OSError, UnicodeDecodeError) as e:
log.error(f" ✗ shnsplit 调用失败: {e}")
_mark_all_failed(album_entry, f"shnsplit 调用失败: {e}")
return False
if result.returncode != 0:
stderr = (result.stderr or "").strip()
err_msg = stderr.splitlines()[-1] if stderr else f"shnsplit exit {result.returncode}"
log.error(f" ✗ shnsplit 失败: {err_msg}")
_mark_all_failed(album_entry, err_msg[-400:])
return False
# ── Step 2: verify expected temp files ──
tmp_files: list[Path] = []
for i in range(1, total + 1):
tmp_name = f"{i:0{num_digits}d}{out_ext}"
f = tmpdir / tmp_name
try:
sz = f.stat().st_size
except OSError:
sz = 0
if sz < 1024:
err = f"shnsplit 未产生预期文件 {tmp_name} (大小 {sz})"
log.error(f" ✗ {err}")
_mark_all_failed(album_entry, err)
return False
tmp_files.append(f)
# ── Step 3: cuetag (positional metadata write) ──
tag_cmd = [cfg.cuetag, str(prepared_cue)] + [str(f) for f in tmp_files]
try:
tag_result = subprocess.run(
tag_cmd, capture_output=True, text=True, errors="replace",
timeout=cfg.tag_timeout_sec,
)
if tag_result.returncode != 0:
stderr = (tag_result.stderr or "").strip()
brief = stderr.splitlines()[-1] if stderr else f"exit {tag_result.returncode}"
log.warning(f" cuetag 失败, 音频仍可用但可能无标签: {brief}")
album_entry["issues"].append(f"cuetag 失败: {brief[:200]}")
except (subprocess.TimeoutExpired, OSError, UnicodeDecodeError) as e:
log.warning(f" cuetag 调用失败, 音频仍可用: {e}")
album_entry["issues"].append(f"cuetag 调用失败: {e}")
# ── Step 4: rename each temp file to sanitized final name in dir_path ──
all_ok = True
for tmp_f, t in zip(tmp_files, tracks):
dst = dir_path / t["target_name"]
if dst.exists() and not cfg.force:
# Previous-run leftover with matching name — keep it, discard our temp.
log.debug(f" [已存在, 保留] {t['target_name']}")
_try_unlink(tmp_f)
t["status"] = Status.COMPLETED.value
t["error"] = None
if not t.get("converted_at"):
t["converted_at"] = _now_iso()
continue
if dst.exists() and cfg.force:
try:
dst.unlink()
except OSError as e:
log.error(f" ✗ 无法覆盖 {dst.name}: {e}")
t["status"] = Status.FAILED.value
t["error"] = f"无法覆盖: {e}"
all_ok = False
continue
try:
os.replace(str(tmp_f), str(dst))
except OSError as e:
if e.errno == errno.EXDEV:
try:
shutil.move(str(tmp_f), str(dst))
except OSError as e2:
log.error(f" ✗ move {tmp_f.name} → {dst.name}: {e2}")
t["status"] = Status.FAILED.value
t["error"] = f"move 失败: {e2}"
all_ok = False
continue
else:
log.error(f" ✗ rename {tmp_f.name} → {dst.name}: {e}")
t["status"] = Status.FAILED.value
t["error"] = f"rename 失败: {e}"
all_ok = False
continue
t["status"] = Status.COMPLETED.value
t["error"] = None
t["converted_at"] = _now_iso()
log.info(f" ✓ [{t['num']:02d}/{total}] {t['target_name']}")
return all_ok
finally:
_remove_dir(tmpdir)
# ────────────────────────── manifest ──────────────────────────
class Manifest:
def __init__(self, path: Path, cfg: Config):
self.path = path
self.cfg = cfg
self.data: dict = {}
def load_or_init(self) -> None:
if self.path.exists() and not self.cfg.force:
try:
text = self.path.read_text(encoding="utf-8")
data = json.loads(text)
if not isinstance(data, dict) or not isinstance(data.get("albums"), dict):
raise ValueError("bad structure")
self.data = data
logging.info(f"读取 manifest: {len(self.data['albums'])} 个专辑记录")
return
except (json.JSONDecodeError, ValueError, OSError) as e:
quarantine_corrupt(self.path, str(e))
self.data = self._new()
def _new(self) -> dict:
now = _now_iso()
return {
"manifest_version": MANIFEST_SCHEMA_VERSION,
"created_at": now,
"updated_at": now,
"source_root": str(self.cfg.source_root),
"stats": {},
"albums": {},
}
def save(self) -> None:
self.data["updated_at"] = _now_iso()
self.data["stats"] = self._compute_stats()
atomic_write_json(self.path, self.data)
def _compute_stats(self) -> dict:
by_album: dict[str, int] = {}
by_track: dict[str, int] = {}
total_tracks = 0
for album in self.data.get("albums", {}).values():
s = album.get("status", "unknown")
by_album[s] = by_album.get(s, 0) + 1
for t in album.get("tracks") or []:
ts = t.get("status", "unknown")
by_track[ts] = by_track.get(ts, 0) + 1
total_tracks += 1
return {
"total_albums": len(self.data.get("albums", {})),
"by_album_status": by_album,
"by_track_status": by_track,
"total_tracks": total_tracks,
}
# ────────────────────────── driver ──────────────────────────
def find_all_dirs(source_root: Path) -> list[Path]:
"""Include source_root itself so single-album invocations work."""
result: list[Path] = [source_root]
for dirpath, dirnames, _ in os.walk(source_root):
dirnames.sort()
for d in dirnames:
result.append(Path(dirpath) / d)
return result
def analyze_all(cfg: Config, deps: Deps, manifest: Manifest, log: logging.Logger) -> None:
log.info(f"扫描 {cfg.source_root} ...")
all_dirs = find_all_dirs(cfg.source_root)
log.info(f"发现 {len(all_dirs)} 个目录 (含根)")
for d in all_dirs:
rel = "." if d == cfg.source_root else str(d.relative_to(cfg.source_root))
existing = manifest.data["albums"].get(rel)
if existing and existing.get("status") == Status.COMPLETED.value and not cfg.force:
log.debug(f"[已完成 skip] {rel}")
continue
log.info(f"[分析] {rel}")
fresh = analyze_directory(d, cfg, deps)
if existing:
fresh = merge_entry(existing, fresh)
manifest.data["albums"][rel] = fresh
manifest.save()
def process_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
albums = manifest.data["albums"]
keys = sorted(albums.keys())
total = len(keys)
actionable = {Status.ANALYZED.value, Status.PROCESSING.value, Status.FAILED.value}
for i, rel in enumerate(keys, 1):
album = albums[rel]
status = album.get("status")
if status == Status.COMPLETED.value and not cfg.force:
log.debug(f"[{i}/{total}] SKIP (已完成): {rel}")
continue
if status == Status.SKIPPED.value:
log.debug(f"[{i}/{total}] SKIP (无需切割): {rel}")
continue
if status == Status.UNSUPPORTED.value:
log.warning(f"[{i}/{total}] SKIP (源格式不支持): {rel}")
continue
if status not in actionable or not album.get("tracks"):
log.debug(f"[{i}/{total}] SKIP (状态 {status} 或无计划): {rel}")
continue
log.info(f"[{i}/{total}] 切割: {rel}")
log.info(f" 源: {album['source_audio']} → {len(album['tracks'])} 轨"
f" ({album.get('output_format', '?')})")
album["status"] = Status.PROCESSING.value
if album.get("started_at") is None:
album["started_at"] = _now_iso()
manifest.save()
dir_path = cfg.source_root / rel if rel != "." else cfg.source_root
try:
all_ok = split_album(dir_path, album, cfg, log)
except KeyboardInterrupt:
manifest.save()
raise
except Exception as e: # last-resort safety net: single album must not kill batch
log.error(f" ✗ 未捕获异常: {type(e).__name__}: {e}")
album["issues"].append(f"未捕获异常 {type(e).__name__}: {str(e)[:200]}")
_mark_all_failed(album, f"未捕获异常: {type(e).__name__}: {str(e)[:100]}")
all_ok = False
finally:
manifest.save()
album["status"] = Status.COMPLETED.value if all_ok else Status.FAILED.value
if all_ok:
album["completed_at"] = _now_iso()
manifest.save()
def print_report(manifest: Manifest, log: logging.Logger) -> None:
albums = manifest.data.get("albums", {})
stats = manifest.data.get("stats") or {}
log.info("")
log.info("=" * 70)
log.info("摘要")
log.info("=" * 70)
log.info(f"总目录数: {stats.get('total_albums', 0)}")
for k, v in sorted((stats.get("by_album_status") or {}).items()):
log.info(f" · album {k}: {v}")
log.info(f"分轨文件: {stats.get('total_tracks', 0)}")
for k, v in sorted((stats.get("by_track_status") or {}).items()):
log.info(f" · track {k}: {v}")
to_split = [(r, a) for r, a in albums.items()
if a.get("status") == Status.ANALYZED.value]
if to_split:
log.info("")
log.info(f"[i] {len(to_split)} 个专辑待切割:")
for r, a in to_split[:20]:
log.info(f" • {r}: {a['source_audio']} → {len(a['tracks'])} 轨")
if len(to_split) > 20:
log.info(f" ... 及另外 {len(to_split) - 20} 个")
unsupported = [(r, a) for r, a in albums.items()
if a.get("status") == Status.UNSUPPORTED.value]
if unsupported:
log.warning("")
log.warning(f"[!] {len(unsupported)} 个专辑源格式不支持 (跳过):")
for r, a in unsupported:
log.warning(f" • {r}")
for issue in a.get("issues", []):
log.warning(f" {issue}")
failed = [(r, a) for r, a in albums.items()
if a.get("status") == Status.FAILED.value]
if failed:
log.error("")
log.error(f"[x] {len(failed)} 个专辑失败 (再次运行会重试):")
for r, a in failed:
log.error(f" • {r}")
for issue in a.get("issues", []):
log.error(f" {issue}")
failed_tracks = [t for t in a.get("tracks", [])
if t.get("status") == Status.FAILED.value]
for t in failed_tracks[:3]:
log.error(f" ✗ 轨 {t['num']:02d}: {t.get('error')}")
def _check_startup_deps(deps: Deps, log: logging.Logger) -> bool:
ok = True
if not deps.shnsplit:
log.error("未找到 shnsplit (shntool). 安装: apt install shntool")
ok = False
if not deps.cuetag:
log.error("未找到 cuetag (cuetools). 安装: apt install cuetools")
ok = False
if not deps.ffprobe:
log.warning("未找到 ffprobe (仅用于源文件元数据展示, 不影响切割)")
if not deps.flac:
log.warning("未找到 flac 二进制 → FLAC 源将无法切割 (仅支持 WAV)")
if not deps.mac:
log.info("未找到 mac (APE 解码器) → APE 源将标记 UNSUPPORTED. "
"安装可选: apt install monkeys-audio (或从源码构建)")
return ok
def main() -> int:
ap = argparse.ArgumentParser(
description="批量用 CUE 切割整轨无损音乐为分轨 (原地切割, 保留原文件, "
"shnsplit + cuetag 后端).",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog="""示例:
# 最简用法 (位置参数)
%(prog)s /mnt/z/music
# 仅分析, 检查切割计划
%(prog)s /mnt/z/music --analyze-only
# 命名参数
%(prog)s -s /mnt/z/music
# 强制重跑 (覆盖已有分轨)
%(prog)s /mnt/z/music --force
""",
)
ap.add_argument("source_pos", nargs="?", metavar="SOURCE",
help="源目录 (位置参数). 也可用 --source/-s")
ap.add_argument("--source", "-s", dest="source_flag",
help="源目录 (等价于位置参数 SOURCE)")
ap.add_argument("--analyze-only", action="store_true",
help="只分析生成 manifest, 不实际切割")
ap.add_argument("--force", action="store_true",
help="忽略现有 manifest, 强制重新分析并覆盖已有分轨")
ap.add_argument("--verbose", "-v", action="store_true", help="详细日志")
ap.add_argument("--shnsplit", default="shnsplit", help="shnsplit 路径 (默认从 PATH 查)")
ap.add_argument("--cuetag", default="cuetag", help="cuetag 路径")
ap.add_argument("--ffprobe", default="ffprobe", help="ffprobe 路径")
ap.add_argument("--split-timeout", type=int, default=1800,
help="shnsplit 超时秒数 (默认 1800)")
args = ap.parse_args()
source_arg = args.source_pos or args.source_flag
if not source_arg:
ap.error("必须提供源目录 (位置参数 SOURCE 或 --source/-s)")
logging.basicConfig(
level=logging.DEBUG if args.verbose else logging.INFO,
format="%(asctime)s %(levelname)-5s %(message)s",
datefmt="%H:%M:%S",
)
log = logging.getLogger("split_cue")
source = Path(source_arg).resolve()
if not source.exists():
log.error(f"源目录不存在: {source}")
return 2
if not source.is_dir():
log.error(f"源路径不是目录: {source}")
return 2
cfg = Config(
source_root=source,
force=args.force,
analyze_only=args.analyze_only,
shnsplit=args.shnsplit,
cuetag=args.cuetag,
ffprobe=args.ffprobe,
split_timeout_sec=args.split_timeout,
)
deps = probe_deps(cfg)
if not _check_startup_deps(deps, log):
return 3
manifest_path = source / MANIFEST_NAME
manifest = Manifest(manifest_path, cfg)
manifest.load_or_init()
log.info(f"源目录: {source}")
log.info(f"Manifest: {manifest_path}")
log.info(f"支持源格式: {sorted(deps.supported_source_exts())}")
log.info(f"模式: {'仅分析' if cfg.analyze_only else '分析 + 切割'}")
log.info(f"Resume: {'关闭 (--force)' if cfg.force else '开启 (跳过已完成)'}")
log.info("")
try:
analyze_all(cfg, deps, manifest, log)
if cfg.analyze_only:
print_report(manifest, log)
log.info("")
log.info("仅分析完成. 查看 manifest 确认切割计划后, 再跑一次进行切割.")
return 0
process_all(cfg, manifest, log)
print_report(manifest, log)
except KeyboardInterrupt:
log.warning("")
log.warning("用户中断 (Ctrl+C) - 保存 manifest 后退出")
try:
manifest.save()
except OSError as e:
log.error(f"保存 manifest 失败: {e}")
return 130
return 0
if __name__ == "__main__":
sys.exit(main())