#!/usr/bin/env python3 """ split_cue.py - 批量用 CUE 文件切割整轨无损音乐为分轨. Companion to transcode_music.py. Walks source dir, detects "needs_cue_split" albums (single CUE + single whole-disc lossless file), splits them into per-track files IN PLACE using shntool (shnsplit) + cuetag. Preserves source format when possible: FLAC → FLAC (via shnsplit -o flac, needs `flac` binary) WAV → WAV (via shnsplit -o wav, native) APE → FLAC (needs `mac` decoder; UNSUPPORTED if missing) DSF/DFF → UNSUPPORTED (shntool doesn't handle DSD) Whole-disc file is NEVER modified. transcode_music.py's CUE analyzer will subsequently see it as "situation C" (CUE + whole-disc + per-track files) and skip the whole-disc file, transcoding only the per-track outputs. Split pipeline per album: 1. shnsplit -f cue -o -t "%n" -O always -d produces /01., 02., ... 2. cuetag /01. ... applies CUE tags positionally 3. os.replace each tmp file → dir/. (sanitized name) Dependencies: Python 3.10+, shntool, cuetools (cuetag), ffprobe (for source metadata display only). `flac` for FLAC output. `mac` for APE input. Usage: # Analyze only (recommended first run — inspect split plan before splitting) python3 split_cue.py /mnt/z/music --analyze-only # Split all detected whole-disc albums in-place (auto-resumes) python3 split_cue.py /mnt/z/music # Force re-analyze + overwrite existing per-track files python3 split_cue.py /mnt/z/music --force """ from __future__ import annotations import argparse import datetime as dt import errno import json import logging import os import re import shutil import subprocess import sys import threading from dataclasses import dataclass from enum import Enum from pathlib import Path from typing import Optional __version__ = "1.0.0" MANIFEST_SCHEMA_VERSION = 1 MANIFEST_NAME = ".split_cue_manifest.json" LOSSLESS_EXTS = {".flac", ".wav", ".dsf", ".dff", ".ape"} CUE_EXTS = {".cue"} IGNORE_NAMES = {".ds_store", "thumbs.db", "desktop.ini", ".directory"} IGNORE_PREFIXES = ("._",) # macOS resource forks — not user content # CUE files from Chinese/Japanese/Korean releases often use legacy encodings. CUE_ENCODINGS = ("utf-8-sig", "utf-8", "gbk", "big5", "shift_jis", "cp936", "latin1") # Filesystem-illegal characters (Windows-superset, safe on Linux/macOS too). _ILLEGAL_CHARS_RE = re.compile(r'[<>:"/\\|?*\x00-\x1f]') class Status(str, Enum): PENDING = "pending" ANALYZED = "analyzed" PROCESSING = "processing" COMPLETED = "completed" FAILED = "failed" SKIPPED = "skipped" UNSUPPORTED = "unsupported" @dataclass class Config: source_root: Path force: bool = False analyze_only: bool = False shnsplit: str = "shnsplit" cuetag: str = "cuetag" ffprobe: str = "ffprobe" split_timeout_sec: int = 1800 tag_timeout_sec: int = 60 # ────────────────────────── helpers ────────────────────────── def _now_iso() -> str: return dt.datetime.now(dt.timezone.utc).astimezone().isoformat(timespec="seconds") def _to_int(x) -> Optional[int]: if x in (None, "", "N/A"): return None try: return int(x) except (TypeError, ValueError): return None def _to_float(x) -> Optional[float]: if x in (None, "", "N/A"): return None try: return float(x) except (TypeError, ValueError): return None def _try_unlink(p: Path) -> None: try: if p.exists(): p.unlink() except OSError: pass def _remove_dir(path: Path) -> None: try: shutil.rmtree(str(path)) except OSError: pass def _is_ignored(p: Path) -> bool: name = p.name.lower() if name in IGNORE_NAMES: return True for pref in IGNORE_PREFIXES: if name.startswith(pref): return True if MANIFEST_NAME.lower() in name: return True return False def atomic_write_json(path: Path, data: dict) -> None: payload = json.dumps(data, ensure_ascii=False, indent=2) tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.tmp") fd = os.open(str(tmp), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644) try: os.write(fd, payload.encode("utf-8")) try: os.fsync(fd) except OSError: pass # network mounts (CIFS/NFS) may not implement fsync finally: os.close(fd) os.replace(str(tmp), str(path)) try: dfd = os.open(str(path.parent), os.O_RDONLY) try: os.fsync(dfd) # persist the rename itself (durability guarantee) except OSError: pass finally: os.close(dfd) except OSError: pass def quarantine_corrupt(path: Path, reason: str) -> Path: ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S") q = path.with_name(f"{path.name}.corrupt-{ts}") try: os.replace(str(path), str(q)) logging.error(f"Manifest 损坏, 已隔离到 {q}: {reason}") except OSError as e: logging.error(f"隔离失败 {path}: {e}") return q def sanitize_filename(name: str, max_len: int = 200) -> str: s = _ILLEGAL_CHARS_RE.sub("_", name) s = re.sub(r"\s+", " ", s).strip() s = s.rstrip(". ") # Windows dislikes trailing dots/spaces if len(s) > max_len: s = s[:max_len].rstrip() return s or "untitled" # ────────────────────────── ffprobe ────────────────────────── def ffprobe_audio(path: Path, cfg: Config) -> dict: try: result = subprocess.run( [ cfg.ffprobe, "-v", "error", "-select_streams", "a:0", "-show_entries", "stream=codec_name,sample_rate,channels,bits_per_raw_sample", "-show_entries", "format=duration", "-of", "json", str(path), ], capture_output=True, text=True, errors="replace", timeout=60, ) if result.returncode != 0: return {"error": (result.stderr or "ffprobe failed").strip()[-200:]} data = json.loads(result.stdout or "{}") stream = (data.get("streams") or [{}])[0] fmt = data.get("format") or {} return { "codec": stream.get("codec_name"), "sample_rate": _to_int(stream.get("sample_rate")), "channels": _to_int(stream.get("channels")), "duration_sec": _to_float(fmt.get("duration")), "bits_per_sample": _to_int(stream.get("bits_per_raw_sample")), } except subprocess.TimeoutExpired: return {"error": "ffprobe timeout"} except (json.JSONDecodeError, OSError, UnicodeDecodeError) as e: return {"error": str(e)} # ────────────────────────── CUE parser ────────────────────────── def read_cue_text(path: Path) -> Optional[str]: try: raw = path.read_bytes() except OSError: return None for enc in CUE_ENCODINGS: try: return raw.decode(enc) except UnicodeDecodeError: continue return None def _unquote(s: str) -> str: s = s.strip() if len(s) >= 2 and s[0] == '"' and s[-1] == '"': return s[1:-1] return s def parse_cue_full(path: Path) -> Optional[dict]: """ Parse a CUE file into structured form. CUE spec semantics: TITLE / PERFORMER before any TRACK = album-level TITLE / PERFORMER inside a TRACK block = track-level REM DATE / REM GENRE at album level = album-level metadata """ text = read_cue_text(path) if text is None: return None album_title: Optional[str] = None album_performer: Optional[str] = None album_date: Optional[str] = None album_genre: Optional[str] = None files: list[dict] = [] tracks: list[dict] = [] current_track: Optional[dict] = None for raw_line in text.splitlines(): line = raw_line.strip() if not line: continue parts = line.split(None, 1) if not parts: continue keyword = parts[0].upper() rest = parts[1] if len(parts) > 1 else "" if keyword == "FILE": m = re.match(r'"([^"]*)"\s+(\S+)', rest) or re.match(r"(\S+)\s+(\S+)", rest) if m: files.append({"name": m.group(1), "format": m.group(2).upper()}) elif keyword == "TRACK": m = re.match(r"(\d+)\s+(\S+)", rest) if m: current_track = { "num": int(m.group(1)), "type": m.group(2).upper(), "title": None, "performer": None, } tracks.append(current_track) elif keyword == "TITLE": val = _unquote(rest) if current_track is None: if album_title is None: album_title = val else: current_track["title"] = val elif keyword == "PERFORMER": val = _unquote(rest) if current_track is None: if album_performer is None: album_performer = val else: current_track["performer"] = val elif keyword == "REM": rem_parts = rest.split(None, 1) if len(rem_parts) >= 2 and current_track is None: rem_kw = rem_parts[0].upper() rem_val = _unquote(rem_parts[1]) if rem_kw == "DATE" and album_date is None: album_date = rem_val elif rem_kw == "GENRE" and album_genre is None: album_genre = rem_val return { "album_title": album_title, "album_performer": album_performer, "album_date": album_date, "album_genre": album_genre, "files": files, "tracks": tracks, } def match_cue_ref(ref_name: str, audio_paths: list[Path]) -> Optional[Path]: """ Match a CUE FILE reference to an actual audio file, with fallbacks for case differences and extension mismatch (e.g., FILE "album.wav" WAVE where the actual file is album.flac). """ for p in audio_paths: if p.name == ref_name: return p ref_lower = ref_name.lower() for p in audio_paths: if p.name.lower() == ref_lower: return p ref_stem = Path(ref_name).stem.lower() matches = [p for p in audio_paths if p.stem.lower() == ref_stem] if len(matches) == 1: return matches[0] return None # ────────────────────────── dependency probing ────────────────────────── @dataclass class Deps: shnsplit: bool cuetag: bool ffprobe: bool flac: bool mac: bool def supported_source_exts(self) -> set[str]: supported = {".wav"} # shntool native if self.flac: supported.add(".flac") if self.mac and self.flac: supported.add(".ape") return supported def unsupported_reason(self, ext: str) -> str: if ext == ".ape": missing = [] if not self.mac: missing.append("mac (APE 解码器)") if not self.flac: missing.append("flac") return f"缺少依赖: {', '.join(missing)}" if ext in (".dsf", ".dff"): return "shntool 不支持 DSD, 请先手动转 FLAC" if ext == ".flac" and not self.flac: return "缺少依赖: flac" return f"未知源格式: {ext}" def probe_deps(cfg: Config) -> Deps: return Deps( shnsplit=shutil.which(cfg.shnsplit) is not None, cuetag=shutil.which(cfg.cuetag) is not None, ffprobe=shutil.which(cfg.ffprobe) is not None, flac=shutil.which("flac") is not None, mac=shutil.which("mac") is not None, ) def _output_format_for(src_ext: str) -> tuple[Optional[str], Optional[str]]: if src_ext == ".flac": return "flac", ".flac" if src_ext == ".wav": return "wav", ".wav" if src_ext == ".ape": return "flac", ".flac" # ape → flac (mac decodes, flac encodes) return None, None # ────────────────────────── analysis ────────────────────────── def analyze_directory(dir_path: Path, cfg: Config, deps: Deps) -> dict: """ Analyze one directory. Detects the "whole-disc CUE + big file" pattern and builds a per-track split plan. Non-matching dirs get SKIPPED. Sources whose decoder is missing get UNSUPPORTED with reason. """ entry: dict = { "status": Status.PENDING.value, "source_audio": None, "cue": None, "album_title": None, "album_performer": None, "album_date": None, "album_genre": None, "source_codec": None, "source_sample_rate": None, "source_channels": None, "duration_sec": None, "output_format": None, "tracks": [], "issues": [], "started_at": None, "completed_at": None, } try: files = sorted(p for p in dir_path.iterdir() if p.is_file()) except OSError as e: entry["status"] = Status.FAILED.value entry["issues"].append(f"无法读取目录: {e}") return entry audio_files = [p for p in files if p.suffix.lower() in LOSSLESS_EXTS and not _is_ignored(p)] cue_files = [p for p in files if p.suffix.lower() in CUE_EXTS and not _is_ignored(p)] if not cue_files or not audio_files: entry["status"] = Status.SKIPPED.value return entry supported = deps.supported_source_exts() for cue in cue_files: parsed = parse_cue_full(cue) if parsed is None: entry["issues"].append(f"{cue.name}: 无法解析 (编码问题)") continue refs = parsed["files"] if len(refs) == 0: entry["issues"].append(f"{cue.name}: 无 FILE entry") continue if len(refs) > 1: continue # multi-FILE CUE = already per-track index ref_name = refs[0]["name"] matched = match_cue_ref(ref_name, audio_files) if matched is None: entry["issues"].append(f"{cue.name}: 引用 '{ref_name}' 但目录内不存在") continue others = [p for p in audio_files if p != matched] if others: continue # situation C — already split # ── Whole-disc pattern confirmed ── cue_tracks = parsed["tracks"] if not cue_tracks: entry["issues"].append(f"{cue.name}: 无 TRACK 条目") continue src_ext = matched.suffix.lower() entry["cue"] = cue.name entry["source_audio"] = matched.name entry["album_title"] = parsed["album_title"] entry["album_performer"] = parsed["album_performer"] entry["album_date"] = parsed["album_date"] entry["album_genre"] = parsed["album_genre"] if src_ext not in supported: entry["status"] = Status.UNSUPPORTED.value entry["issues"].append(deps.unsupported_reason(src_ext)) return entry out_format, out_ext = _output_format_for(src_ext) entry["output_format"] = out_format probe = ffprobe_audio(matched, cfg) if "error" in probe: entry["issues"].append(f"ffprobe '{matched.name}': {probe['error']}") else: entry["source_codec"] = probe.get("codec") entry["source_sample_rate"] = probe.get("sample_rate") entry["source_channels"] = probe.get("channels") entry["duration_sec"] = probe.get("duration_sec") num_digits = max(2, len(str(len(cue_tracks)))) total_tracks = len(cue_tracks) target_names: set[str] = set() for i, cue_track in enumerate(cue_tracks, start=1): title = cue_track.get("title") or f"Track {cue_track['num']:02d}" num_str = str(cue_track["num"]).zfill(num_digits) base = sanitize_filename(f"{num_str} - {title}") target = base + out_ext n = 2 while target in target_names: target = f"{base} ({n}){out_ext}" n += 1 target_names.add(target) entry["tracks"].append({ "num": cue_track["num"], "shnsplit_index": i, # 1-based, matches shnsplit's -t "%n" counter "title": title, "performer": cue_track.get("performer"), "target_name": target, "total_tracks": total_tracks, "status": Status.PENDING.value, "error": None, "converted_at": None, }) entry["status"] = Status.ANALYZED.value return entry entry["status"] = Status.SKIPPED.value return entry def merge_entry(existing: dict, fresh: dict) -> dict: """ Preserve completion state across re-analysis: if a previously-completed track has the same target_name, mark it completed in the fresh entry. """ old_by_name = {t.get("target_name"): t for t in existing.get("tracks") or []} for t in fresh.get("tracks", []): old = old_by_name.get(t["target_name"]) if old and old.get("status") == Status.COMPLETED.value: t["status"] = Status.COMPLETED.value t["converted_at"] = old.get("converted_at") t["error"] = None if existing.get("started_at"): fresh["started_at"] = existing["started_at"] return fresh # ────────────────────────── album splitter (shntool + cuetag) ────────────────────────── def _mark_all_failed(album_entry: dict, err: str) -> None: for t in album_entry["tracks"]: if t["status"] != Status.COMPLETED.value: t["status"] = Status.FAILED.value t["error"] = err # INDEX / PREGAP / POSTGAP with 1-digit fractional (e.g. "57:12.0" or "57:12:0") # — non-standard, both shnsplit AND cuetag's internal cueprint reject them. # Normalize to canonical CUE "MM:SS:FF" (2-digit CD frame count, 0-74) by # interpreting the 1 digit as raw frame count and left-padding to 2 digits. # For the overwhelmingly common ".0" case this is exact (0 frames); other # values may drift by up to 9 frames (~120ms) which is unavoidable given # the original CUE's ambiguity. _BAD_CUE_TIME_RE = re.compile( r"^([ \t]*(?:INDEX\s+\d+|PREGAP|POSTGAP)\s+)(\d+):(\d+)[.:](\d)[ \t]*$", re.IGNORECASE | re.MULTILINE, ) def normalize_cue_for_shntool(cue_text: str) -> str: cue_text = cue_text.replace("\r\n", "\n").replace("\r", "\n") normalized = _BAD_CUE_TIME_RE.sub( lambda m: f"{m.group(1)}{m.group(2)}:{m.group(3)}:0{m.group(4)}", cue_text, ) # cueprint (used internally by cuetag) rejects files without a final \n. if not normalized.endswith("\n"): normalized += "\n" return normalized def split_album( dir_path: Path, album_entry: dict, cfg: Config, log: logging.Logger, ) -> bool: """ Split all pending tracks via shnsplit + cuetag + rename. Returns True on full success (all tracks COMPLETED). Design: shnsplit produces `/01.`, `02.`, ... (zero-padded per `-n` format). cuetag writes CUE metadata into each file positionally. We then os.replace() each temp file to its sanitized final name in dir_path. """ src = dir_path / album_entry["source_audio"] cue = dir_path / album_entry["cue"] tracks = album_entry["tracks"] total = len(tracks) if not src.exists(): log.error(f" 源文件消失: {src}") album_entry["issues"].append(f"处理时源文件不存在: {src.name}") _mark_all_failed(album_entry, "源文件不存在") return False if not cue.exists(): log.error(f" CUE 消失: {cue}") album_entry["issues"].append(f"处理时 CUE 不存在: {cue.name}") _mark_all_failed(album_entry, "CUE 不存在") return False out_format, out_ext = _output_format_for(src.suffix.lower()) if out_format is None: _mark_all_failed(album_entry, f"源格式 {src.suffix} 不支持") return False # Skip fast-path: all tracks already completed and files exist on disk. if not cfg.force: all_done = all( t["status"] == Status.COMPLETED.value and (dir_path / t["target_name"]).exists() for t in tracks ) if all_done: log.debug(f" 所有轨已完成且文件存在, skip") return True tmpdir = dir_path / f".split_cue_tmp_{os.getpid()}" if tmpdir.exists(): _remove_dir(tmpdir) try: tmpdir.mkdir() except OSError as e: log.error(f" 无法创建临时目录 {tmpdir}: {e}") _mark_all_failed(album_entry, f"临时目录创建失败: {e}") return False try: num_digits = max(2, len(str(total))) n_format = f"%0{num_digits}d" # ── Step 0: prepare a UTF-8 + time-normalized CUE for both shnsplit and cuetag ── # Feeding original CUE to shnsplit trips on non-standard MM:SS.n times. # Feeding original CUE to cuetag corrupts CJK tags because Vorbis Comment # requires UTF-8. One prepared CUE solves both. prepared_cue: Optional[Path] = None cue_text = read_cue_text(cue) if cue_text is None: log.error(f" CUE 无法解码: {cue.name}") _mark_all_failed(album_entry, "CUE 无法解码") return False cue_text = normalize_cue_for_shntool(cue_text) prepared_cue = tmpdir / "_prepared.cue" try: prepared_cue.write_text(cue_text, encoding="utf-8") except OSError as e: log.error(f" 临时 CUE 写入失败: {e}") _mark_all_failed(album_entry, f"临时 CUE 写入失败: {e}") return False # ── Step 1: shnsplit ── cmd = [ cfg.shnsplit, "-f", str(prepared_cue), "-o", out_format, "-t", "%n", "-n", n_format, "-O", "always", "-d", str(tmpdir), str(src), ] log.info(f" shnsplit -o {out_format} → {total} 轨") try: result = subprocess.run( cmd, capture_output=True, text=True, errors="replace", timeout=cfg.split_timeout_sec, ) except subprocess.TimeoutExpired: log.error(f" ✗ shnsplit 超时") _mark_all_failed(album_entry, f"shnsplit 超时 ({cfg.split_timeout_sec}s)") return False except (OSError, UnicodeDecodeError) as e: log.error(f" ✗ shnsplit 调用失败: {e}") _mark_all_failed(album_entry, f"shnsplit 调用失败: {e}") return False if result.returncode != 0: stderr = (result.stderr or "").strip() err_msg = stderr.splitlines()[-1] if stderr else f"shnsplit exit {result.returncode}" log.error(f" ✗ shnsplit 失败: {err_msg}") _mark_all_failed(album_entry, err_msg[-400:]) return False # ── Step 2: verify expected temp files ── tmp_files: list[Path] = [] for i in range(1, total + 1): tmp_name = f"{i:0{num_digits}d}{out_ext}" f = tmpdir / tmp_name try: sz = f.stat().st_size except OSError: sz = 0 if sz < 1024: err = f"shnsplit 未产生预期文件 {tmp_name} (大小 {sz})" log.error(f" ✗ {err}") _mark_all_failed(album_entry, err) return False tmp_files.append(f) # ── Step 3: cuetag (positional metadata write) ── tag_cmd = [cfg.cuetag, str(prepared_cue)] + [str(f) for f in tmp_files] try: tag_result = subprocess.run( tag_cmd, capture_output=True, text=True, errors="replace", timeout=cfg.tag_timeout_sec, ) if tag_result.returncode != 0: stderr = (tag_result.stderr or "").strip() brief = stderr.splitlines()[-1] if stderr else f"exit {tag_result.returncode}" log.warning(f" cuetag 失败, 音频仍可用但可能无标签: {brief}") album_entry["issues"].append(f"cuetag 失败: {brief[:200]}") except (subprocess.TimeoutExpired, OSError, UnicodeDecodeError) as e: log.warning(f" cuetag 调用失败, 音频仍可用: {e}") album_entry["issues"].append(f"cuetag 调用失败: {e}") # ── Step 4: rename each temp file to sanitized final name in dir_path ── all_ok = True for tmp_f, t in zip(tmp_files, tracks): dst = dir_path / t["target_name"] if dst.exists() and not cfg.force: # Previous-run leftover with matching name — keep it, discard our temp. log.debug(f" [已存在, 保留] {t['target_name']}") _try_unlink(tmp_f) t["status"] = Status.COMPLETED.value t["error"] = None if not t.get("converted_at"): t["converted_at"] = _now_iso() continue if dst.exists() and cfg.force: try: dst.unlink() except OSError as e: log.error(f" ✗ 无法覆盖 {dst.name}: {e}") t["status"] = Status.FAILED.value t["error"] = f"无法覆盖: {e}" all_ok = False continue try: os.replace(str(tmp_f), str(dst)) except OSError as e: if e.errno == errno.EXDEV: try: shutil.move(str(tmp_f), str(dst)) except OSError as e2: log.error(f" ✗ move {tmp_f.name} → {dst.name}: {e2}") t["status"] = Status.FAILED.value t["error"] = f"move 失败: {e2}" all_ok = False continue else: log.error(f" ✗ rename {tmp_f.name} → {dst.name}: {e}") t["status"] = Status.FAILED.value t["error"] = f"rename 失败: {e}" all_ok = False continue t["status"] = Status.COMPLETED.value t["error"] = None t["converted_at"] = _now_iso() log.info(f" ✓ [{t['num']:02d}/{total}] {t['target_name']}") return all_ok finally: _remove_dir(tmpdir) # ────────────────────────── manifest ────────────────────────── class Manifest: def __init__(self, path: Path, cfg: Config): self.path = path self.cfg = cfg self.data: dict = {} def load_or_init(self) -> None: if self.path.exists() and not self.cfg.force: try: text = self.path.read_text(encoding="utf-8") data = json.loads(text) if not isinstance(data, dict) or not isinstance(data.get("albums"), dict): raise ValueError("bad structure") self.data = data logging.info(f"读取 manifest: {len(self.data['albums'])} 个专辑记录") return except (json.JSONDecodeError, ValueError, OSError) as e: quarantine_corrupt(self.path, str(e)) self.data = self._new() def _new(self) -> dict: now = _now_iso() return { "manifest_version": MANIFEST_SCHEMA_VERSION, "created_at": now, "updated_at": now, "source_root": str(self.cfg.source_root), "stats": {}, "albums": {}, } def save(self) -> None: self.data["updated_at"] = _now_iso() self.data["stats"] = self._compute_stats() atomic_write_json(self.path, self.data) def _compute_stats(self) -> dict: by_album: dict[str, int] = {} by_track: dict[str, int] = {} total_tracks = 0 for album in self.data.get("albums", {}).values(): s = album.get("status", "unknown") by_album[s] = by_album.get(s, 0) + 1 for t in album.get("tracks") or []: ts = t.get("status", "unknown") by_track[ts] = by_track.get(ts, 0) + 1 total_tracks += 1 return { "total_albums": len(self.data.get("albums", {})), "by_album_status": by_album, "by_track_status": by_track, "total_tracks": total_tracks, } # ────────────────────────── driver ────────────────────────── def find_all_dirs(source_root: Path) -> list[Path]: """Include source_root itself so single-album invocations work.""" result: list[Path] = [source_root] for dirpath, dirnames, _ in os.walk(source_root): dirnames.sort() for d in dirnames: result.append(Path(dirpath) / d) return result def analyze_all(cfg: Config, deps: Deps, manifest: Manifest, log: logging.Logger) -> None: log.info(f"扫描 {cfg.source_root} ...") all_dirs = find_all_dirs(cfg.source_root) log.info(f"发现 {len(all_dirs)} 个目录 (含根)") for d in all_dirs: rel = "." if d == cfg.source_root else str(d.relative_to(cfg.source_root)) existing = manifest.data["albums"].get(rel) if existing and existing.get("status") == Status.COMPLETED.value and not cfg.force: log.debug(f"[已完成 skip] {rel}") continue log.info(f"[分析] {rel}") fresh = analyze_directory(d, cfg, deps) if existing: fresh = merge_entry(existing, fresh) manifest.data["albums"][rel] = fresh manifest.save() def process_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None: albums = manifest.data["albums"] keys = sorted(albums.keys()) total = len(keys) actionable = {Status.ANALYZED.value, Status.PROCESSING.value, Status.FAILED.value} for i, rel in enumerate(keys, 1): album = albums[rel] status = album.get("status") if status == Status.COMPLETED.value and not cfg.force: log.debug(f"[{i}/{total}] SKIP (已完成): {rel}") continue if status == Status.SKIPPED.value: log.debug(f"[{i}/{total}] SKIP (无需切割): {rel}") continue if status == Status.UNSUPPORTED.value: log.warning(f"[{i}/{total}] SKIP (源格式不支持): {rel}") continue if status not in actionable or not album.get("tracks"): log.debug(f"[{i}/{total}] SKIP (状态 {status} 或无计划): {rel}") continue log.info(f"[{i}/{total}] 切割: {rel}") log.info(f" 源: {album['source_audio']} → {len(album['tracks'])} 轨" f" ({album.get('output_format', '?')})") album["status"] = Status.PROCESSING.value if album.get("started_at") is None: album["started_at"] = _now_iso() manifest.save() dir_path = cfg.source_root / rel if rel != "." else cfg.source_root try: all_ok = split_album(dir_path, album, cfg, log) except KeyboardInterrupt: manifest.save() raise except Exception as e: # last-resort safety net: single album must not kill batch log.error(f" ✗ 未捕获异常: {type(e).__name__}: {e}") album["issues"].append(f"未捕获异常 {type(e).__name__}: {str(e)[:200]}") _mark_all_failed(album, f"未捕获异常: {type(e).__name__}: {str(e)[:100]}") all_ok = False finally: manifest.save() album["status"] = Status.COMPLETED.value if all_ok else Status.FAILED.value if all_ok: album["completed_at"] = _now_iso() manifest.save() def print_report(manifest: Manifest, log: logging.Logger) -> None: albums = manifest.data.get("albums", {}) stats = manifest.data.get("stats") or {} log.info("") log.info("=" * 70) log.info("摘要") log.info("=" * 70) log.info(f"总目录数: {stats.get('total_albums', 0)}") for k, v in sorted((stats.get("by_album_status") or {}).items()): log.info(f" · album {k}: {v}") log.info(f"分轨文件: {stats.get('total_tracks', 0)}") for k, v in sorted((stats.get("by_track_status") or {}).items()): log.info(f" · track {k}: {v}") to_split = [(r, a) for r, a in albums.items() if a.get("status") == Status.ANALYZED.value] if to_split: log.info("") log.info(f"[i] {len(to_split)} 个专辑待切割:") for r, a in to_split[:20]: log.info(f" • {r}: {a['source_audio']} → {len(a['tracks'])} 轨") if len(to_split) > 20: log.info(f" ... 及另外 {len(to_split) - 20} 个") unsupported = [(r, a) for r, a in albums.items() if a.get("status") == Status.UNSUPPORTED.value] if unsupported: log.warning("") log.warning(f"[!] {len(unsupported)} 个专辑源格式不支持 (跳过):") for r, a in unsupported: log.warning(f" • {r}") for issue in a.get("issues", []): log.warning(f" {issue}") failed = [(r, a) for r, a in albums.items() if a.get("status") == Status.FAILED.value] if failed: log.error("") log.error(f"[x] {len(failed)} 个专辑失败 (再次运行会重试):") for r, a in failed: log.error(f" • {r}") for issue in a.get("issues", []): log.error(f" {issue}") failed_tracks = [t for t in a.get("tracks", []) if t.get("status") == Status.FAILED.value] for t in failed_tracks[:3]: log.error(f" ✗ 轨 {t['num']:02d}: {t.get('error')}") def _check_startup_deps(deps: Deps, log: logging.Logger) -> bool: ok = True if not deps.shnsplit: log.error("未找到 shnsplit (shntool). 安装: apt install shntool") ok = False if not deps.cuetag: log.error("未找到 cuetag (cuetools). 安装: apt install cuetools") ok = False if not deps.ffprobe: log.warning("未找到 ffprobe (仅用于源文件元数据展示, 不影响切割)") if not deps.flac: log.warning("未找到 flac 二进制 → FLAC 源将无法切割 (仅支持 WAV)") if not deps.mac: log.info("未找到 mac (APE 解码器) → APE 源将标记 UNSUPPORTED. " "安装可选: apt install monkeys-audio (或从源码构建)") return ok def main() -> int: ap = argparse.ArgumentParser( description="批量用 CUE 切割整轨无损音乐为分轨 (原地切割, 保留原文件, " "shnsplit + cuetag 后端).", formatter_class=argparse.RawDescriptionHelpFormatter, epilog="""示例: # 最简用法 (位置参数) %(prog)s /mnt/z/music # 仅分析, 检查切割计划 %(prog)s /mnt/z/music --analyze-only # 命名参数 %(prog)s -s /mnt/z/music # 强制重跑 (覆盖已有分轨) %(prog)s /mnt/z/music --force """, ) ap.add_argument("source_pos", nargs="?", metavar="SOURCE", help="源目录 (位置参数). 也可用 --source/-s") ap.add_argument("--source", "-s", dest="source_flag", help="源目录 (等价于位置参数 SOURCE)") ap.add_argument("--analyze-only", action="store_true", help="只分析生成 manifest, 不实际切割") ap.add_argument("--force", action="store_true", help="忽略现有 manifest, 强制重新分析并覆盖已有分轨") ap.add_argument("--verbose", "-v", action="store_true", help="详细日志") ap.add_argument("--shnsplit", default="shnsplit", help="shnsplit 路径 (默认从 PATH 查)") ap.add_argument("--cuetag", default="cuetag", help="cuetag 路径") ap.add_argument("--ffprobe", default="ffprobe", help="ffprobe 路径") ap.add_argument("--split-timeout", type=int, default=1800, help="shnsplit 超时秒数 (默认 1800)") args = ap.parse_args() source_arg = args.source_pos or args.source_flag if not source_arg: ap.error("必须提供源目录 (位置参数 SOURCE 或 --source/-s)") logging.basicConfig( level=logging.DEBUG if args.verbose else logging.INFO, format="%(asctime)s %(levelname)-5s %(message)s", datefmt="%H:%M:%S", ) log = logging.getLogger("split_cue") source = Path(source_arg).resolve() if not source.exists(): log.error(f"源目录不存在: {source}") return 2 if not source.is_dir(): log.error(f"源路径不是目录: {source}") return 2 cfg = Config( source_root=source, force=args.force, analyze_only=args.analyze_only, shnsplit=args.shnsplit, cuetag=args.cuetag, ffprobe=args.ffprobe, split_timeout_sec=args.split_timeout, ) deps = probe_deps(cfg) if not _check_startup_deps(deps, log): return 3 manifest_path = source / MANIFEST_NAME manifest = Manifest(manifest_path, cfg) manifest.load_or_init() log.info(f"源目录: {source}") log.info(f"Manifest: {manifest_path}") log.info(f"支持源格式: {sorted(deps.supported_source_exts())}") log.info(f"模式: {'仅分析' if cfg.analyze_only else '分析 + 切割'}") log.info(f"Resume: {'关闭 (--force)' if cfg.force else '开启 (跳过已完成)'}") log.info("") try: analyze_all(cfg, deps, manifest, log) if cfg.analyze_only: print_report(manifest, log) log.info("") log.info("仅分析完成. 查看 manifest 确认切割计划后, 再跑一次进行切割.") return 0 process_all(cfg, manifest, log) print_report(manifest, log) except KeyboardInterrupt: log.warning("") log.warning("用户中断 (Ctrl+C) - 保存 manifest 后退出") try: manifest.save() except OSError as e: log.error(f"保存 manifest 失败: {e}") return 130 return 0 if __name__ == "__main__": sys.exit(main())