1094 lines
39 KiB
Python
1094 lines
39 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
split_cue.py - 批量用 CUE 文件切割整轨无损音乐为分轨.
|
|
|
|
Companion to transcode_music.py. Walks source dir, detects "needs_cue_split"
|
|
albums (single CUE + single whole-disc lossless file), splits them into
|
|
per-track files IN PLACE using shntool (shnsplit) + cuetag. Preserves source
|
|
format when possible:
|
|
FLAC → FLAC (via shnsplit -o flac, needs `flac` binary)
|
|
WAV → WAV (via shnsplit -o wav, native)
|
|
APE → FLAC (needs `mac` decoder; UNSUPPORTED if missing)
|
|
DSF/DFF → UNSUPPORTED (shntool doesn't handle DSD)
|
|
|
|
Whole-disc file is NEVER modified. transcode_music.py's CUE analyzer will
|
|
subsequently see it as "situation C" (CUE + whole-disc + per-track files)
|
|
and skip the whole-disc file, transcoding only the per-track outputs.
|
|
|
|
Split pipeline per album:
|
|
1. shnsplit -f cue -o <fmt> -t "%n" -O always -d <tmpdir> <src>
|
|
produces <tmpdir>/01.<ext>, 02.<ext>, ...
|
|
2. cuetag <cue> <tmpdir>/01.<ext> ... applies CUE tags positionally
|
|
3. os.replace each tmp file → dir/<NN - Title>.<ext> (sanitized name)
|
|
|
|
Dependencies: Python 3.10+, shntool, cuetools (cuetag), ffprobe (for source
|
|
metadata display only). `flac` for FLAC output. `mac` for APE input.
|
|
|
|
Usage:
|
|
# Analyze only (recommended first run — inspect split plan before splitting)
|
|
python3 split_cue.py /mnt/z/music --analyze-only
|
|
|
|
# Split all detected whole-disc albums in-place (auto-resumes)
|
|
python3 split_cue.py /mnt/z/music
|
|
|
|
# Force re-analyze + overwrite existing per-track files
|
|
python3 split_cue.py /mnt/z/music --force
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import datetime as dt
|
|
import errno
|
|
import json
|
|
import logging
|
|
import os
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import threading
|
|
from dataclasses import dataclass
|
|
from enum import Enum
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
__version__ = "1.0.0"
|
|
|
|
MANIFEST_SCHEMA_VERSION = 1
|
|
MANIFEST_NAME = ".split_cue_manifest.json"
|
|
|
|
LOSSLESS_EXTS = {".flac", ".wav", ".dsf", ".dff", ".ape"}
|
|
CUE_EXTS = {".cue"}
|
|
|
|
IGNORE_NAMES = {".ds_store", "thumbs.db", "desktop.ini", ".directory"}
|
|
IGNORE_PREFIXES = ("._",) # macOS resource forks — not user content
|
|
|
|
# CUE files from Chinese/Japanese/Korean releases often use legacy encodings.
|
|
CUE_ENCODINGS = ("utf-8-sig", "utf-8", "gbk", "big5", "shift_jis", "cp936", "latin1")
|
|
|
|
# Filesystem-illegal characters (Windows-superset, safe on Linux/macOS too).
|
|
_ILLEGAL_CHARS_RE = re.compile(r'[<>:"/\\|?*\x00-\x1f]')
|
|
|
|
|
|
class Status(str, Enum):
|
|
PENDING = "pending"
|
|
ANALYZED = "analyzed"
|
|
PROCESSING = "processing"
|
|
COMPLETED = "completed"
|
|
FAILED = "failed"
|
|
SKIPPED = "skipped"
|
|
UNSUPPORTED = "unsupported"
|
|
|
|
|
|
@dataclass
|
|
class Config:
|
|
source_root: Path
|
|
force: bool = False
|
|
analyze_only: bool = False
|
|
shnsplit: str = "shnsplit"
|
|
cuetag: str = "cuetag"
|
|
ffprobe: str = "ffprobe"
|
|
split_timeout_sec: int = 1800
|
|
tag_timeout_sec: int = 60
|
|
|
|
|
|
# ────────────────────────── helpers ──────────────────────────
|
|
|
|
def _now_iso() -> str:
|
|
return dt.datetime.now(dt.timezone.utc).astimezone().isoformat(timespec="seconds")
|
|
|
|
|
|
def _to_int(x) -> Optional[int]:
|
|
if x in (None, "", "N/A"):
|
|
return None
|
|
try:
|
|
return int(x)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
|
|
def _to_float(x) -> Optional[float]:
|
|
if x in (None, "", "N/A"):
|
|
return None
|
|
try:
|
|
return float(x)
|
|
except (TypeError, ValueError):
|
|
return None
|
|
|
|
|
|
def _try_unlink(p: Path) -> None:
|
|
try:
|
|
if p.exists():
|
|
p.unlink()
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _remove_dir(path: Path) -> None:
|
|
try:
|
|
shutil.rmtree(str(path))
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def _is_ignored(p: Path) -> bool:
|
|
name = p.name.lower()
|
|
if name in IGNORE_NAMES:
|
|
return True
|
|
for pref in IGNORE_PREFIXES:
|
|
if name.startswith(pref):
|
|
return True
|
|
if MANIFEST_NAME.lower() in name:
|
|
return True
|
|
return False
|
|
|
|
|
|
def atomic_write_json(path: Path, data: dict) -> None:
|
|
payload = json.dumps(data, ensure_ascii=False, indent=2)
|
|
tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.tmp")
|
|
|
|
fd = os.open(str(tmp), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644)
|
|
try:
|
|
os.write(fd, payload.encode("utf-8"))
|
|
try:
|
|
os.fsync(fd)
|
|
except OSError:
|
|
pass # network mounts (CIFS/NFS) may not implement fsync
|
|
finally:
|
|
os.close(fd)
|
|
|
|
os.replace(str(tmp), str(path))
|
|
|
|
try:
|
|
dfd = os.open(str(path.parent), os.O_RDONLY)
|
|
try:
|
|
os.fsync(dfd) # persist the rename itself (durability guarantee)
|
|
except OSError:
|
|
pass
|
|
finally:
|
|
os.close(dfd)
|
|
except OSError:
|
|
pass
|
|
|
|
|
|
def quarantine_corrupt(path: Path, reason: str) -> Path:
|
|
ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S")
|
|
q = path.with_name(f"{path.name}.corrupt-{ts}")
|
|
try:
|
|
os.replace(str(path), str(q))
|
|
logging.error(f"Manifest 损坏, 已隔离到 {q}: {reason}")
|
|
except OSError as e:
|
|
logging.error(f"隔离失败 {path}: {e}")
|
|
return q
|
|
|
|
|
|
def sanitize_filename(name: str, max_len: int = 200) -> str:
|
|
s = _ILLEGAL_CHARS_RE.sub("_", name)
|
|
s = re.sub(r"\s+", " ", s).strip()
|
|
s = s.rstrip(". ") # Windows dislikes trailing dots/spaces
|
|
if len(s) > max_len:
|
|
s = s[:max_len].rstrip()
|
|
return s or "untitled"
|
|
|
|
|
|
# ────────────────────────── ffprobe ──────────────────────────
|
|
|
|
def ffprobe_audio(path: Path, cfg: Config) -> dict:
|
|
try:
|
|
result = subprocess.run(
|
|
[
|
|
cfg.ffprobe, "-v", "error",
|
|
"-select_streams", "a:0",
|
|
"-show_entries", "stream=codec_name,sample_rate,channels,bits_per_raw_sample",
|
|
"-show_entries", "format=duration",
|
|
"-of", "json",
|
|
str(path),
|
|
],
|
|
capture_output=True, text=True, errors="replace", timeout=60,
|
|
)
|
|
if result.returncode != 0:
|
|
return {"error": (result.stderr or "ffprobe failed").strip()[-200:]}
|
|
data = json.loads(result.stdout or "{}")
|
|
stream = (data.get("streams") or [{}])[0]
|
|
fmt = data.get("format") or {}
|
|
return {
|
|
"codec": stream.get("codec_name"),
|
|
"sample_rate": _to_int(stream.get("sample_rate")),
|
|
"channels": _to_int(stream.get("channels")),
|
|
"duration_sec": _to_float(fmt.get("duration")),
|
|
"bits_per_sample": _to_int(stream.get("bits_per_raw_sample")),
|
|
}
|
|
except subprocess.TimeoutExpired:
|
|
return {"error": "ffprobe timeout"}
|
|
except (json.JSONDecodeError, OSError, UnicodeDecodeError) as e:
|
|
return {"error": str(e)}
|
|
|
|
|
|
# ────────────────────────── CUE parser ──────────────────────────
|
|
|
|
def read_cue_text(path: Path) -> Optional[str]:
|
|
try:
|
|
raw = path.read_bytes()
|
|
except OSError:
|
|
return None
|
|
for enc in CUE_ENCODINGS:
|
|
try:
|
|
return raw.decode(enc)
|
|
except UnicodeDecodeError:
|
|
continue
|
|
return None
|
|
|
|
|
|
def _unquote(s: str) -> str:
|
|
s = s.strip()
|
|
if len(s) >= 2 and s[0] == '"' and s[-1] == '"':
|
|
return s[1:-1]
|
|
return s
|
|
|
|
|
|
def parse_cue_full(path: Path) -> Optional[dict]:
|
|
"""
|
|
Parse a CUE file into structured form.
|
|
|
|
CUE spec semantics:
|
|
TITLE / PERFORMER before any TRACK = album-level
|
|
TITLE / PERFORMER inside a TRACK block = track-level
|
|
REM DATE / REM GENRE at album level = album-level metadata
|
|
"""
|
|
text = read_cue_text(path)
|
|
if text is None:
|
|
return None
|
|
|
|
album_title: Optional[str] = None
|
|
album_performer: Optional[str] = None
|
|
album_date: Optional[str] = None
|
|
album_genre: Optional[str] = None
|
|
files: list[dict] = []
|
|
tracks: list[dict] = []
|
|
current_track: Optional[dict] = None
|
|
|
|
for raw_line in text.splitlines():
|
|
line = raw_line.strip()
|
|
if not line:
|
|
continue
|
|
parts = line.split(None, 1)
|
|
if not parts:
|
|
continue
|
|
keyword = parts[0].upper()
|
|
rest = parts[1] if len(parts) > 1 else ""
|
|
|
|
if keyword == "FILE":
|
|
m = re.match(r'"([^"]*)"\s+(\S+)', rest) or re.match(r"(\S+)\s+(\S+)", rest)
|
|
if m:
|
|
files.append({"name": m.group(1), "format": m.group(2).upper()})
|
|
elif keyword == "TRACK":
|
|
m = re.match(r"(\d+)\s+(\S+)", rest)
|
|
if m:
|
|
current_track = {
|
|
"num": int(m.group(1)),
|
|
"type": m.group(2).upper(),
|
|
"title": None,
|
|
"performer": None,
|
|
}
|
|
tracks.append(current_track)
|
|
elif keyword == "TITLE":
|
|
val = _unquote(rest)
|
|
if current_track is None:
|
|
if album_title is None:
|
|
album_title = val
|
|
else:
|
|
current_track["title"] = val
|
|
elif keyword == "PERFORMER":
|
|
val = _unquote(rest)
|
|
if current_track is None:
|
|
if album_performer is None:
|
|
album_performer = val
|
|
else:
|
|
current_track["performer"] = val
|
|
elif keyword == "REM":
|
|
rem_parts = rest.split(None, 1)
|
|
if len(rem_parts) >= 2 and current_track is None:
|
|
rem_kw = rem_parts[0].upper()
|
|
rem_val = _unquote(rem_parts[1])
|
|
if rem_kw == "DATE" and album_date is None:
|
|
album_date = rem_val
|
|
elif rem_kw == "GENRE" and album_genre is None:
|
|
album_genre = rem_val
|
|
|
|
return {
|
|
"album_title": album_title,
|
|
"album_performer": album_performer,
|
|
"album_date": album_date,
|
|
"album_genre": album_genre,
|
|
"files": files,
|
|
"tracks": tracks,
|
|
}
|
|
|
|
|
|
def match_cue_ref(ref_name: str, audio_paths: list[Path]) -> Optional[Path]:
|
|
"""
|
|
Match a CUE FILE reference to an actual audio file, with fallbacks for
|
|
case differences and extension mismatch (e.g., FILE "album.wav" WAVE
|
|
where the actual file is album.flac).
|
|
"""
|
|
for p in audio_paths:
|
|
if p.name == ref_name:
|
|
return p
|
|
ref_lower = ref_name.lower()
|
|
for p in audio_paths:
|
|
if p.name.lower() == ref_lower:
|
|
return p
|
|
ref_stem = Path(ref_name).stem.lower()
|
|
matches = [p for p in audio_paths if p.stem.lower() == ref_stem]
|
|
if len(matches) == 1:
|
|
return matches[0]
|
|
return None
|
|
|
|
|
|
# ────────────────────────── dependency probing ──────────────────────────
|
|
|
|
@dataclass
|
|
class Deps:
|
|
shnsplit: bool
|
|
cuetag: bool
|
|
ffprobe: bool
|
|
flac: bool
|
|
mac: bool
|
|
|
|
def supported_source_exts(self) -> set[str]:
|
|
supported = {".wav"} # shntool native
|
|
if self.flac:
|
|
supported.add(".flac")
|
|
if self.mac and self.flac:
|
|
supported.add(".ape")
|
|
return supported
|
|
|
|
def unsupported_reason(self, ext: str) -> str:
|
|
if ext == ".ape":
|
|
missing = []
|
|
if not self.mac:
|
|
missing.append("mac (APE 解码器)")
|
|
if not self.flac:
|
|
missing.append("flac")
|
|
return f"缺少依赖: {', '.join(missing)}"
|
|
if ext in (".dsf", ".dff"):
|
|
return "shntool 不支持 DSD, 请先手动转 FLAC"
|
|
if ext == ".flac" and not self.flac:
|
|
return "缺少依赖: flac"
|
|
return f"未知源格式: {ext}"
|
|
|
|
|
|
def probe_deps(cfg: Config) -> Deps:
|
|
return Deps(
|
|
shnsplit=shutil.which(cfg.shnsplit) is not None,
|
|
cuetag=shutil.which(cfg.cuetag) is not None,
|
|
ffprobe=shutil.which(cfg.ffprobe) is not None,
|
|
flac=shutil.which("flac") is not None,
|
|
mac=shutil.which("mac") is not None,
|
|
)
|
|
|
|
|
|
def _output_format_for(src_ext: str) -> tuple[Optional[str], Optional[str]]:
|
|
if src_ext == ".flac":
|
|
return "flac", ".flac"
|
|
if src_ext == ".wav":
|
|
return "wav", ".wav"
|
|
if src_ext == ".ape":
|
|
return "flac", ".flac" # ape → flac (mac decodes, flac encodes)
|
|
return None, None
|
|
|
|
|
|
# ────────────────────────── analysis ──────────────────────────
|
|
|
|
def analyze_directory(dir_path: Path, cfg: Config, deps: Deps) -> dict:
|
|
"""
|
|
Analyze one directory. Detects the "whole-disc CUE + big file" pattern
|
|
and builds a per-track split plan. Non-matching dirs get SKIPPED.
|
|
Sources whose decoder is missing get UNSUPPORTED with reason.
|
|
"""
|
|
entry: dict = {
|
|
"status": Status.PENDING.value,
|
|
"source_audio": None,
|
|
"cue": None,
|
|
"album_title": None,
|
|
"album_performer": None,
|
|
"album_date": None,
|
|
"album_genre": None,
|
|
"source_codec": None,
|
|
"source_sample_rate": None,
|
|
"source_channels": None,
|
|
"duration_sec": None,
|
|
"output_format": None,
|
|
"tracks": [],
|
|
"issues": [],
|
|
"started_at": None,
|
|
"completed_at": None,
|
|
}
|
|
|
|
try:
|
|
files = sorted(p for p in dir_path.iterdir() if p.is_file())
|
|
except OSError as e:
|
|
entry["status"] = Status.FAILED.value
|
|
entry["issues"].append(f"无法读取目录: {e}")
|
|
return entry
|
|
|
|
audio_files = [p for p in files
|
|
if p.suffix.lower() in LOSSLESS_EXTS and not _is_ignored(p)]
|
|
cue_files = [p for p in files
|
|
if p.suffix.lower() in CUE_EXTS and not _is_ignored(p)]
|
|
|
|
if not cue_files or not audio_files:
|
|
entry["status"] = Status.SKIPPED.value
|
|
return entry
|
|
|
|
supported = deps.supported_source_exts()
|
|
|
|
for cue in cue_files:
|
|
parsed = parse_cue_full(cue)
|
|
if parsed is None:
|
|
entry["issues"].append(f"{cue.name}: 无法解析 (编码问题)")
|
|
continue
|
|
|
|
refs = parsed["files"]
|
|
if len(refs) == 0:
|
|
entry["issues"].append(f"{cue.name}: 无 FILE entry")
|
|
continue
|
|
if len(refs) > 1:
|
|
continue # multi-FILE CUE = already per-track index
|
|
|
|
ref_name = refs[0]["name"]
|
|
matched = match_cue_ref(ref_name, audio_files)
|
|
if matched is None:
|
|
entry["issues"].append(f"{cue.name}: 引用 '{ref_name}' 但目录内不存在")
|
|
continue
|
|
|
|
others = [p for p in audio_files if p != matched]
|
|
if others:
|
|
continue # situation C — already split
|
|
|
|
# ── Whole-disc pattern confirmed ──
|
|
cue_tracks = parsed["tracks"]
|
|
if not cue_tracks:
|
|
entry["issues"].append(f"{cue.name}: 无 TRACK 条目")
|
|
continue
|
|
|
|
src_ext = matched.suffix.lower()
|
|
entry["cue"] = cue.name
|
|
entry["source_audio"] = matched.name
|
|
entry["album_title"] = parsed["album_title"]
|
|
entry["album_performer"] = parsed["album_performer"]
|
|
entry["album_date"] = parsed["album_date"]
|
|
entry["album_genre"] = parsed["album_genre"]
|
|
|
|
if src_ext not in supported:
|
|
entry["status"] = Status.UNSUPPORTED.value
|
|
entry["issues"].append(deps.unsupported_reason(src_ext))
|
|
return entry
|
|
|
|
out_format, out_ext = _output_format_for(src_ext)
|
|
entry["output_format"] = out_format
|
|
|
|
probe = ffprobe_audio(matched, cfg)
|
|
if "error" in probe:
|
|
entry["issues"].append(f"ffprobe '{matched.name}': {probe['error']}")
|
|
else:
|
|
entry["source_codec"] = probe.get("codec")
|
|
entry["source_sample_rate"] = probe.get("sample_rate")
|
|
entry["source_channels"] = probe.get("channels")
|
|
entry["duration_sec"] = probe.get("duration_sec")
|
|
|
|
num_digits = max(2, len(str(len(cue_tracks))))
|
|
total_tracks = len(cue_tracks)
|
|
target_names: set[str] = set()
|
|
|
|
for i, cue_track in enumerate(cue_tracks, start=1):
|
|
title = cue_track.get("title") or f"Track {cue_track['num']:02d}"
|
|
num_str = str(cue_track["num"]).zfill(num_digits)
|
|
base = sanitize_filename(f"{num_str} - {title}")
|
|
target = base + out_ext
|
|
n = 2
|
|
while target in target_names:
|
|
target = f"{base} ({n}){out_ext}"
|
|
n += 1
|
|
target_names.add(target)
|
|
entry["tracks"].append({
|
|
"num": cue_track["num"],
|
|
"shnsplit_index": i, # 1-based, matches shnsplit's -t "%n" counter
|
|
"title": title,
|
|
"performer": cue_track.get("performer"),
|
|
"target_name": target,
|
|
"total_tracks": total_tracks,
|
|
"status": Status.PENDING.value,
|
|
"error": None,
|
|
"converted_at": None,
|
|
})
|
|
|
|
entry["status"] = Status.ANALYZED.value
|
|
return entry
|
|
|
|
entry["status"] = Status.SKIPPED.value
|
|
return entry
|
|
|
|
|
|
def merge_entry(existing: dict, fresh: dict) -> dict:
|
|
"""
|
|
Preserve completion state across re-analysis: if a previously-completed
|
|
track has the same target_name, mark it completed in the fresh entry.
|
|
"""
|
|
old_by_name = {t.get("target_name"): t for t in existing.get("tracks") or []}
|
|
for t in fresh.get("tracks", []):
|
|
old = old_by_name.get(t["target_name"])
|
|
if old and old.get("status") == Status.COMPLETED.value:
|
|
t["status"] = Status.COMPLETED.value
|
|
t["converted_at"] = old.get("converted_at")
|
|
t["error"] = None
|
|
if existing.get("started_at"):
|
|
fresh["started_at"] = existing["started_at"]
|
|
return fresh
|
|
|
|
|
|
# ────────────────────────── album splitter (shntool + cuetag) ──────────────────────────
|
|
|
|
def _mark_all_failed(album_entry: dict, err: str) -> None:
|
|
for t in album_entry["tracks"]:
|
|
if t["status"] != Status.COMPLETED.value:
|
|
t["status"] = Status.FAILED.value
|
|
t["error"] = err
|
|
|
|
|
|
# INDEX / PREGAP / POSTGAP with 1-digit fractional (e.g. "57:12.0" or "57:12:0")
|
|
# — non-standard, both shnsplit AND cuetag's internal cueprint reject them.
|
|
# Normalize to canonical CUE "MM:SS:FF" (2-digit CD frame count, 0-74) by
|
|
# interpreting the 1 digit as raw frame count and left-padding to 2 digits.
|
|
# For the overwhelmingly common ".0" case this is exact (0 frames); other
|
|
# values may drift by up to 9 frames (~120ms) which is unavoidable given
|
|
# the original CUE's ambiguity.
|
|
_BAD_CUE_TIME_RE = re.compile(
|
|
r"^([ \t]*(?:INDEX\s+\d+|PREGAP|POSTGAP)\s+)(\d+):(\d+)[.:](\d)[ \t]*$",
|
|
re.IGNORECASE | re.MULTILINE,
|
|
)
|
|
|
|
|
|
def normalize_cue_for_shntool(cue_text: str) -> str:
|
|
cue_text = cue_text.replace("\r\n", "\n").replace("\r", "\n")
|
|
normalized = _BAD_CUE_TIME_RE.sub(
|
|
lambda m: f"{m.group(1)}{m.group(2)}:{m.group(3)}:0{m.group(4)}",
|
|
cue_text,
|
|
)
|
|
# cueprint (used internally by cuetag) rejects files without a final \n.
|
|
if not normalized.endswith("\n"):
|
|
normalized += "\n"
|
|
return normalized
|
|
|
|
|
|
def split_album(
|
|
dir_path: Path,
|
|
album_entry: dict,
|
|
cfg: Config,
|
|
log: logging.Logger,
|
|
) -> bool:
|
|
"""
|
|
Split all pending tracks via shnsplit + cuetag + rename.
|
|
|
|
Returns True on full success (all tracks COMPLETED).
|
|
|
|
Design: shnsplit produces `<tmpdir>/01.<ext>`, `02.<ext>`, ... (zero-padded
|
|
per `-n` format). cuetag writes CUE metadata into each file positionally.
|
|
We then os.replace() each temp file to its sanitized final name in dir_path.
|
|
"""
|
|
src = dir_path / album_entry["source_audio"]
|
|
cue = dir_path / album_entry["cue"]
|
|
tracks = album_entry["tracks"]
|
|
total = len(tracks)
|
|
|
|
if not src.exists():
|
|
log.error(f" 源文件消失: {src}")
|
|
album_entry["issues"].append(f"处理时源文件不存在: {src.name}")
|
|
_mark_all_failed(album_entry, "源文件不存在")
|
|
return False
|
|
if not cue.exists():
|
|
log.error(f" CUE 消失: {cue}")
|
|
album_entry["issues"].append(f"处理时 CUE 不存在: {cue.name}")
|
|
_mark_all_failed(album_entry, "CUE 不存在")
|
|
return False
|
|
|
|
out_format, out_ext = _output_format_for(src.suffix.lower())
|
|
if out_format is None:
|
|
_mark_all_failed(album_entry, f"源格式 {src.suffix} 不支持")
|
|
return False
|
|
|
|
# Skip fast-path: all tracks already completed and files exist on disk.
|
|
if not cfg.force:
|
|
all_done = all(
|
|
t["status"] == Status.COMPLETED.value and (dir_path / t["target_name"]).exists()
|
|
for t in tracks
|
|
)
|
|
if all_done:
|
|
log.debug(f" 所有轨已完成且文件存在, skip")
|
|
return True
|
|
|
|
tmpdir = dir_path / f".split_cue_tmp_{os.getpid()}"
|
|
if tmpdir.exists():
|
|
_remove_dir(tmpdir)
|
|
try:
|
|
tmpdir.mkdir()
|
|
except OSError as e:
|
|
log.error(f" 无法创建临时目录 {tmpdir}: {e}")
|
|
_mark_all_failed(album_entry, f"临时目录创建失败: {e}")
|
|
return False
|
|
|
|
try:
|
|
num_digits = max(2, len(str(total)))
|
|
n_format = f"%0{num_digits}d"
|
|
|
|
# ── Step 0: prepare a UTF-8 + time-normalized CUE for both shnsplit and cuetag ──
|
|
# Feeding original CUE to shnsplit trips on non-standard MM:SS.n times.
|
|
# Feeding original CUE to cuetag corrupts CJK tags because Vorbis Comment
|
|
# requires UTF-8. One prepared CUE solves both.
|
|
prepared_cue: Optional[Path] = None
|
|
cue_text = read_cue_text(cue)
|
|
if cue_text is None:
|
|
log.error(f" CUE 无法解码: {cue.name}")
|
|
_mark_all_failed(album_entry, "CUE 无法解码")
|
|
return False
|
|
cue_text = normalize_cue_for_shntool(cue_text)
|
|
prepared_cue = tmpdir / "_prepared.cue"
|
|
try:
|
|
prepared_cue.write_text(cue_text, encoding="utf-8")
|
|
except OSError as e:
|
|
log.error(f" 临时 CUE 写入失败: {e}")
|
|
_mark_all_failed(album_entry, f"临时 CUE 写入失败: {e}")
|
|
return False
|
|
|
|
# ── Step 1: shnsplit ──
|
|
cmd = [
|
|
cfg.shnsplit,
|
|
"-f", str(prepared_cue),
|
|
"-o", out_format,
|
|
"-t", "%n",
|
|
"-n", n_format,
|
|
"-O", "always",
|
|
"-d", str(tmpdir),
|
|
str(src),
|
|
]
|
|
log.info(f" shnsplit -o {out_format} → {total} 轨")
|
|
try:
|
|
result = subprocess.run(
|
|
cmd, capture_output=True, text=True, errors="replace",
|
|
timeout=cfg.split_timeout_sec,
|
|
)
|
|
except subprocess.TimeoutExpired:
|
|
log.error(f" ✗ shnsplit 超时")
|
|
_mark_all_failed(album_entry, f"shnsplit 超时 ({cfg.split_timeout_sec}s)")
|
|
return False
|
|
except (OSError, UnicodeDecodeError) as e:
|
|
log.error(f" ✗ shnsplit 调用失败: {e}")
|
|
_mark_all_failed(album_entry, f"shnsplit 调用失败: {e}")
|
|
return False
|
|
|
|
if result.returncode != 0:
|
|
stderr = (result.stderr or "").strip()
|
|
err_msg = stderr.splitlines()[-1] if stderr else f"shnsplit exit {result.returncode}"
|
|
log.error(f" ✗ shnsplit 失败: {err_msg}")
|
|
_mark_all_failed(album_entry, err_msg[-400:])
|
|
return False
|
|
|
|
# ── Step 2: verify expected temp files ──
|
|
tmp_files: list[Path] = []
|
|
for i in range(1, total + 1):
|
|
tmp_name = f"{i:0{num_digits}d}{out_ext}"
|
|
f = tmpdir / tmp_name
|
|
try:
|
|
sz = f.stat().st_size
|
|
except OSError:
|
|
sz = 0
|
|
if sz < 1024:
|
|
err = f"shnsplit 未产生预期文件 {tmp_name} (大小 {sz})"
|
|
log.error(f" ✗ {err}")
|
|
_mark_all_failed(album_entry, err)
|
|
return False
|
|
tmp_files.append(f)
|
|
|
|
# ── Step 3: cuetag (positional metadata write) ──
|
|
tag_cmd = [cfg.cuetag, str(prepared_cue)] + [str(f) for f in tmp_files]
|
|
try:
|
|
tag_result = subprocess.run(
|
|
tag_cmd, capture_output=True, text=True, errors="replace",
|
|
timeout=cfg.tag_timeout_sec,
|
|
)
|
|
if tag_result.returncode != 0:
|
|
stderr = (tag_result.stderr or "").strip()
|
|
brief = stderr.splitlines()[-1] if stderr else f"exit {tag_result.returncode}"
|
|
log.warning(f" cuetag 失败, 音频仍可用但可能无标签: {brief}")
|
|
album_entry["issues"].append(f"cuetag 失败: {brief[:200]}")
|
|
except (subprocess.TimeoutExpired, OSError, UnicodeDecodeError) as e:
|
|
log.warning(f" cuetag 调用失败, 音频仍可用: {e}")
|
|
album_entry["issues"].append(f"cuetag 调用失败: {e}")
|
|
|
|
# ── Step 4: rename each temp file to sanitized final name in dir_path ──
|
|
all_ok = True
|
|
for tmp_f, t in zip(tmp_files, tracks):
|
|
dst = dir_path / t["target_name"]
|
|
|
|
if dst.exists() and not cfg.force:
|
|
# Previous-run leftover with matching name — keep it, discard our temp.
|
|
log.debug(f" [已存在, 保留] {t['target_name']}")
|
|
_try_unlink(tmp_f)
|
|
t["status"] = Status.COMPLETED.value
|
|
t["error"] = None
|
|
if not t.get("converted_at"):
|
|
t["converted_at"] = _now_iso()
|
|
continue
|
|
|
|
if dst.exists() and cfg.force:
|
|
try:
|
|
dst.unlink()
|
|
except OSError as e:
|
|
log.error(f" ✗ 无法覆盖 {dst.name}: {e}")
|
|
t["status"] = Status.FAILED.value
|
|
t["error"] = f"无法覆盖: {e}"
|
|
all_ok = False
|
|
continue
|
|
|
|
try:
|
|
os.replace(str(tmp_f), str(dst))
|
|
except OSError as e:
|
|
if e.errno == errno.EXDEV:
|
|
try:
|
|
shutil.move(str(tmp_f), str(dst))
|
|
except OSError as e2:
|
|
log.error(f" ✗ move {tmp_f.name} → {dst.name}: {e2}")
|
|
t["status"] = Status.FAILED.value
|
|
t["error"] = f"move 失败: {e2}"
|
|
all_ok = False
|
|
continue
|
|
else:
|
|
log.error(f" ✗ rename {tmp_f.name} → {dst.name}: {e}")
|
|
t["status"] = Status.FAILED.value
|
|
t["error"] = f"rename 失败: {e}"
|
|
all_ok = False
|
|
continue
|
|
|
|
t["status"] = Status.COMPLETED.value
|
|
t["error"] = None
|
|
t["converted_at"] = _now_iso()
|
|
log.info(f" ✓ [{t['num']:02d}/{total}] {t['target_name']}")
|
|
|
|
return all_ok
|
|
finally:
|
|
_remove_dir(tmpdir)
|
|
|
|
|
|
# ────────────────────────── manifest ──────────────────────────
|
|
|
|
class Manifest:
|
|
def __init__(self, path: Path, cfg: Config):
|
|
self.path = path
|
|
self.cfg = cfg
|
|
self.data: dict = {}
|
|
|
|
def load_or_init(self) -> None:
|
|
if self.path.exists() and not self.cfg.force:
|
|
try:
|
|
text = self.path.read_text(encoding="utf-8")
|
|
data = json.loads(text)
|
|
if not isinstance(data, dict) or not isinstance(data.get("albums"), dict):
|
|
raise ValueError("bad structure")
|
|
self.data = data
|
|
logging.info(f"读取 manifest: {len(self.data['albums'])} 个专辑记录")
|
|
return
|
|
except (json.JSONDecodeError, ValueError, OSError) as e:
|
|
quarantine_corrupt(self.path, str(e))
|
|
self.data = self._new()
|
|
|
|
def _new(self) -> dict:
|
|
now = _now_iso()
|
|
return {
|
|
"manifest_version": MANIFEST_SCHEMA_VERSION,
|
|
"created_at": now,
|
|
"updated_at": now,
|
|
"source_root": str(self.cfg.source_root),
|
|
"stats": {},
|
|
"albums": {},
|
|
}
|
|
|
|
def save(self) -> None:
|
|
self.data["updated_at"] = _now_iso()
|
|
self.data["stats"] = self._compute_stats()
|
|
atomic_write_json(self.path, self.data)
|
|
|
|
def _compute_stats(self) -> dict:
|
|
by_album: dict[str, int] = {}
|
|
by_track: dict[str, int] = {}
|
|
total_tracks = 0
|
|
for album in self.data.get("albums", {}).values():
|
|
s = album.get("status", "unknown")
|
|
by_album[s] = by_album.get(s, 0) + 1
|
|
for t in album.get("tracks") or []:
|
|
ts = t.get("status", "unknown")
|
|
by_track[ts] = by_track.get(ts, 0) + 1
|
|
total_tracks += 1
|
|
return {
|
|
"total_albums": len(self.data.get("albums", {})),
|
|
"by_album_status": by_album,
|
|
"by_track_status": by_track,
|
|
"total_tracks": total_tracks,
|
|
}
|
|
|
|
|
|
# ────────────────────────── driver ──────────────────────────
|
|
|
|
def find_all_dirs(source_root: Path) -> list[Path]:
|
|
"""Include source_root itself so single-album invocations work."""
|
|
result: list[Path] = [source_root]
|
|
for dirpath, dirnames, _ in os.walk(source_root):
|
|
dirnames.sort()
|
|
for d in dirnames:
|
|
result.append(Path(dirpath) / d)
|
|
return result
|
|
|
|
|
|
def analyze_all(cfg: Config, deps: Deps, manifest: Manifest, log: logging.Logger) -> None:
|
|
log.info(f"扫描 {cfg.source_root} ...")
|
|
all_dirs = find_all_dirs(cfg.source_root)
|
|
log.info(f"发现 {len(all_dirs)} 个目录 (含根)")
|
|
|
|
for d in all_dirs:
|
|
rel = "." if d == cfg.source_root else str(d.relative_to(cfg.source_root))
|
|
existing = manifest.data["albums"].get(rel)
|
|
if existing and existing.get("status") == Status.COMPLETED.value and not cfg.force:
|
|
log.debug(f"[已完成 skip] {rel}")
|
|
continue
|
|
log.info(f"[分析] {rel}")
|
|
fresh = analyze_directory(d, cfg, deps)
|
|
if existing:
|
|
fresh = merge_entry(existing, fresh)
|
|
manifest.data["albums"][rel] = fresh
|
|
|
|
manifest.save()
|
|
|
|
|
|
def process_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None:
|
|
albums = manifest.data["albums"]
|
|
keys = sorted(albums.keys())
|
|
total = len(keys)
|
|
|
|
actionable = {Status.ANALYZED.value, Status.PROCESSING.value, Status.FAILED.value}
|
|
|
|
for i, rel in enumerate(keys, 1):
|
|
album = albums[rel]
|
|
status = album.get("status")
|
|
|
|
if status == Status.COMPLETED.value and not cfg.force:
|
|
log.debug(f"[{i}/{total}] SKIP (已完成): {rel}")
|
|
continue
|
|
if status == Status.SKIPPED.value:
|
|
log.debug(f"[{i}/{total}] SKIP (无需切割): {rel}")
|
|
continue
|
|
if status == Status.UNSUPPORTED.value:
|
|
log.warning(f"[{i}/{total}] SKIP (源格式不支持): {rel}")
|
|
continue
|
|
if status not in actionable or not album.get("tracks"):
|
|
log.debug(f"[{i}/{total}] SKIP (状态 {status} 或无计划): {rel}")
|
|
continue
|
|
|
|
log.info(f"[{i}/{total}] 切割: {rel}")
|
|
log.info(f" 源: {album['source_audio']} → {len(album['tracks'])} 轨"
|
|
f" ({album.get('output_format', '?')})")
|
|
|
|
album["status"] = Status.PROCESSING.value
|
|
if album.get("started_at") is None:
|
|
album["started_at"] = _now_iso()
|
|
manifest.save()
|
|
|
|
dir_path = cfg.source_root / rel if rel != "." else cfg.source_root
|
|
try:
|
|
all_ok = split_album(dir_path, album, cfg, log)
|
|
except KeyboardInterrupt:
|
|
manifest.save()
|
|
raise
|
|
except Exception as e: # last-resort safety net: single album must not kill batch
|
|
log.error(f" ✗ 未捕获异常: {type(e).__name__}: {e}")
|
|
album["issues"].append(f"未捕获异常 {type(e).__name__}: {str(e)[:200]}")
|
|
_mark_all_failed(album, f"未捕获异常: {type(e).__name__}: {str(e)[:100]}")
|
|
all_ok = False
|
|
finally:
|
|
manifest.save()
|
|
|
|
album["status"] = Status.COMPLETED.value if all_ok else Status.FAILED.value
|
|
if all_ok:
|
|
album["completed_at"] = _now_iso()
|
|
manifest.save()
|
|
|
|
|
|
def print_report(manifest: Manifest, log: logging.Logger) -> None:
|
|
albums = manifest.data.get("albums", {})
|
|
stats = manifest.data.get("stats") or {}
|
|
|
|
log.info("")
|
|
log.info("=" * 70)
|
|
log.info("摘要")
|
|
log.info("=" * 70)
|
|
log.info(f"总目录数: {stats.get('total_albums', 0)}")
|
|
for k, v in sorted((stats.get("by_album_status") or {}).items()):
|
|
log.info(f" · album {k}: {v}")
|
|
log.info(f"分轨文件: {stats.get('total_tracks', 0)}")
|
|
for k, v in sorted((stats.get("by_track_status") or {}).items()):
|
|
log.info(f" · track {k}: {v}")
|
|
|
|
to_split = [(r, a) for r, a in albums.items()
|
|
if a.get("status") == Status.ANALYZED.value]
|
|
if to_split:
|
|
log.info("")
|
|
log.info(f"[i] {len(to_split)} 个专辑待切割:")
|
|
for r, a in to_split[:20]:
|
|
log.info(f" • {r}: {a['source_audio']} → {len(a['tracks'])} 轨")
|
|
if len(to_split) > 20:
|
|
log.info(f" ... 及另外 {len(to_split) - 20} 个")
|
|
|
|
unsupported = [(r, a) for r, a in albums.items()
|
|
if a.get("status") == Status.UNSUPPORTED.value]
|
|
if unsupported:
|
|
log.warning("")
|
|
log.warning(f"[!] {len(unsupported)} 个专辑源格式不支持 (跳过):")
|
|
for r, a in unsupported:
|
|
log.warning(f" • {r}")
|
|
for issue in a.get("issues", []):
|
|
log.warning(f" {issue}")
|
|
|
|
failed = [(r, a) for r, a in albums.items()
|
|
if a.get("status") == Status.FAILED.value]
|
|
if failed:
|
|
log.error("")
|
|
log.error(f"[x] {len(failed)} 个专辑失败 (再次运行会重试):")
|
|
for r, a in failed:
|
|
log.error(f" • {r}")
|
|
for issue in a.get("issues", []):
|
|
log.error(f" {issue}")
|
|
failed_tracks = [t for t in a.get("tracks", [])
|
|
if t.get("status") == Status.FAILED.value]
|
|
for t in failed_tracks[:3]:
|
|
log.error(f" ✗ 轨 {t['num']:02d}: {t.get('error')}")
|
|
|
|
|
|
def _check_startup_deps(deps: Deps, log: logging.Logger) -> bool:
|
|
ok = True
|
|
if not deps.shnsplit:
|
|
log.error("未找到 shnsplit (shntool). 安装: apt install shntool")
|
|
ok = False
|
|
if not deps.cuetag:
|
|
log.error("未找到 cuetag (cuetools). 安装: apt install cuetools")
|
|
ok = False
|
|
if not deps.ffprobe:
|
|
log.warning("未找到 ffprobe (仅用于源文件元数据展示, 不影响切割)")
|
|
if not deps.flac:
|
|
log.warning("未找到 flac 二进制 → FLAC 源将无法切割 (仅支持 WAV)")
|
|
if not deps.mac:
|
|
log.info("未找到 mac (APE 解码器) → APE 源将标记 UNSUPPORTED. "
|
|
"安装可选: apt install monkeys-audio (或从源码构建)")
|
|
return ok
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser(
|
|
description="批量用 CUE 切割整轨无损音乐为分轨 (原地切割, 保留原文件, "
|
|
"shnsplit + cuetag 后端).",
|
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
epilog="""示例:
|
|
# 最简用法 (位置参数)
|
|
%(prog)s /mnt/z/music
|
|
|
|
# 仅分析, 检查切割计划
|
|
%(prog)s /mnt/z/music --analyze-only
|
|
|
|
# 命名参数
|
|
%(prog)s -s /mnt/z/music
|
|
|
|
# 强制重跑 (覆盖已有分轨)
|
|
%(prog)s /mnt/z/music --force
|
|
""",
|
|
)
|
|
ap.add_argument("source_pos", nargs="?", metavar="SOURCE",
|
|
help="源目录 (位置参数). 也可用 --source/-s")
|
|
ap.add_argument("--source", "-s", dest="source_flag",
|
|
help="源目录 (等价于位置参数 SOURCE)")
|
|
ap.add_argument("--analyze-only", action="store_true",
|
|
help="只分析生成 manifest, 不实际切割")
|
|
ap.add_argument("--force", action="store_true",
|
|
help="忽略现有 manifest, 强制重新分析并覆盖已有分轨")
|
|
ap.add_argument("--verbose", "-v", action="store_true", help="详细日志")
|
|
ap.add_argument("--shnsplit", default="shnsplit", help="shnsplit 路径 (默认从 PATH 查)")
|
|
ap.add_argument("--cuetag", default="cuetag", help="cuetag 路径")
|
|
ap.add_argument("--ffprobe", default="ffprobe", help="ffprobe 路径")
|
|
ap.add_argument("--split-timeout", type=int, default=1800,
|
|
help="shnsplit 超时秒数 (默认 1800)")
|
|
args = ap.parse_args()
|
|
|
|
source_arg = args.source_pos or args.source_flag
|
|
if not source_arg:
|
|
ap.error("必须提供源目录 (位置参数 SOURCE 或 --source/-s)")
|
|
|
|
logging.basicConfig(
|
|
level=logging.DEBUG if args.verbose else logging.INFO,
|
|
format="%(asctime)s %(levelname)-5s %(message)s",
|
|
datefmt="%H:%M:%S",
|
|
)
|
|
log = logging.getLogger("split_cue")
|
|
|
|
source = Path(source_arg).resolve()
|
|
if not source.exists():
|
|
log.error(f"源目录不存在: {source}")
|
|
return 2
|
|
if not source.is_dir():
|
|
log.error(f"源路径不是目录: {source}")
|
|
return 2
|
|
|
|
cfg = Config(
|
|
source_root=source,
|
|
force=args.force,
|
|
analyze_only=args.analyze_only,
|
|
shnsplit=args.shnsplit,
|
|
cuetag=args.cuetag,
|
|
ffprobe=args.ffprobe,
|
|
split_timeout_sec=args.split_timeout,
|
|
)
|
|
|
|
deps = probe_deps(cfg)
|
|
if not _check_startup_deps(deps, log):
|
|
return 3
|
|
|
|
manifest_path = source / MANIFEST_NAME
|
|
manifest = Manifest(manifest_path, cfg)
|
|
manifest.load_or_init()
|
|
|
|
log.info(f"源目录: {source}")
|
|
log.info(f"Manifest: {manifest_path}")
|
|
log.info(f"支持源格式: {sorted(deps.supported_source_exts())}")
|
|
log.info(f"模式: {'仅分析' if cfg.analyze_only else '分析 + 切割'}")
|
|
log.info(f"Resume: {'关闭 (--force)' if cfg.force else '开启 (跳过已完成)'}")
|
|
log.info("")
|
|
|
|
try:
|
|
analyze_all(cfg, deps, manifest, log)
|
|
if cfg.analyze_only:
|
|
print_report(manifest, log)
|
|
log.info("")
|
|
log.info("仅分析完成. 查看 manifest 确认切割计划后, 再跑一次进行切割.")
|
|
return 0
|
|
process_all(cfg, manifest, log)
|
|
print_report(manifest, log)
|
|
except KeyboardInterrupt:
|
|
log.warning("")
|
|
log.warning("用户中断 (Ctrl+C) - 保存 manifest 后退出")
|
|
try:
|
|
manifest.save()
|
|
except OSError as e:
|
|
log.error(f"保存 manifest 失败: {e}")
|
|
return 130
|
|
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|