From 31d02dd1bd1f6a3eeb47c3a40d5033ad890c6107 Mon Sep 17 00:00:00 2001 From: admin Date: Tue, 8 Sep 2026 06:47:00 +0800 Subject: [PATCH] add split cue script --- transcode_music/split_cue.md | 1152 ++++++++++++++++++++++++++++++++++ transcode_music/split_cue.py | 1093 ++++++++++++++++++++++++++++++++ 2 files changed, 2245 insertions(+) create mode 100644 transcode_music/split_cue.md create mode 100644 transcode_music/split_cue.py diff --git a/transcode_music/split_cue.md b/transcode_music/split_cue.md new file mode 100644 index 0000000..a0a0309 --- /dev/null +++ b/transcode_music/split_cue.md @@ -0,0 +1,1152 @@ +# `split_cue.py` 使用手册 + +批量用 CUE 文件把"整轨大文件 + CUE"式专辑切成分轨的 Python 脚本,支持断点续跑、CJK 编码 CUE、多种源格式,以 shntool + cuetools 作为切割/标签后端。 + +设计上是 [`transcode_music.py`](./transcode_music.md) 的**前置伴生工具**:先跑 `split_cue.py` 把整轨拆开,再跑 `transcode_music.py` 转 MP3,两者组合无缝衔接。 + +--- + +## 目录 + +- [1. 简介](#1-简介) +- [2. 系统要求](#2-系统要求) +- [3. 快速上手](#3-快速上手) +- [4. 命令行参数详解](#4-命令行参数详解) +- [5. 常用使用场景](#5-常用使用场景) +- [6. Manifest 文件说明](#6-manifest-文件说明) +- [7. 支持的源格式与依赖矩阵](#7-支持的源格式与依赖矩阵) +- [8. CUE 检测与匹配逻辑](#8-cue-检测与匹配逻辑) +- [9. 分轨命名与元数据](#9-分轨命名与元数据) +- [10. 与 transcode_music.py 的组合工作流](#10-与-transcode_musicpy-的组合工作流) +- [11. 错误处理与恢复](#11-错误处理与恢复) +- [12. 已知局限](#12-已知局限) +- [13. FAQ](#13-faq) + +--- + +## 1. 简介 + +**这个工具解决什么问题?** + +你有很多"整轨型"专辑:一张 CD 打包成一个大 FLAC / WAV / APE 文件,配一个 CUE 索引,例如: + +``` +王菲 - 唱游 [FLAC+CUE]/ +├── CDImage.flac (整张 CD 的一个 40 分钟大文件) +└── CDImage.cue (12 个 TRACK 条目, 标注每首歌的起止) +``` + +这类文件不方便用普通播放器逐首播放(除非播放器专门支持 CUE),也不方便直接转成 MP3(`transcode_music.py` 会检测出这种情况并报警跳过)。 + +**`split_cue.py` 就是用来把这类专辑批量拆成分轨的**:扫描整个源目录树,找出所有"整轨 CUE"型专辑,用 `shnsplit` 按 CD 帧边界(1/75 秒精度)切成 `01 - 红豆.flac`、`02 - 催眠.flac` 这样的分轨文件,用 `cuetag` 写入 CUE 里的标题/艺人/专辑等元数据,全部**原地切割**、**保留原文件不动**。 + +**核心特性**: + +| 特性 | 说明 | +|---|---| +| 原地切割 | 分轨文件写在原专辑目录里,与整轨大文件并存 | +| 保留原文件 | 整轨的 `CDImage.flac` 从不被删改,可事后自己决定去留 | +| 保持源格式 | FLAC → FLAC 无损、WAV → WAV、APE → FLAC(ffmpeg 无 APE 编码器) | +| Sample-exact 切割 | shntool 按 CD 帧对齐,每首歌起止精确到 1/75 秒 | +| 完整元数据 | cuetag 自动写入 TITLE / ALBUM / ARTIST / track / TRACKTOTAL / DATE / GENRE | +| CJK 支持 | CUE 编码自动识别(utf-8/utf-8-sig/gbk/big5/shift_jis/cp936),中日韩音源都能处理 | +| 断点续跑 | `.split_cue_manifest.json` 记录状态,任意时刻中断都能从上次位置继续 | +| 幂等 | 已完成的专辑再次运行秒跳过;新增专辑增量处理 | +| 自动排除已分轨 | 若目录里已有分轨文件(situation C),自动 SKIP 不误切 | +| 依赖自适应 | 启动时 probe 依赖,缺 `mac` 时把 APE 源标 UNSUPPORTED 而非崩溃 | + +--- + +## 2. 系统要求 + +| 组件 | 版本 | 必需/可选 | 说明 | +|---|---|---|---| +| Python | 3.10+ | 必需 | 只用标准库,无需 pip install | +| shntool | 3.0+ | **必需** | 提供 `shnsplit` 切割命令 | +| cuetools | 1.4+ | **必需** | 提供 `cuetag` 元数据写入命令 | +| flac | 1.3+ | 处理 FLAC 时必需 | shntool 编码 FLAC 输出时调用 | +| mac (Monkey's Audio) | 4.x+ | 处理 APE 时必需 | shntool 解码 APE 输入时调用 | +| ffprobe | 6.0+ | 可选 | 仅用于源文件元数据展示(时长、采样率),不影响切割 | + +**检查环境**: + +```bash +python3 --version +shnsplit -v | head -1 +cuetag --help 2>&1 | head -1 +flac --version | head -1 +mac 2>&1 | head -1 # 无输出/命令不存在 = APE 源将标 UNSUPPORTED +ffprobe -version | head -1 +``` + +**安装依赖(Ubuntu / Debian / WSL)**: + +```bash +sudo apt install shntool cuetools flac +# APE 支持(可选):Ubuntu 官方仓库通常没有,需要第三方 PPA 或源码编译 +sudo apt install monkeys-audio # 如果仓库有的话 +``` + +**macOS**: + +```bash +brew install shntool cuetools flac +# APE:需从 https://monkeysaudio.com 下载或用其他方式 +``` + +**Windows**:直接跑不推荐(有 `os.fsync(dir_fd)` 的 POSIX 依赖),建议 WSL2。 + +**依赖缺失时的行为**: + +启动时脚本会 probe 所有依赖并打印状态。缺 `shnsplit` 或 `cuetag` 直接退出(exit code 3)。缺 `mac` 或 `flac` 只影响对应格式,其他格式仍能处理。 + +--- + +## 3. 快速上手 + +**最常用的三个命令**: + +```bash +# 1. 先分析看看哪些专辑需要切割 (不实际切) +python3 split_cue.py /mnt/z/我的音乐 --analyze-only + +# 2. 确认没问题后正式切割 (中断可再跑, 自动 resume) +python3 split_cue.py /mnt/z/我的音乐 + +# 3. 切完了继续转 MP3 (transcode_music.py 会自动跳过整轨大文件) +python3 transcode_music.py /mnt/z/我的音乐 /mnt/x/mp3输出 +``` + +就这么简单。整轨型专辑切成分轨后,`transcode_music.py` 会自动识别为"situation C"(CUE 引用整轨但存在分轨),跳过整轨文件只转分轨——两个脚本天然衔接。 + +--- + +## 4. 命令行参数详解 + +### 完整语法 + +``` +split_cue.py [SOURCE] [选项...] +``` + +或 + +``` +split_cue.py --source SOURCE [选项...] +``` + +### 参数列表 + +| 参数 | 短选项 | 类型 | 是否必需 | 默认 | 说明 | +|---|---|---|---|---|---| +| `SOURCE` | — | 位置参数 | 必需¹ | — | 源目录,含整轨型专辑的位置 | +| `--source` | `-s` | 字符串 | 必需¹ | — | 源目录(与位置参数 SOURCE 等价) | +| `--analyze-only` | — | 开关 | 可选 | 关 | 只分析生成 manifest,不实际切割 | +| `--force` | — | 开关 | 可选 | 关 | 忽略现有 manifest,全部重新分析并**覆盖**已有分轨 | +| `--verbose` | `-v` | 开关 | 可选 | 关 | 显示 DEBUG 级别日志(含跳过项) | +| `--shnsplit` | — | 字符串 | 可选 | `shnsplit` | shnsplit 可执行路径 | +| `--cuetag` | — | 字符串 | 可选 | `cuetag` | cuetag 可执行路径 | +| `--ffprobe` | — | 字符串 | 可选 | `ffprobe` | ffprobe 可执行路径 | +| `--split-timeout` | — | 整数 | 可选 | `1800` | 单个专辑 shnsplit 超时秒数 | +| `--help` | `-h` | 开关 | — | — | 显示帮助并退出 | + +¹ 源目录必须提供,可用位置参数或 flag 两种形式: + +```bash +# 位置参数 (更简洁) +split_cue.py /mnt/z/music + +# flag 长形式 +split_cue.py --source /mnt/z/music + +# flag 短形式 +split_cue.py -s /mnt/z/music +``` + +注意:**`split_cue.py` 不需要目标目录**,因为它是原地切割——分轨文件直接写到原专辑目录里。这与 `transcode_music.py` 需要 `-s / -t` 两个参数不同。 + +### 参数详细说明 + +#### `SOURCE` / `--source` / `-s` + +**源目录**。脚本会递归扫描此目录及所有子目录,找出每个"整轨 CUE"型专辑处理。 + +- 支持含空格、中文、特殊字符的路径 +- 网盘挂载(`/mnt/z/`、`/mnt/x/` 等)都可以 +- Manifest 文件 `.split_cue_manifest.json` 写入此目录 + +**特殊约定**:如果 SOURCE 本身就是一个专辑目录(而不是一堆专辑的父目录),也能正常工作。脚本会把源目录自身也作为分析候选。 + +#### `--analyze-only` + +**只分析,不切割**。用途: + +- 首次跑一个新的大目录时,先看看总共有几个专辑需要切、切成几轨 +- 检查 CUE 解析是否有问题(编码、缺 FILE 引用等) +- 生成 manifest 后可以手动查看/编辑 + +**强烈推荐**每次对新目录都先跑一次 `--analyze-only`。切割是**原地写文件**的操作,先看清楚计划再动手更安全。 + +#### `--force` + +**强制重跑**。忽略现有 manifest,重新分析并**覆盖**已存在的分轨文件。 + +**什么时候用**: +- 之前切错了想重来 +- CUE 修改过了,想重新按新 CUE 切 +- 想清理并重新开始 + +**警告**:`--force` 会覆盖已存在的同名分轨文件。如果之前有你手动编辑过的分轨,会丢失。**不确定时先备份**。 + +**边界情况**:如果专辑的 status 已经因为"situation C"被判为 `skipped`(因为目录里已有分轨),`--force` 也**不会**自动 un-skip。因为脚本无法可靠区分"我上次生成的分轨"和"用户已有的分轨"——安全起见不动。想强制重切,请先手动删掉旧的分轨文件再跑。 + +#### `--verbose` / `-v` + +打开 DEBUG 级别日志。默认 INFO 级别只显示每首切割和错误。加上 `-v` 会额外显示: + +- `[已完成 skip] ...` 每个跳过的专辑 +- `[SKIP (无需切割): ...]` 每个 skipped 目录 +- `[已存在, 保留] ...` 每个已有分轨文件 + +大量输出,通常只在排查问题时用。 + +#### `--shnsplit` / `--cuetag` / `--ffprobe` + +指定可执行文件路径。默认从 `PATH` 查找。 + +**什么时候用**: +- 装了多个版本想指定用哪个 +- 二进制不在标准 PATH 里 +- 用自己编译的版本 + +示例: + +```bash +split_cue.py /src --shnsplit /opt/shntool-3.0.10/bin/shnsplit +``` + +#### `--split-timeout` + +单个专辑 shnsplit 命令的超时时间(秒),默认 `1800`(30 分钟)。超时算失败,写入 manifest 后继续下一个。 + +**通常不需要调**。除非你有超长专辑(比如一个 3 小时的整轨 audiobook)在慢网盘上,可以调大到 3600 或更多。 + +注意:shnsplit 是**单次调用切完整个专辑**的,不像 ffmpeg 是逐轨调用,所以超时值应该覆盖整个专辑的切割时间。 + +--- + +## 5. 常用使用场景 + +### 场景 1:第一次处理新目录(推荐流程) + +假设你刚下载了一堆 `[FLAC+CUE]` 到 `/mnt/z/新收藏/`。 + +**第一步:先分析** + +```bash +python3 split_cue.py /mnt/z/新收藏 --analyze-only +``` + +输出会告诉你: + +- 总共扫了几个目录 +- 有几个专辑属于"整轨型"需要切割(`analyzed`) +- 有几个源格式不支持(`unsupported`) +- 有几个已经是分轨的(`skipped`) + +**第二步:查看切割计划** + +```bash +python3 -c " +import json +d = json.load(open('/mnt/z/新收藏/.split_cue_manifest.json')) +for rel, a in d['albums'].items(): + if a['status'] == 'analyzed': + print(f'{rel}: {a[\"source_audio\"]} → {len(a[\"tracks\"])} 轨') + for t in a['tracks'][:3]: + print(f' · {t[\"target_name\"]}') + if len(a['tracks']) > 3: + print(f' ... 及另外 {len(a[\"tracks\"]) - 3} 轨') +" +``` + +**第三步:确认无误后正式切割** + +```bash +python3 split_cue.py /mnt/z/新收藏 +``` + +已经 `analyzed` 的专辑会开始切割,`skipped` / `unsupported` 的自动跳过。 + +**第四步:接着跑 transcode_music.py 转 MP3** + +```bash +python3 transcode_music.py /mnt/z/新收藏 /mnt/x/music/新收藏 +``` + +`transcode_music.py` 会自动识别切出来的分轨(situation C),跳过整轨大文件只转分轨。 + +### 场景 2:中断后恢复 + +切到一半按 Ctrl+C 中断(或断电、脚本崩溃)。只需要**再跑同样的命令**: + +```bash +python3 split_cue.py /mnt/z/新收藏 +``` + +脚本会读取 manifest,跳过所有 `completed` 状态的专辑,从上次中断的地方继续。不需要任何特殊参数。 + +### 场景 3:增量同步(新增了整轨专辑) + +以后又下载了几个 `[FLAC+CUE]` 到 `/mnt/z/新收藏/`: + +```bash +python3 split_cue.py /mnt/z/新收藏 +``` + +已完成的秒跳过,只有新目录会被分析和切割。最舒服的用法。 + +### 场景 4:修复失败的专辑 + +如果某个专辑因为 CUE 解析错误、shnsplit 崩溃等失败了(`status: failed`),修好后: + +```bash +# 直接再跑, failed 的会被自动重试 +python3 split_cue.py /mnt/z/新收藏 +``` + +已成功的不动,只重试失败的。 + +### 场景 5:只切一个专辑 + +比如只想切某一张: + +```bash +python3 split_cue.py "/mnt/z/新收藏/王菲 - 唱游 [FLAC+CUE]" +``` + +Manifest 会写到那个专辑目录里,其他专辑不受影响。 + +### 场景 6:想重新切某个已完成的专辑 + +比如切完发现 CUE 元数据有错,改了 CUE 想重切: + +**方法一:删掉分轨文件 + 编辑 manifest** + +删除该专辑目录里的分轨文件(`01 - Title.flac` 等),然后编辑 `.split_cue_manifest.json`,把该专辑的 status 改成 `analyzed`、每个 track 的 status 改成 `pending`。再跑一次即可。 + +**方法二:全局 --force** + +```bash +python3 split_cue.py /mnt/z/新收藏 --force +``` + +会重新分析所有专辑并覆盖分轨。但注意:**如果目录里已有旧分轨且脚本无法区分**,可能会被判为 situation C 而 skip。最保险的做法还是先手动删旧分轨。 + +### 场景 7:排查某个专辑的问题 + +某个专辑 status=failed,用 verbose 模式跑: + +```bash +python3 split_cue.py "/mnt/z/新收藏/有问题的专辑" -v +``` + +输出会显示 shnsplit 的完整错误、cuetag 警告等。 + +### 场景 8:切完不满意,回滚 + +因为原文件从不被删改,回滚很简单: + +```bash +# 删掉所有分轨文件, 只留原整轨 + CUE +cd "/mnt/z/新收藏/王菲 - 唱游" +rm -f *.flac # 删所有 flac +# 找回原整轨 (脚本没删过它, 只是你上面 rm 一起删了; 如果只想删分轨:) +# 或更精确: 只删符合 "NN - Title.flac" 命名的 +``` + +更精确的清理: + +```bash +# 只删 "01 - ...", "02 - ..." 这种格式的文件 +find "/mnt/z/新收藏/王菲 - 唱游" -maxdepth 1 -regex '.*/[0-9]\{2\} - .*\.\(flac\|wav\)$' -delete + +# 编辑 manifest 把该专辑 status 改回 analyzed +``` + +--- + +## 6. Manifest 文件说明 + +### 位置 + +Manifest 保存在**源目录**下,文件名 `.split_cue_manifest.json`(隐藏文件)。 + +### 顶层结构 + +```json +{ + "manifest_version": 1, + "created_at": "2026-09-06T13:41:03+08:00", + "updated_at": "2026-09-06T13:41:04+08:00", + "source_root": "/mnt/z/新收藏", + "stats": { + "total_albums": 4, + "by_album_status": {"completed": 1, "skipped": 3}, + "by_track_status": {"completed": 3}, + "total_tracks": 3 + }, + "albums": { + "专辑相对路径1": { ... }, + "专辑相对路径2": { ... } + } +} +``` + +### 单个专辑 (album) 结构 + +```json +{ + "status": "completed", + "source_audio": "CDImage.flac", + "cue": "CDImage.cue", + "album_title": "唱游", + "album_performer": "王菲", + "album_date": "1998", + "album_genre": "Chinese Pop", + "source_codec": "flac", + "source_sample_rate": 44100, + "source_channels": 2, + "duration_sec": 40.0, + "output_format": "flac", + "tracks": [ + { + "num": 1, + "shnsplit_index": 1, + "title": "红豆", + "performer": "王菲", + "target_name": "01 - 红豆.flac", + "total_tracks": 3, + "status": "completed", + "error": null, + "converted_at": "2026-09-06T13:41:04+08:00" + }, + { "num": 2, "shnsplit_index": 2, ... }, + { "num": 3, "shnsplit_index": 3, ... } + ], + "issues": [], + "started_at": "2026-09-06T13:41:03+08:00", + "completed_at": "2026-09-06T13:41:04+08:00" +} +``` + +**字段含义**: + +| 字段 | 含义 | +|---|---| +| `status` | 专辑级状态,见下表 | +| `source_audio` | 检测到的整轨大文件名(相对路径) | +| `cue` | 匹配的 CUE 文件名 | +| `album_title` / `album_performer` | 从 CUE 的顶级 TITLE / PERFORMER 解析 | +| `album_date` / `album_genre` | 从 CUE 的 REM DATE / REM GENRE 解析 | +| `source_codec` / `source_sample_rate` / `source_channels` / `duration_sec` | 由 ffprobe 探测(仅用于展示) | +| `output_format` | shnsplit `-o` 参数(`flac` 或 `wav`) | +| `tracks[]` | 切割计划,每首歌一项 | +| `tracks[].num` | CUE 里声明的 TRACK 编号 | +| `tracks[].shnsplit_index` | shnsplit 内部计数器(1-based,用于对应 tmp 文件名) | +| `tracks[].title` / `tracks[].performer` | 从 CUE 的 TRACK 级 TITLE / PERFORMER 解析 | +| `tracks[].target_name` | 最终文件名(已 sanitize) | +| `tracks[].status` | 分轨级状态 | +| `issues[]` | 人类可读的警告/说明 | + +### 状态值 (status) + +**专辑级状态**: + +| status | 说明 | 下次运行时行为 | +|---|---|---| +| `pending` | 初始状态(一般用不到) | 重新分析 | +| `analyzed` | 已识别为整轨 CUE 型,切割计划就绪 | 执行切割 | +| `processing` | 正在切割(若看到,说明上次中断了) | 重新切割整个专辑 | +| `completed` | 所有分轨都成功输出 | **秒跳过** | +| `failed` | 切割失败(shnsplit / cuetag / 重命名任一失败) | 重试整个专辑 | +| `skipped` | 无需切割(无 CUE / 已分轨 situation C / CUE 引用多文件 / 引用不存在) | 跳过 | +| `unsupported` | 源格式当前依赖无法处理(如 APE 但 `mac` 未装、DSF) | 跳过 | + +**分轨级状态**(tracks[] 里): + +| status | 说明 | +|---|---| +| `pending` | 待处理 | +| `completed` | 已完成 | +| `failed` | 失败(`error` 字段有原因) | + +### 手动干预 manifest + +Manifest 是纯 JSON,可以用文本编辑器直接改。 + +**例子 1:让某个专辑重新切割** + +```json +"status": "completed" → "status": "analyzed" +``` + +同时把该专辑内所有 track 的 status 从 `completed` 改成 `pending`,删掉现有分轨文件,再跑脚本。 + +**例子 2:跳过某个有问题的专辑** + +```json +"status": "analyzed" → "status": "skipped" +``` + +**例子 3:删除某个专辑的记录重新扫描** + +删除对应的 album 条目,下次运行会自动重新扫描添加。 + +**改完保存直接跑**: + +```bash +python3 split_cue.py /mnt/z/新收藏 +``` + +### Manifest 损坏怎么办? + +如果 JSON 损坏(编辑错、盘故障),脚本启动时会自动检测,把坏的 manifest 重命名为 `.split_cue_manifest.json.corrupt-<时间戳>`,然后**重新生成新的 manifest**。 + +已切出来的分轨文件不会丢,但脚本会重新扫描分析。因为已切分的目录会呈现为 situation C(有分轨 + 有整轨),会自动 SKIP,不会重复切割。**幂等性保护**。 + +--- + +## 7. 支持的源格式与依赖矩阵 + +### 支持矩阵 + +| 源格式 | 输出格式 | 依赖 | 备注 | +|---|---|---|---| +| `.flac` | `.flac` | `shntool` + `flac` | 无损 re-encode,sample-exact | +| `.wav` | `.wav` | `shntool` 原生 | 无外部依赖,PCM 样本直切 | +| `.ape` | `.flac` | `shntool` + `mac` + `flac` | mac 解码 APE,flac 编码输出(ffmpeg 无 APE 编码器) | +| `.dsf` / `.dff` | — | 无 | **不支持** (DSD)。请先手动转 FLAC 再切 | + +### 依赖启动时 probe + +启动时会打印所有依赖状态: + +``` +未找到 mac (APE 解码器) → APE 源将标记 UNSUPPORTED. 安装可选: apt install monkeys-audio +支持源格式: ['.flac', '.wav'] +``` + +**缺 shnsplit 或 cuetag**:直接退出(exit 3)。这是硬依赖。 + +**缺 flac**:FLAC 源会被标为 UNSUPPORTED。只有 WAV 源能处理。 + +**缺 mac**:APE 源会被标为 UNSUPPORTED,其他格式不受影响。 + +**缺 ffprobe**:警告但不影响切割,只是 manifest 里 `source_codec` / `duration_sec` 等字段是空的。 + +### 为什么不支持 DSD? + +`shntool` 本身不支持 DSD 格式。理论上可以用 ffmpeg 走另一条路,但: + +1. DSD 打包成整轨很少见(DSD 音源通常出货就是分轨的) +2. DSD 解码涉及低通滤波、降采样等,输出到 FLAC 会显著改变数据 +3. 处理逻辑复杂度不值得 + +如果你确实有 DSD 整轨 + CUE,建议先手动用 ffmpeg 转成 FLAC,再跑 split_cue.py: + +```bash +ffmpeg -i album.dsf -c:a flac -ar 88200 album.flac +# 然后修改 CUE 里的 FILE 指向 album.flac +split_cue.py /path/to/album +``` + +--- + +## 8. CUE 检测与匹配逻辑 + +### 什么样的目录会被识别为"整轨 CUE 型"? + +严格判定条件(**全部满足**才算): + +1. 目录里**至少**有 1 个 CUE 文件 +2. 目录里**至少**有 1 个无损音频文件(FLAC/WAV/APE/DSF/DFF) +3. 某个 CUE 里 `FILE "xxx" WAVE` 条目**恰好 1 个** +4. 这个 FILE 引用能匹配到目录里的音频文件 +5. 目录里**没有其他无损音频文件**(除了这个被引用的) +6. CUE 里**有 TRACK 条目**(不能是空 CUE) + +只要有一项不满足,就不会被切割: + +| 情况 | 判定 | 原因 | +|---|---|---| +| 无 CUE | `skipped` | 没有切割依据 | +| 无无损音频 | `skipped` | 没东西可切 | +| CUE 无 FILE 条目 | `skipped` + issues 记录 | CUE 无效 | +| CUE 有多个 FILE (>1) | `skipped` | 已经是分轨 CUE 索引 | +| CUE 引用的文件不存在 | `skipped` + issues 记录 | 可能是错配的 CUE | +| CUE 引用整轨但同时存在分轨 | `skipped` (situation C) | 已切过或用户手动分轨了 | +| 匹配成功但源格式依赖缺失 | `unsupported` | 例如 APE 但 mac 未装 | + +### CUE FILE 引用的匹配策略 + +CUE 里的 `FILE "album.wav" WAVE` 未必和实际文件名完全一致。脚本用三级 fallback: + +1. **精确匹配**:文件名完全一致 +2. **不区分大小写**:`ALBUM.FLAC` 匹配 `album.flac` +3. **忽略扩展名**:`FILE "album.wav" WAVE` 匹配到 `album.flac`(假设目录里只有一个 `album.*` 无损文件) + +三级都不匹配才判为"引用不存在"。 + +### CUE 编码识别 + +CUE 文件常见的编码: + +- `utf-8-sig`(有 BOM 的 UTF-8) +- `utf-8` +- `gbk`(简体中文) +- `big5`(繁体中文) +- `shift_jis`(日文) +- `cp936`(GBK 别名) +- `latin1`(最后兜底) + +脚本按顺序尝试解码,第一个成功的即采用。中日韩音源基本都能识别。 + +### CUE 里 TITLE / PERFORMER 的语义 + +CUE 里 `TITLE "..."` 和 `PERFORMER "..."` 可能出现在两个位置: + +- **顶级**(在任何 TRACK 之前):整张专辑的标题和艺人 +- **TRACK 块内**:这一首歌的标题和艺人 + +脚本区分这两种上下文,分别写入 `album_title` / `album_performer` 和每首歌的 `title` / `performer`。 + +### REM 元数据 + +CUE 里 `REM DATE "1998"` 和 `REM GENRE "Chinese Pop"` 在顶级出现时,会被解析为专辑的年份和流派,写入 manifest 并通过 cuetag 传给分轨的 tag。 + +### CUE 与 transcode_music.py 的区别 + +`transcode_music.py` 也有 CUE 检测逻辑(在其文档第 8 节详述了三种情况 A/B/C)。它的目的是**报警**"这个专辑需要 CUE split",而 `split_cue.py` 是**执行**这个 split。 + +两者对 CUE 的看法一致: + +| 情况 | transcode_music.py 视角 | split_cue.py 视角 | +|---|---|---| +| A: CUE + 整轨大文件 | `needs_cue_split` 报警跳过 | **`analyzed`,执行切割** | +| B: CUE + 分轨 | 正常转码 | `skipped` | +| C: CUE 引用整轨 + 分轨都在 | 跳过整轨只转分轨 | `skipped` | + +这就是为什么它们能天然衔接:`split_cue.py` 把 A 转成 C,然后 `transcode_music.py` 处理 C。 + +--- + +## 9. 分轨命名与元数据 + +### 文件命名规则 + +分轨文件名格式:**`NN - Title.ext`** + +- **NN**:CUE 里 TRACK 声明的编号,零填充到至少 2 位(如果总轨数 ≥ 100 会用 3 位) +- **Title**:CUE 里 TRACK 块内的 TITLE,经过文件名净化 +- **ext**:由源格式决定的输出扩展名(见第 7 节表格) + +**Title 净化规则**: + +1. 替换文件系统非法字符 `<>:"/\|?*` 以及 ASCII 控制字符(`\x00-\x1f`)为 `_` +2. 折叠连续空白为单个空格 +3. 去掉结尾的点号和空格(Windows 要求) +4. 截断到最多 200 字符(避免文件系统限制) +5. 净化后为空则用 `untitled` 兜底 + +**示例**: + +| CUE 里的 TITLE | 输出文件名(假设 num=2) | +|---|---| +| `红豆` | `02 - 红豆.flac` | +| `催眠 / 静夜` | `02 - 催眠 _ 静夜.flac`(`/` → `_`) | +| `"Track: A"` | `02 - _Track_ A_.flac`(`"` 和 `:` → `_`) | +| (无 TITLE) | `02 - Track 02.flac`(兜底) | + +**注意**:**文件名**里的非法字符被替换,但**元数据 tag**(写入文件内部)保留 CUE 里的原样。所以 `催眠 / 静夜.flac` 里的 `TITLE` tag 仍然是 `催眠 / 静夜`(含 `/`),播放器显示正常。 + +### 命名冲突处理 + +极少数情况下多首歌的净化后名字撞车(比如两首歌都叫 `未命名` 而且 num 不同但被截断)。脚本会在冲突时追加 `(2)`、`(3)`: + +- `05 - 未命名.flac` +- `06 - 未命名 (2).flac` + +### 元数据 (Tag) 映射 + +由 `cuetag` 自动写入。对 FLAC 输出,写的是 Vorbis Comment tags: + +| Tag | 来源 | +|---|---| +| `TITLE` | CUE 里 TRACK 块内的 `TITLE` | +| `ARTIST` / `PERFORMER` | CUE 里 TRACK 块内的 `PERFORMER`(若无则用专辑 PERFORMER) | +| `ALBUM` | CUE 里顶级 `TITLE` | +| `track` | CUE 里 TRACK 编号(补零至 2 位,如 `01`) | +| `TRACKTOTAL` | CUE 里总 TRACK 数 | +| `DATE` | CUE 里顶级 `REM DATE` | +| `GENRE` | CUE 里顶级 `REM GENRE` | + +**注意**:cuetag 的行为是**按位置对应**:`cuetag album.cue file1.flac file2.flac file3.flac` 会把 CUE 的 TRACK 1 写到 file1,TRACK 2 写到 file2,以此类推。脚本按 shnsplit_index 严格排序传参,保证对应正确。 + +### cuetag 失败的后果 + +如果 cuetag 因某种原因失败(罕见),脚本会**警告但不算专辑失败**——分轨音频文件本身是完整的,只是没有元数据 tag。可以事后手动跑一次: + +```bash +cd "专辑目录" +cuetag *.cue 01*.flac 02*.flac 03*.flac # 按顺序传 +``` + +--- + +## 10. 与 transcode_music.py 的组合工作流 + +### 典型完整流程 + +```bash +# ┌─ 前期整理 (split_cue.py) +python3 split_cue.py /mnt/z/新收藏 --analyze-only +python3 split_cue.py /mnt/z/新收藏 + +# └─ 转码到便携格式 (transcode_music.py) +python3 transcode_music.py /mnt/z/新收藏 /mnt/x/mp3输出 --analyze-only +python3 transcode_music.py /mnt/z/新收藏 /mnt/x/mp3输出 +``` + +### 两个脚本对同一个专辑的看法 + +举例:`王菲 - 唱游 [FLAC+CUE]/` 目录里有 `CDImage.flac` + `CDImage.cue`(12 轨)。 + +**只跑 transcode_music.py**: + +``` +[!] 1 个专辑需先手动 CUE split (已跳过转码): + • 王菲 - 唱游 [FLAC+CUE] + CDImage.cue: 整轨型 (单文件 'CDImage.flac', 12 轨) — 需先手动 CUE split 再转码 +``` + +**先跑 split_cue.py 再跑 transcode_music.py**: + +``` +[split_cue.py] +[1/1] 切割: 王菲 - 唱游 [FLAC+CUE] + 源: CDImage.flac → 12 轨 (flac) + ✓ [01/12] 01 - 红豆.flac + ✓ [02/12] 02 - 催眠.flac + ... + +[transcode_music.py] + CDImage.cue: 引用 'CDImage.flac' 但存在 12 个分轨 → 跳过 'CDImage.flac' 只转分轨 + 转码: 01 - 红豆.flac (flac, 44100Hz) + 转码: 02 - 催眠.flac (flac, 44100Hz) + ... +``` + +`transcode_music.py` 自动检测到 situation C,只转 12 个分轨、不动整轨,产生 12 个 MP3。 + +### Manifest 隔离 + +两个脚本的 manifest 文件名不同: + +- `split_cue.py` → `.split_cue_manifest.json` +- `transcode_music.py` → `.transcode_manifest.json` + +互不干扰。可以随时来回跑。 + +### 输出结构对比 + +**运行前**: + +``` +/mnt/z/新收藏/ +└── 王菲 - 唱游 [FLAC+CUE]/ + ├── CDImage.flac (整张 CD 一个大文件) + └── CDImage.cue +``` + +**运行 split_cue.py 后**: + +``` +/mnt/z/新收藏/ ← 源目录, 分轨写这里 +├── .split_cue_manifest.json ← split_cue 状态 +└── 王菲 - 唱游 [FLAC+CUE]/ + ├── CDImage.flac ← 原文件保留 + ├── CDImage.cue ← 保留 + ├── 01 - 红豆.flac ← 新: split_cue 输出 + ├── 02 - 催眠.flac + └── ... (12 个分轨) +``` + +**接着运行 transcode_music.py 后**: + +``` +/mnt/z/新收藏/ +├── .split_cue_manifest.json +├── .transcode_manifest.json ← transcode_music 状态 +└── 王菲 - 唱游 [FLAC+CUE]/ (未变) + +/mnt/x/mp3输出/ ← 目标目录 +└── 王菲 - 唱游 [FLAC+CUE]/ + ├── CDImage.cue ← CUE 也复制过去 + ├── 01 - 红豆.mp3 ← 320k CBR + ├── 02 - 催眠.mp3 + └── ... (12 个 MP3) +``` + +注意:`CDImage.flac` 不会转成 MP3(因为 situation C 检测跳过);`CDImage.cue` 会被复制到目标目录(作为资源)。 + +### 关于 CUE 的一个细节 + +切完后,`CDImage.cue` 里的 `FILE "CDImage.flac" WAVE` 仍指向原整轨。如果用户在目标目录里想用 CUE 播放分轨 MP3,这个 CUE 是不正确的——它指向整轨、但目标目录里没有那个整轨。 + +如果你实际会用 CUE 索引,考虑: + +1. 手动编辑复制到目标的 CUE,把 FILE 指向 MP3(但 CUE 传统上不支持多 FILE 分轨映射,改起来麻烦) +2. 直接删除目标目录里的 CUE(分轨 MP3 有 tag,播放器不需要 CUE 也能显示正确信息) + +**推荐做法**:直接删。分轨 MP3 的 ID3 tag 已经包含所有 CUE 元数据。 + +--- + +## 11. 错误处理与恢复 + +### 错误捕获原则 + +**单个专辑失败不影响其他**。shnsplit / cuetag / 重命名任一失败时: + +1. 该专辑内所有未完成的 track 状态设为 `failed`,错误消息写入 `error` 字段 +2. 该专辑整体状态设为 `failed` +3. `issues[]` 追加人类可读的说明 +4. **继续处理**下一个专辑 +5. 最后在报告里列出所有失败项 + +### 再次运行时 + +`failed` 状态的专辑会**自动重试**(整个专辑重切,因为 shnsplit 是全轨一次调用的原子操作,无法只重切失败的分轨)。 + +```bash +# 修好源文件后, 直接再跑 +python3 split_cue.py /mnt/z/新收藏 +``` + +### 常见错误 + +**"shnsplit 失败: unknown format"** + +CUE 里 `FILE "xxx" TYPE` 的 TYPE 值 shntool 不认。常见于 `FLAC` / `WAVE` / `MP3` / `BINARY`,其他值可能有问题。检查 CUE 内容并手动修正。 + +**"shnsplit 失败: cannot open file"** + +CUE 的 FILE 引用文件名和实际文件名不一致(尽管脚本会 fallback 匹配到实际文件,但 shnsplit 用的是 CUE 里的原字符串)。可能因为脚本 fallback 匹配到了但 shntool 找不到。手动把 CUE 里的 FILE 名改成实际文件名,再跑。 + +**"cuetag 失败, 音频仍可用但可能无标签"** + +cuetag 有时对某些 CUE 或 FLAC 版本行为异常。不影响切割本身,可事后手动跑 cuetag 或用别的 tagger(Picard、mp3tag 等)补 tag。 + +**"预期临时文件缺失/过小"** + +shnsplit 返回 0 但实际没写全所有分轨。极罕见,通常是磁盘满或权限问题。检查目标目录空间和权限。 + +**"临时目录创建失败"** + +无法在专辑目录里创建 `.split_cue_tmp_`。通常是目录只读或磁盘满。 + +**"源文件消失"** + +分析阶段之后、切割阶段之前源文件被删/移动。删除对应专辑 manifest 条目重新扫描即可。 + +### 中断(Ctrl+C) + +脚本捕获 SIGINT,保存 manifest 后退出。**中断时临时目录 `.split_cue_tmp_` 里的 shnsplit 中间产物会保留**(因为 finally 清理只在 split_album 函数正常返回时执行)——下次跑会重建。中断专辑的状态可能停留在 `processing`,重跑时会被视为需要重切。 + +如果发现有遗留的 `.split_cue_tmp_*` 目录(比如脚本被 kill -9 强杀),可以手动清理: + +```bash +find /mnt/z/新收藏 -maxdepth 2 -type d -name '.split_cue_tmp_*' -exec rm -rf {} + +``` + +### 查看失败列表 + +```bash +# 用 jq 快速查看失败的专辑 +jq -r '.albums | to_entries[] | select(.value.status=="failed") | .key' \ + /mnt/z/新收藏/.split_cue_manifest.json + +# 查看某专辑的具体失败原因 +jq '.albums["专辑名"] | {status, issues, tracks: [.tracks[] | select(.status=="failed") | {num, error}]}' \ + /mnt/z/新收藏/.split_cue_manifest.json +``` + +### 关于 unsupported + +`unsupported` 状态的专辑不算失败,是主动跳过。原因写在 `issues[]` 里,例如: + +- `缺少依赖: mac (APE 解码器)` → 装 `mac` 后再跑,会自动切 +- `shntool 不支持 DSD, 请先手动转 FLAC` → 手动预处理 + +装好依赖后,需要用 `--force` 或删掉对应 manifest 条目才会重新分析。 + +--- + +## 12. 已知局限 + +- **不支持 DSD**(`.dsf` / `.dff`):shntool 不识别 DSD。若有整轨 DSD + CUE,需先手动 ffmpeg 转 FLAC 再切。 +- **shnsplit 是一次调用切完整个专辑**:无法只重切失败的某一首歌。失败时整个专辑重切(幂等,无副作用)。 +- **不做 replaygain / 章节标记 / 高级 tag**:只做 cuetag 支持的标准 tag(TITLE/ARTIST/ALBUM/DATE/GENRE/track/TRACKTOTAL)。 +- **不支持并行切割**:为保证 manifest 状态一致性,专辑串行处理。可对不同子目录同时跑多个脚本实例。 +- **CUE 里 track 编号必须连续从 1 开始**:极少数非标准 CUE(比如从 TRACK 03 开始)会导致 shnsplit 内部计数器和 CUE 编号错位。目前未特殊处理,可能造成 tag 错乱。 +- **网盘 mtime 不能保留**:分轨是新建文件,mtime 是当下时间。原文件不动。 +- **--force 不会自动清理旧分轨**:如果目录里已有旧分轨且脚本判为 situation C 而 skip,`--force` 也不会强制切。需手动删旧分轨。 +- **shnsplit 依赖 `flac` 二进制来编码 FLAC**:这是 shntool 3.x 的架构选择(它调用外部 encoder)。装了 `flac` 才能切 FLAC 源;装了 `mac` 才能切 APE 源。 +- **APE 输出为 FLAC,不能原格式**:ffmpeg 和 shntool 都没 APE 编码器(APE 是私有格式)。所以 APE → FLAC 是唯一路径。 +- **Windows 直接跑未测试**:脚本用了 POSIX `os.fsync(dir_fd)`,Windows 上会 best-effort。理论可用,实际推荐 Linux / WSL2 / macOS。 + +--- + +## 13. FAQ + +**Q: 为什么要单独一个 split_cue.py?直接让 transcode_music.py 顺便切了不好吗?** + +A: 分离关注点。切割是**修改源目录**的操作(写新文件),转码是**读源目录写目标目录**的操作。两者错误影响面不同:切错了改源,转错了改目标。分开可以: +1. 先充分 review 切割计划再动手(`--analyze-only`) +2. 切割失败不牵连转码,反之亦然 +3. 用户可以只切不转(本地播放器直接消费分轨 FLAC) +4. 用户可以直接手动切好再让 transcode_music 转(跳过 split_cue) + +**Q: 切完 CUE 文件删不删?** + +A: 保留。原文件(`CDImage.flac`)也保留。用户想删的话手动来,脚本不做破坏性操作。删原文件前建议: +1. 先播放几首分轨确认没问题 +2. 用 `metaflac -l 01*.flac` 确认 tag 完整 +3. 备份 CUE(换新 CUE 想重切时有用) + +**Q: 切割精度和音质怎么样?** + +A: shntool 是按 CD 帧(1/75 秒)对齐切的,与原始 CUE INDEX 完全一致。对 FLAC → FLAC,音频数据无损(decode + re-encode 是数学等价的 PCM 转换)。对 WAV → WAV 是纯样本操作,字节级精确。对 APE → FLAC,音频数据也无损(FLAC 压缩是无损的),只是容器格式变了。 + +**Q: 分轨没有 gapless 播放会不会有咔嗒声?** + +A: 用了 shntool 的默认 `--append-gaps` 模式,pregap 追加到上一轨结尾。相邻曲目之间的过渡应该 sample-exact 无缝,播放器如支持 gapless 就无咔嗒声。 + +**Q: 切完后音频总长度对不上原文件?** + +A: 应该是精确对上的(每首歌的 duration 加起来 = 原文件 duration)。如果偏差 < 1 秒,可能是取整误差(ffprobe 显示的 duration 有精度限制)。偏差大就是有 bug,请提 issue。 + +**Q: 中文 / 日文 / 韩文的 CUE 处理有问题吗?** + +A: 应该没问题。脚本会按 `utf-8 / gbk / big5 / shift_jis / cp936 / latin1` 顺序尝试解码,绝大部分 CJK CUE 都能识别。切割输出的 tag(Vorbis Comment)总是 UTF-8。如果发现某个 CUE 解码错乱,可以用别的工具(比如 `iconv` 或文本编辑器)先把 CUE 转成 UTF-8 再跑脚本。 + +**Q: `cuetag` 输出 warning 说什么 metaflac 不 exist?** + +A: 有些 cuetools 版本的 cuetag 是 Python 或 shell 脚本,内部调用 `metaflac`(来自 `flac` 包)。装了 `flac` 就有 `metaflac`。如果 `cuetag` 抱怨 metaflac 不存在,装 `flac` 包即可(`sudo apt install flac`)。 + +**Q: 我的 CUE 有奇怪的编码/结构 shntool 处理不了怎么办?** + +A: 先用文本编辑器把 CUE 转成 UTF-8 无 BOM,去掉不必要的 REM 行,看 TRACK / FILE / INDEX 结构是否规范(每个 TRACK 至少要有 `INDEX 01 MM:SS:FF`)。规范化后重试。 + +**Q: 能不能自定义分轨文件名格式?比如 "%p - %t.flac"?** + +A: 目前不能。命名格式固定为 `NN - Title.ext`。这是刻意的: +1. 保证目录内文件按 track 顺序排列 +2. 简化文件名,避免复杂替换的 bug +3. 元数据 tag 里已经有完整信息(艺人、专辑等),不需要塞进文件名 + +**Q: 支持 track 里带 gap(HTOA / hidden track)吗?** + +A: 部分支持。shnsplit 默认 `--append-gaps` 会把 gap 归到前一轨末尾。TRACK 01 之前的 pregap(HTOA)目前算作 TRACK 01 的开头。如果你需要独立提取 HTOA,得手动调 shnsplit 参数(脚本目前不暴露此选项)。 + +**Q: 切错了想撤回?** + +A: 因为**原文件从不被删改**,撤回很简单:找出符合 `NN - Title.ext` 格式的新生成分轨全部删除即可。示例见场景 8。 + +**Q: 支持嵌套目录(比如多碟 CD1/CD2/)吗?** + +A: 支持。脚本会 `os.walk` 全递归,每一层子目录独立分析。多碟专辑的每个 CDx 目录都会被视为一个独立的专辑单元。 + +**Q: 我可以在 Docker 里跑吗?** + +A: 可以。写个简单 Dockerfile: + +```dockerfile +FROM debian:stable-slim +RUN apt-get update && apt-get install -y \ + python3 shntool cuetools flac ffmpeg \ + && rm -rf /var/lib/apt/lists/* +COPY split_cue.py /usr/local/bin/ +ENTRYPOINT ["python3", "/usr/local/bin/split_cue.py"] +``` + +然后: + +```bash +docker run --rm -v /mnt/z/music:/music my-split-cue /music --analyze-only +``` + +**Q: 我想只切某一个格式(比如只 FLAC 不 APE)怎么办?** + +A: 目前没提供 filter 参数。可以用 shell 循环: + +```bash +find /mnt/z/新收藏 -type d | while read d; do + if ls "$d"/*.flac >/dev/null 2>&1; then + python3 split_cue.py "$d" + fi +done +``` + +**Q: 生成的 manifest 会很大吗?** + +A: 一般不大。50 个专辑的 manifest 约 30 KB。上千个专辑估计几 MB。不影响使用。 + +**Q: transcode_music.py 和 split_cue.py 能同时跑吗?** + +A: 不建议。两者都会写源目录(split_cue 写分轨、transcode_music 写自己的 manifest)。虽然写的是不同文件不会冲突,但 transcode_music 分析时看到 split_cue 正在切的中间状态可能判断错误。**串行跑更安全**:split_cue 先跑完,transcode_music 再跑。 + +--- + +## 附录 A:完整命令示例集 + +```bash +# 分析 (不切割) +python3 split_cue.py /mnt/z/music --analyze-only + +# 正式切割 (自动 resume) +python3 split_cue.py /mnt/z/music + +# 详细日志 +python3 split_cue.py /mnt/z/music -v + +# 强制重跑 (覆盖已有分轨) +python3 split_cue.py /mnt/z/music --force + +# 使用长参数 +python3 split_cue.py --source /mnt/z/music + +# 使用短参数 +python3 split_cue.py -s /mnt/z/music + +# 组合: 位置 + 选项 +python3 split_cue.py /mnt/z/music -v --force + +# 指定 shnsplit / cuetag 路径 +python3 split_cue.py /mnt/z/music \ + --shnsplit /opt/shntool-3/bin/shnsplit \ + --cuetag /opt/cuetools/bin/cuetag + +# 加大超时 (超长专辑) +python3 split_cue.py /mnt/z/music --split-timeout 3600 + +# 只处理某一个专辑 +python3 split_cue.py "/mnt/z/music/王菲 - 唱游 [FLAC+CUE]" + +# 完整工作流: split 然后 transcode +python3 split_cue.py /mnt/z/music && \ +python3 transcode_music.py /mnt/z/music /mnt/x/mp3 +``` + +## 附录 B:文件结构示例 + +**运行前**(典型的整轨型音乐盘): + +``` +/mnt/z/新收藏/ +├── 王菲 - 唱游 [FLAC+CUE]/ +│ ├── CDImage.flac ← 整张 CD 一个大文件, 40 分钟 +│ └── CDImage.cue ← 12 个 TRACK +├── 蔡琴 - 老歌 [FLAC+CUE]/ +│ ├── album.flac ← 整张 CD 一个大文件 +│ ├── album.cue +│ ├── cover.jpg +│ └── booklet.pdf +├── 张学友精选 [已分轨]/ ← 已经是分轨的, 会 SKIP +│ ├── 01 吻别.flac +│ ├── 02 一路上有你.flac +│ ├── 03 情网.flac +│ └── (无 CUE) +└── Beatles - White Album [WAV+CUE]/ ← WAV 源, 会切成 WAV + ├── disc1.wav + └── disc1.cue +``` + +**运行 `split_cue.py /mnt/z/新收藏` 后**: + +``` +/mnt/z/新收藏/ +├── .split_cue_manifest.json ← 状态记录 +├── 王菲 - 唱游 [FLAC+CUE]/ +│ ├── CDImage.flac ← 保留不动 +│ ├── CDImage.cue ← 保留 +│ ├── 01 - 红豆.flac ← 新: 分轨输出 (含 tag) +│ ├── 02 - 催眠.flac +│ ├── 03 - 无常.flac +│ ├── 04 - 童 (国语).flac +│ ├── ... (共 12 个) +│ └── 12 - 精彩.flac +├── 蔡琴 - 老歌 [FLAC+CUE]/ +│ ├── album.flac ← 保留 +│ ├── album.cue ← 保留 +│ ├── cover.jpg ← 保留 +│ ├── booklet.pdf ← 保留 +│ ├── 01 - 恰似你的温柔.flac ← 新: 分轨 +│ ├── 02 - 你的眼神.flac +│ └── ... (12 个) +├── 张学友精选 [已分轨]/ ← 无变化 (SKIP) +│ ├── 01 吻别.flac +│ ├── 02 一路上有你.flac +│ └── 03 情网.flac +└── Beatles - White Album [WAV+CUE]/ + ├── disc1.wav ← 保留 WAV 整轨 + ├── disc1.cue ← 保留 + ├── 01 - Back In The U.S.S.R..wav ← 新: WAV 分轨 + ├── 02 - Dear Prudence.wav + └── ... (30 个) +``` + +**接着运行 `transcode_music.py /mnt/z/新收藏 /mnt/x/mp3`**: + +``` +/mnt/z/新收藏/ ← 未变 +├── .split_cue_manifest.json +├── .transcode_manifest.json ← 新: transcode 状态 +└── ... (专辑目录未变) + +/mnt/x/mp3/ ← 目标目录 +├── 王菲 - 唱游 [FLAC+CUE]/ +│ ├── CDImage.cue ← CUE 复制过来 +│ ├── 01 - 红豆.mp3 ← 320k CBR +│ ├── 02 - 催眠.mp3 +│ └── ... (12 个 MP3) +├── 蔡琴 - 老歌 [FLAC+CUE]/ +│ ├── album.cue +│ ├── cover.jpg +│ ├── booklet.pdf +│ ├── 01 - 恰似你的温柔.mp3 +│ └── ... +├── 张学友精选 [已分轨]/ +│ ├── 01 吻别.mp3 +│ ├── 02 一路上有你.mp3 +│ └── 03 情网.mp3 +└── Beatles - White Album [WAV+CUE]/ + ├── disc1.cue + ├── 01 - Back In The U.S.S.R..mp3 + └── ... (30 个 MP3) +``` + +**注意**: +- 整轨大文件(`CDImage.flac`、`album.flac`、`disc1.wav`)**不会**被转成 MP3,因为 transcode_music.py 检测到 situation C 会跳过它们 +- 所有 CUE 文件会被复制到目标目录(作为资源文件) +- 分轨 MP3 保留完整 ID3 tag(继承自分轨 FLAC 的 Vorbis Comment) + +**Manifest 位置**: + +``` +/mnt/z/新收藏/.split_cue_manifest.json ← split_cue 的状态 +/mnt/z/新收藏/.transcode_manifest.json ← transcode_music 的状态 +``` + +两个 manifest 都在源目录里,文件名不同,互不冲突。 diff --git a/transcode_music/split_cue.py b/transcode_music/split_cue.py new file mode 100644 index 0000000..b093665 --- /dev/null +++ b/transcode_music/split_cue.py @@ -0,0 +1,1093 @@ +#!/usr/bin/env python3 +""" +split_cue.py - 批量用 CUE 文件切割整轨无损音乐为分轨. + +Companion to transcode_music.py. Walks source dir, detects "needs_cue_split" +albums (single CUE + single whole-disc lossless file), splits them into +per-track files IN PLACE using shntool (shnsplit) + cuetag. Preserves source +format when possible: + FLAC → FLAC (via shnsplit -o flac, needs `flac` binary) + WAV → WAV (via shnsplit -o wav, native) + APE → FLAC (needs `mac` decoder; UNSUPPORTED if missing) + DSF/DFF → UNSUPPORTED (shntool doesn't handle DSD) + +Whole-disc file is NEVER modified. transcode_music.py's CUE analyzer will +subsequently see it as "situation C" (CUE + whole-disc + per-track files) +and skip the whole-disc file, transcoding only the per-track outputs. + +Split pipeline per album: + 1. shnsplit -f cue -o -t "%n" -O always -d + produces /01., 02., ... + 2. cuetag /01. ... applies CUE tags positionally + 3. os.replace each tmp file → dir/. (sanitized name) + +Dependencies: Python 3.10+, shntool, cuetools (cuetag), ffprobe (for source +metadata display only). `flac` for FLAC output. `mac` for APE input. + +Usage: + # Analyze only (recommended first run — inspect split plan before splitting) + python3 split_cue.py /mnt/z/music --analyze-only + + # Split all detected whole-disc albums in-place (auto-resumes) + python3 split_cue.py /mnt/z/music + + # Force re-analyze + overwrite existing per-track files + python3 split_cue.py /mnt/z/music --force +""" +from __future__ import annotations + +import argparse +import datetime as dt +import errno +import json +import logging +import os +import re +import shutil +import subprocess +import sys +import threading +from dataclasses import dataclass +from enum import Enum +from pathlib import Path +from typing import Optional + +__version__ = "1.0.0" + +MANIFEST_SCHEMA_VERSION = 1 +MANIFEST_NAME = ".split_cue_manifest.json" + +LOSSLESS_EXTS = {".flac", ".wav", ".dsf", ".dff", ".ape"} +CUE_EXTS = {".cue"} + +IGNORE_NAMES = {".ds_store", "thumbs.db", "desktop.ini", ".directory"} +IGNORE_PREFIXES = ("._",) # macOS resource forks — not user content + +# CUE files from Chinese/Japanese/Korean releases often use legacy encodings. +CUE_ENCODINGS = ("utf-8-sig", "utf-8", "gbk", "big5", "shift_jis", "cp936", "latin1") + +# Filesystem-illegal characters (Windows-superset, safe on Linux/macOS too). +_ILLEGAL_CHARS_RE = re.compile(r'[<>:"/\\|?*\x00-\x1f]') + + +class Status(str, Enum): + PENDING = "pending" + ANALYZED = "analyzed" + PROCESSING = "processing" + COMPLETED = "completed" + FAILED = "failed" + SKIPPED = "skipped" + UNSUPPORTED = "unsupported" + + +@dataclass +class Config: + source_root: Path + force: bool = False + analyze_only: bool = False + shnsplit: str = "shnsplit" + cuetag: str = "cuetag" + ffprobe: str = "ffprobe" + split_timeout_sec: int = 1800 + tag_timeout_sec: int = 60 + + +# ────────────────────────── helpers ────────────────────────── + +def _now_iso() -> str: + return dt.datetime.now(dt.timezone.utc).astimezone().isoformat(timespec="seconds") + + +def _to_int(x) -> Optional[int]: + if x in (None, "", "N/A"): + return None + try: + return int(x) + except (TypeError, ValueError): + return None + + +def _to_float(x) -> Optional[float]: + if x in (None, "", "N/A"): + return None + try: + return float(x) + except (TypeError, ValueError): + return None + + +def _try_unlink(p: Path) -> None: + try: + if p.exists(): + p.unlink() + except OSError: + pass + + +def _remove_dir(path: Path) -> None: + try: + shutil.rmtree(str(path)) + except OSError: + pass + + +def _is_ignored(p: Path) -> bool: + name = p.name.lower() + if name in IGNORE_NAMES: + return True + for pref in IGNORE_PREFIXES: + if name.startswith(pref): + return True + if MANIFEST_NAME.lower() in name: + return True + return False + + +def atomic_write_json(path: Path, data: dict) -> None: + payload = json.dumps(data, ensure_ascii=False, indent=2) + tmp = path.with_name(f".{path.name}.{os.getpid()}.{threading.get_ident()}.tmp") + + fd = os.open(str(tmp), os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644) + try: + os.write(fd, payload.encode("utf-8")) + try: + os.fsync(fd) + except OSError: + pass # network mounts (CIFS/NFS) may not implement fsync + finally: + os.close(fd) + + os.replace(str(tmp), str(path)) + + try: + dfd = os.open(str(path.parent), os.O_RDONLY) + try: + os.fsync(dfd) # persist the rename itself (durability guarantee) + except OSError: + pass + finally: + os.close(dfd) + except OSError: + pass + + +def quarantine_corrupt(path: Path, reason: str) -> Path: + ts = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%S") + q = path.with_name(f"{path.name}.corrupt-{ts}") + try: + os.replace(str(path), str(q)) + logging.error(f"Manifest 损坏, 已隔离到 {q}: {reason}") + except OSError as e: + logging.error(f"隔离失败 {path}: {e}") + return q + + +def sanitize_filename(name: str, max_len: int = 200) -> str: + s = _ILLEGAL_CHARS_RE.sub("_", name) + s = re.sub(r"\s+", " ", s).strip() + s = s.rstrip(". ") # Windows dislikes trailing dots/spaces + if len(s) > max_len: + s = s[:max_len].rstrip() + return s or "untitled" + + +# ────────────────────────── ffprobe ────────────────────────── + +def ffprobe_audio(path: Path, cfg: Config) -> dict: + try: + result = subprocess.run( + [ + cfg.ffprobe, "-v", "error", + "-select_streams", "a:0", + "-show_entries", "stream=codec_name,sample_rate,channels,bits_per_raw_sample", + "-show_entries", "format=duration", + "-of", "json", + str(path), + ], + capture_output=True, text=True, errors="replace", timeout=60, + ) + if result.returncode != 0: + return {"error": (result.stderr or "ffprobe failed").strip()[-200:]} + data = json.loads(result.stdout or "{}") + stream = (data.get("streams") or [{}])[0] + fmt = data.get("format") or {} + return { + "codec": stream.get("codec_name"), + "sample_rate": _to_int(stream.get("sample_rate")), + "channels": _to_int(stream.get("channels")), + "duration_sec": _to_float(fmt.get("duration")), + "bits_per_sample": _to_int(stream.get("bits_per_raw_sample")), + } + except subprocess.TimeoutExpired: + return {"error": "ffprobe timeout"} + except (json.JSONDecodeError, OSError, UnicodeDecodeError) as e: + return {"error": str(e)} + + +# ────────────────────────── CUE parser ────────────────────────── + +def read_cue_text(path: Path) -> Optional[str]: + try: + raw = path.read_bytes() + except OSError: + return None + for enc in CUE_ENCODINGS: + try: + return raw.decode(enc) + except UnicodeDecodeError: + continue + return None + + +def _unquote(s: str) -> str: + s = s.strip() + if len(s) >= 2 and s[0] == '"' and s[-1] == '"': + return s[1:-1] + return s + + +def parse_cue_full(path: Path) -> Optional[dict]: + """ + Parse a CUE file into structured form. + + CUE spec semantics: + TITLE / PERFORMER before any TRACK = album-level + TITLE / PERFORMER inside a TRACK block = track-level + REM DATE / REM GENRE at album level = album-level metadata + """ + text = read_cue_text(path) + if text is None: + return None + + album_title: Optional[str] = None + album_performer: Optional[str] = None + album_date: Optional[str] = None + album_genre: Optional[str] = None + files: list[dict] = [] + tracks: list[dict] = [] + current_track: Optional[dict] = None + + for raw_line in text.splitlines(): + line = raw_line.strip() + if not line: + continue + parts = line.split(None, 1) + if not parts: + continue + keyword = parts[0].upper() + rest = parts[1] if len(parts) > 1 else "" + + if keyword == "FILE": + m = re.match(r'"([^"]*)"\s+(\S+)', rest) or re.match(r"(\S+)\s+(\S+)", rest) + if m: + files.append({"name": m.group(1), "format": m.group(2).upper()}) + elif keyword == "TRACK": + m = re.match(r"(\d+)\s+(\S+)", rest) + if m: + current_track = { + "num": int(m.group(1)), + "type": m.group(2).upper(), + "title": None, + "performer": None, + } + tracks.append(current_track) + elif keyword == "TITLE": + val = _unquote(rest) + if current_track is None: + if album_title is None: + album_title = val + else: + current_track["title"] = val + elif keyword == "PERFORMER": + val = _unquote(rest) + if current_track is None: + if album_performer is None: + album_performer = val + else: + current_track["performer"] = val + elif keyword == "REM": + rem_parts = rest.split(None, 1) + if len(rem_parts) >= 2 and current_track is None: + rem_kw = rem_parts[0].upper() + rem_val = _unquote(rem_parts[1]) + if rem_kw == "DATE" and album_date is None: + album_date = rem_val + elif rem_kw == "GENRE" and album_genre is None: + album_genre = rem_val + + return { + "album_title": album_title, + "album_performer": album_performer, + "album_date": album_date, + "album_genre": album_genre, + "files": files, + "tracks": tracks, + } + + +def match_cue_ref(ref_name: str, audio_paths: list[Path]) -> Optional[Path]: + """ + Match a CUE FILE reference to an actual audio file, with fallbacks for + case differences and extension mismatch (e.g., FILE "album.wav" WAVE + where the actual file is album.flac). + """ + for p in audio_paths: + if p.name == ref_name: + return p + ref_lower = ref_name.lower() + for p in audio_paths: + if p.name.lower() == ref_lower: + return p + ref_stem = Path(ref_name).stem.lower() + matches = [p for p in audio_paths if p.stem.lower() == ref_stem] + if len(matches) == 1: + return matches[0] + return None + + +# ────────────────────────── dependency probing ────────────────────────── + +@dataclass +class Deps: + shnsplit: bool + cuetag: bool + ffprobe: bool + flac: bool + mac: bool + + def supported_source_exts(self) -> set[str]: + supported = {".wav"} # shntool native + if self.flac: + supported.add(".flac") + if self.mac and self.flac: + supported.add(".ape") + return supported + + def unsupported_reason(self, ext: str) -> str: + if ext == ".ape": + missing = [] + if not self.mac: + missing.append("mac (APE 解码器)") + if not self.flac: + missing.append("flac") + return f"缺少依赖: {', '.join(missing)}" + if ext in (".dsf", ".dff"): + return "shntool 不支持 DSD, 请先手动转 FLAC" + if ext == ".flac" and not self.flac: + return "缺少依赖: flac" + return f"未知源格式: {ext}" + + +def probe_deps(cfg: Config) -> Deps: + return Deps( + shnsplit=shutil.which(cfg.shnsplit) is not None, + cuetag=shutil.which(cfg.cuetag) is not None, + ffprobe=shutil.which(cfg.ffprobe) is not None, + flac=shutil.which("flac") is not None, + mac=shutil.which("mac") is not None, + ) + + +def _output_format_for(src_ext: str) -> tuple[Optional[str], Optional[str]]: + if src_ext == ".flac": + return "flac", ".flac" + if src_ext == ".wav": + return "wav", ".wav" + if src_ext == ".ape": + return "flac", ".flac" # ape → flac (mac decodes, flac encodes) + return None, None + + +# ────────────────────────── analysis ────────────────────────── + +def analyze_directory(dir_path: Path, cfg: Config, deps: Deps) -> dict: + """ + Analyze one directory. Detects the "whole-disc CUE + big file" pattern + and builds a per-track split plan. Non-matching dirs get SKIPPED. + Sources whose decoder is missing get UNSUPPORTED with reason. + """ + entry: dict = { + "status": Status.PENDING.value, + "source_audio": None, + "cue": None, + "album_title": None, + "album_performer": None, + "album_date": None, + "album_genre": None, + "source_codec": None, + "source_sample_rate": None, + "source_channels": None, + "duration_sec": None, + "output_format": None, + "tracks": [], + "issues": [], + "started_at": None, + "completed_at": None, + } + + try: + files = sorted(p for p in dir_path.iterdir() if p.is_file()) + except OSError as e: + entry["status"] = Status.FAILED.value + entry["issues"].append(f"无法读取目录: {e}") + return entry + + audio_files = [p for p in files + if p.suffix.lower() in LOSSLESS_EXTS and not _is_ignored(p)] + cue_files = [p for p in files + if p.suffix.lower() in CUE_EXTS and not _is_ignored(p)] + + if not cue_files or not audio_files: + entry["status"] = Status.SKIPPED.value + return entry + + supported = deps.supported_source_exts() + + for cue in cue_files: + parsed = parse_cue_full(cue) + if parsed is None: + entry["issues"].append(f"{cue.name}: 无法解析 (编码问题)") + continue + + refs = parsed["files"] + if len(refs) == 0: + entry["issues"].append(f"{cue.name}: 无 FILE entry") + continue + if len(refs) > 1: + continue # multi-FILE CUE = already per-track index + + ref_name = refs[0]["name"] + matched = match_cue_ref(ref_name, audio_files) + if matched is None: + entry["issues"].append(f"{cue.name}: 引用 '{ref_name}' 但目录内不存在") + continue + + others = [p for p in audio_files if p != matched] + if others: + continue # situation C — already split + + # ── Whole-disc pattern confirmed ── + cue_tracks = parsed["tracks"] + if not cue_tracks: + entry["issues"].append(f"{cue.name}: 无 TRACK 条目") + continue + + src_ext = matched.suffix.lower() + entry["cue"] = cue.name + entry["source_audio"] = matched.name + entry["album_title"] = parsed["album_title"] + entry["album_performer"] = parsed["album_performer"] + entry["album_date"] = parsed["album_date"] + entry["album_genre"] = parsed["album_genre"] + + if src_ext not in supported: + entry["status"] = Status.UNSUPPORTED.value + entry["issues"].append(deps.unsupported_reason(src_ext)) + return entry + + out_format, out_ext = _output_format_for(src_ext) + entry["output_format"] = out_format + + probe = ffprobe_audio(matched, cfg) + if "error" in probe: + entry["issues"].append(f"ffprobe '{matched.name}': {probe['error']}") + else: + entry["source_codec"] = probe.get("codec") + entry["source_sample_rate"] = probe.get("sample_rate") + entry["source_channels"] = probe.get("channels") + entry["duration_sec"] = probe.get("duration_sec") + + num_digits = max(2, len(str(len(cue_tracks)))) + total_tracks = len(cue_tracks) + target_names: set[str] = set() + + for i, cue_track in enumerate(cue_tracks, start=1): + title = cue_track.get("title") or f"Track {cue_track['num']:02d}" + num_str = str(cue_track["num"]).zfill(num_digits) + base = sanitize_filename(f"{num_str} - {title}") + target = base + out_ext + n = 2 + while target in target_names: + target = f"{base} ({n}){out_ext}" + n += 1 + target_names.add(target) + entry["tracks"].append({ + "num": cue_track["num"], + "shnsplit_index": i, # 1-based, matches shnsplit's -t "%n" counter + "title": title, + "performer": cue_track.get("performer"), + "target_name": target, + "total_tracks": total_tracks, + "status": Status.PENDING.value, + "error": None, + "converted_at": None, + }) + + entry["status"] = Status.ANALYZED.value + return entry + + entry["status"] = Status.SKIPPED.value + return entry + + +def merge_entry(existing: dict, fresh: dict) -> dict: + """ + Preserve completion state across re-analysis: if a previously-completed + track has the same target_name, mark it completed in the fresh entry. + """ + old_by_name = {t.get("target_name"): t for t in existing.get("tracks") or []} + for t in fresh.get("tracks", []): + old = old_by_name.get(t["target_name"]) + if old and old.get("status") == Status.COMPLETED.value: + t["status"] = Status.COMPLETED.value + t["converted_at"] = old.get("converted_at") + t["error"] = None + if existing.get("started_at"): + fresh["started_at"] = existing["started_at"] + return fresh + + +# ────────────────────────── album splitter (shntool + cuetag) ────────────────────────── + +def _mark_all_failed(album_entry: dict, err: str) -> None: + for t in album_entry["tracks"]: + if t["status"] != Status.COMPLETED.value: + t["status"] = Status.FAILED.value + t["error"] = err + + +# INDEX / PREGAP / POSTGAP with 1-digit fractional (e.g. "57:12.0" or "57:12:0") +# — non-standard, both shnsplit AND cuetag's internal cueprint reject them. +# Normalize to canonical CUE "MM:SS:FF" (2-digit CD frame count, 0-74) by +# interpreting the 1 digit as raw frame count and left-padding to 2 digits. +# For the overwhelmingly common ".0" case this is exact (0 frames); other +# values may drift by up to 9 frames (~120ms) which is unavoidable given +# the original CUE's ambiguity. +_BAD_CUE_TIME_RE = re.compile( + r"^([ \t]*(?:INDEX\s+\d+|PREGAP|POSTGAP)\s+)(\d+):(\d+)[.:](\d)[ \t]*$", + re.IGNORECASE | re.MULTILINE, +) + + +def normalize_cue_for_shntool(cue_text: str) -> str: + cue_text = cue_text.replace("\r\n", "\n").replace("\r", "\n") + normalized = _BAD_CUE_TIME_RE.sub( + lambda m: f"{m.group(1)}{m.group(2)}:{m.group(3)}:0{m.group(4)}", + cue_text, + ) + # cueprint (used internally by cuetag) rejects files without a final \n. + if not normalized.endswith("\n"): + normalized += "\n" + return normalized + + +def split_album( + dir_path: Path, + album_entry: dict, + cfg: Config, + log: logging.Logger, +) -> bool: + """ + Split all pending tracks via shnsplit + cuetag + rename. + + Returns True on full success (all tracks COMPLETED). + + Design: shnsplit produces `/01.`, `02.`, ... (zero-padded + per `-n` format). cuetag writes CUE metadata into each file positionally. + We then os.replace() each temp file to its sanitized final name in dir_path. + """ + src = dir_path / album_entry["source_audio"] + cue = dir_path / album_entry["cue"] + tracks = album_entry["tracks"] + total = len(tracks) + + if not src.exists(): + log.error(f" 源文件消失: {src}") + album_entry["issues"].append(f"处理时源文件不存在: {src.name}") + _mark_all_failed(album_entry, "源文件不存在") + return False + if not cue.exists(): + log.error(f" CUE 消失: {cue}") + album_entry["issues"].append(f"处理时 CUE 不存在: {cue.name}") + _mark_all_failed(album_entry, "CUE 不存在") + return False + + out_format, out_ext = _output_format_for(src.suffix.lower()) + if out_format is None: + _mark_all_failed(album_entry, f"源格式 {src.suffix} 不支持") + return False + + # Skip fast-path: all tracks already completed and files exist on disk. + if not cfg.force: + all_done = all( + t["status"] == Status.COMPLETED.value and (dir_path / t["target_name"]).exists() + for t in tracks + ) + if all_done: + log.debug(f" 所有轨已完成且文件存在, skip") + return True + + tmpdir = dir_path / f".split_cue_tmp_{os.getpid()}" + if tmpdir.exists(): + _remove_dir(tmpdir) + try: + tmpdir.mkdir() + except OSError as e: + log.error(f" 无法创建临时目录 {tmpdir}: {e}") + _mark_all_failed(album_entry, f"临时目录创建失败: {e}") + return False + + try: + num_digits = max(2, len(str(total))) + n_format = f"%0{num_digits}d" + + # ── Step 0: prepare a UTF-8 + time-normalized CUE for both shnsplit and cuetag ── + # Feeding original CUE to shnsplit trips on non-standard MM:SS.n times. + # Feeding original CUE to cuetag corrupts CJK tags because Vorbis Comment + # requires UTF-8. One prepared CUE solves both. + prepared_cue: Optional[Path] = None + cue_text = read_cue_text(cue) + if cue_text is None: + log.error(f" CUE 无法解码: {cue.name}") + _mark_all_failed(album_entry, "CUE 无法解码") + return False + cue_text = normalize_cue_for_shntool(cue_text) + prepared_cue = tmpdir / "_prepared.cue" + try: + prepared_cue.write_text(cue_text, encoding="utf-8") + except OSError as e: + log.error(f" 临时 CUE 写入失败: {e}") + _mark_all_failed(album_entry, f"临时 CUE 写入失败: {e}") + return False + + # ── Step 1: shnsplit ── + cmd = [ + cfg.shnsplit, + "-f", str(prepared_cue), + "-o", out_format, + "-t", "%n", + "-n", n_format, + "-O", "always", + "-d", str(tmpdir), + str(src), + ] + log.info(f" shnsplit -o {out_format} → {total} 轨") + try: + result = subprocess.run( + cmd, capture_output=True, text=True, errors="replace", + timeout=cfg.split_timeout_sec, + ) + except subprocess.TimeoutExpired: + log.error(f" ✗ shnsplit 超时") + _mark_all_failed(album_entry, f"shnsplit 超时 ({cfg.split_timeout_sec}s)") + return False + except (OSError, UnicodeDecodeError) as e: + log.error(f" ✗ shnsplit 调用失败: {e}") + _mark_all_failed(album_entry, f"shnsplit 调用失败: {e}") + return False + + if result.returncode != 0: + stderr = (result.stderr or "").strip() + err_msg = stderr.splitlines()[-1] if stderr else f"shnsplit exit {result.returncode}" + log.error(f" ✗ shnsplit 失败: {err_msg}") + _mark_all_failed(album_entry, err_msg[-400:]) + return False + + # ── Step 2: verify expected temp files ── + tmp_files: list[Path] = [] + for i in range(1, total + 1): + tmp_name = f"{i:0{num_digits}d}{out_ext}" + f = tmpdir / tmp_name + try: + sz = f.stat().st_size + except OSError: + sz = 0 + if sz < 1024: + err = f"shnsplit 未产生预期文件 {tmp_name} (大小 {sz})" + log.error(f" ✗ {err}") + _mark_all_failed(album_entry, err) + return False + tmp_files.append(f) + + # ── Step 3: cuetag (positional metadata write) ── + tag_cmd = [cfg.cuetag, str(prepared_cue)] + [str(f) for f in tmp_files] + try: + tag_result = subprocess.run( + tag_cmd, capture_output=True, text=True, errors="replace", + timeout=cfg.tag_timeout_sec, + ) + if tag_result.returncode != 0: + stderr = (tag_result.stderr or "").strip() + brief = stderr.splitlines()[-1] if stderr else f"exit {tag_result.returncode}" + log.warning(f" cuetag 失败, 音频仍可用但可能无标签: {brief}") + album_entry["issues"].append(f"cuetag 失败: {brief[:200]}") + except (subprocess.TimeoutExpired, OSError, UnicodeDecodeError) as e: + log.warning(f" cuetag 调用失败, 音频仍可用: {e}") + album_entry["issues"].append(f"cuetag 调用失败: {e}") + + # ── Step 4: rename each temp file to sanitized final name in dir_path ── + all_ok = True + for tmp_f, t in zip(tmp_files, tracks): + dst = dir_path / t["target_name"] + + if dst.exists() and not cfg.force: + # Previous-run leftover with matching name — keep it, discard our temp. + log.debug(f" [已存在, 保留] {t['target_name']}") + _try_unlink(tmp_f) + t["status"] = Status.COMPLETED.value + t["error"] = None + if not t.get("converted_at"): + t["converted_at"] = _now_iso() + continue + + if dst.exists() and cfg.force: + try: + dst.unlink() + except OSError as e: + log.error(f" ✗ 无法覆盖 {dst.name}: {e}") + t["status"] = Status.FAILED.value + t["error"] = f"无法覆盖: {e}" + all_ok = False + continue + + try: + os.replace(str(tmp_f), str(dst)) + except OSError as e: + if e.errno == errno.EXDEV: + try: + shutil.move(str(tmp_f), str(dst)) + except OSError as e2: + log.error(f" ✗ move {tmp_f.name} → {dst.name}: {e2}") + t["status"] = Status.FAILED.value + t["error"] = f"move 失败: {e2}" + all_ok = False + continue + else: + log.error(f" ✗ rename {tmp_f.name} → {dst.name}: {e}") + t["status"] = Status.FAILED.value + t["error"] = f"rename 失败: {e}" + all_ok = False + continue + + t["status"] = Status.COMPLETED.value + t["error"] = None + t["converted_at"] = _now_iso() + log.info(f" ✓ [{t['num']:02d}/{total}] {t['target_name']}") + + return all_ok + finally: + _remove_dir(tmpdir) + + +# ────────────────────────── manifest ────────────────────────── + +class Manifest: + def __init__(self, path: Path, cfg: Config): + self.path = path + self.cfg = cfg + self.data: dict = {} + + def load_or_init(self) -> None: + if self.path.exists() and not self.cfg.force: + try: + text = self.path.read_text(encoding="utf-8") + data = json.loads(text) + if not isinstance(data, dict) or not isinstance(data.get("albums"), dict): + raise ValueError("bad structure") + self.data = data + logging.info(f"读取 manifest: {len(self.data['albums'])} 个专辑记录") + return + except (json.JSONDecodeError, ValueError, OSError) as e: + quarantine_corrupt(self.path, str(e)) + self.data = self._new() + + def _new(self) -> dict: + now = _now_iso() + return { + "manifest_version": MANIFEST_SCHEMA_VERSION, + "created_at": now, + "updated_at": now, + "source_root": str(self.cfg.source_root), + "stats": {}, + "albums": {}, + } + + def save(self) -> None: + self.data["updated_at"] = _now_iso() + self.data["stats"] = self._compute_stats() + atomic_write_json(self.path, self.data) + + def _compute_stats(self) -> dict: + by_album: dict[str, int] = {} + by_track: dict[str, int] = {} + total_tracks = 0 + for album in self.data.get("albums", {}).values(): + s = album.get("status", "unknown") + by_album[s] = by_album.get(s, 0) + 1 + for t in album.get("tracks") or []: + ts = t.get("status", "unknown") + by_track[ts] = by_track.get(ts, 0) + 1 + total_tracks += 1 + return { + "total_albums": len(self.data.get("albums", {})), + "by_album_status": by_album, + "by_track_status": by_track, + "total_tracks": total_tracks, + } + + +# ────────────────────────── driver ────────────────────────── + +def find_all_dirs(source_root: Path) -> list[Path]: + """Include source_root itself so single-album invocations work.""" + result: list[Path] = [source_root] + for dirpath, dirnames, _ in os.walk(source_root): + dirnames.sort() + for d in dirnames: + result.append(Path(dirpath) / d) + return result + + +def analyze_all(cfg: Config, deps: Deps, manifest: Manifest, log: logging.Logger) -> None: + log.info(f"扫描 {cfg.source_root} ...") + all_dirs = find_all_dirs(cfg.source_root) + log.info(f"发现 {len(all_dirs)} 个目录 (含根)") + + for d in all_dirs: + rel = "." if d == cfg.source_root else str(d.relative_to(cfg.source_root)) + existing = manifest.data["albums"].get(rel) + if existing and existing.get("status") == Status.COMPLETED.value and not cfg.force: + log.debug(f"[已完成 skip] {rel}") + continue + log.info(f"[分析] {rel}") + fresh = analyze_directory(d, cfg, deps) + if existing: + fresh = merge_entry(existing, fresh) + manifest.data["albums"][rel] = fresh + + manifest.save() + + +def process_all(cfg: Config, manifest: Manifest, log: logging.Logger) -> None: + albums = manifest.data["albums"] + keys = sorted(albums.keys()) + total = len(keys) + + actionable = {Status.ANALYZED.value, Status.PROCESSING.value, Status.FAILED.value} + + for i, rel in enumerate(keys, 1): + album = albums[rel] + status = album.get("status") + + if status == Status.COMPLETED.value and not cfg.force: + log.debug(f"[{i}/{total}] SKIP (已完成): {rel}") + continue + if status == Status.SKIPPED.value: + log.debug(f"[{i}/{total}] SKIP (无需切割): {rel}") + continue + if status == Status.UNSUPPORTED.value: + log.warning(f"[{i}/{total}] SKIP (源格式不支持): {rel}") + continue + if status not in actionable or not album.get("tracks"): + log.debug(f"[{i}/{total}] SKIP (状态 {status} 或无计划): {rel}") + continue + + log.info(f"[{i}/{total}] 切割: {rel}") + log.info(f" 源: {album['source_audio']} → {len(album['tracks'])} 轨" + f" ({album.get('output_format', '?')})") + + album["status"] = Status.PROCESSING.value + if album.get("started_at") is None: + album["started_at"] = _now_iso() + manifest.save() + + dir_path = cfg.source_root / rel if rel != "." else cfg.source_root + try: + all_ok = split_album(dir_path, album, cfg, log) + except KeyboardInterrupt: + manifest.save() + raise + except Exception as e: # last-resort safety net: single album must not kill batch + log.error(f" ✗ 未捕获异常: {type(e).__name__}: {e}") + album["issues"].append(f"未捕获异常 {type(e).__name__}: {str(e)[:200]}") + _mark_all_failed(album, f"未捕获异常: {type(e).__name__}: {str(e)[:100]}") + all_ok = False + finally: + manifest.save() + + album["status"] = Status.COMPLETED.value if all_ok else Status.FAILED.value + if all_ok: + album["completed_at"] = _now_iso() + manifest.save() + + +def print_report(manifest: Manifest, log: logging.Logger) -> None: + albums = manifest.data.get("albums", {}) + stats = manifest.data.get("stats") or {} + + log.info("") + log.info("=" * 70) + log.info("摘要") + log.info("=" * 70) + log.info(f"总目录数: {stats.get('total_albums', 0)}") + for k, v in sorted((stats.get("by_album_status") or {}).items()): + log.info(f" · album {k}: {v}") + log.info(f"分轨文件: {stats.get('total_tracks', 0)}") + for k, v in sorted((stats.get("by_track_status") or {}).items()): + log.info(f" · track {k}: {v}") + + to_split = [(r, a) for r, a in albums.items() + if a.get("status") == Status.ANALYZED.value] + if to_split: + log.info("") + log.info(f"[i] {len(to_split)} 个专辑待切割:") + for r, a in to_split[:20]: + log.info(f" • {r}: {a['source_audio']} → {len(a['tracks'])} 轨") + if len(to_split) > 20: + log.info(f" ... 及另外 {len(to_split) - 20} 个") + + unsupported = [(r, a) for r, a in albums.items() + if a.get("status") == Status.UNSUPPORTED.value] + if unsupported: + log.warning("") + log.warning(f"[!] {len(unsupported)} 个专辑源格式不支持 (跳过):") + for r, a in unsupported: + log.warning(f" • {r}") + for issue in a.get("issues", []): + log.warning(f" {issue}") + + failed = [(r, a) for r, a in albums.items() + if a.get("status") == Status.FAILED.value] + if failed: + log.error("") + log.error(f"[x] {len(failed)} 个专辑失败 (再次运行会重试):") + for r, a in failed: + log.error(f" • {r}") + for issue in a.get("issues", []): + log.error(f" {issue}") + failed_tracks = [t for t in a.get("tracks", []) + if t.get("status") == Status.FAILED.value] + for t in failed_tracks[:3]: + log.error(f" ✗ 轨 {t['num']:02d}: {t.get('error')}") + + +def _check_startup_deps(deps: Deps, log: logging.Logger) -> bool: + ok = True + if not deps.shnsplit: + log.error("未找到 shnsplit (shntool). 安装: apt install shntool") + ok = False + if not deps.cuetag: + log.error("未找到 cuetag (cuetools). 安装: apt install cuetools") + ok = False + if not deps.ffprobe: + log.warning("未找到 ffprobe (仅用于源文件元数据展示, 不影响切割)") + if not deps.flac: + log.warning("未找到 flac 二进制 → FLAC 源将无法切割 (仅支持 WAV)") + if not deps.mac: + log.info("未找到 mac (APE 解码器) → APE 源将标记 UNSUPPORTED. " + "安装可选: apt install monkeys-audio (或从源码构建)") + return ok + + +def main() -> int: + ap = argparse.ArgumentParser( + description="批量用 CUE 切割整轨无损音乐为分轨 (原地切割, 保留原文件, " + "shnsplit + cuetag 后端).", + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog="""示例: + # 最简用法 (位置参数) + %(prog)s /mnt/z/music + + # 仅分析, 检查切割计划 + %(prog)s /mnt/z/music --analyze-only + + # 命名参数 + %(prog)s -s /mnt/z/music + + # 强制重跑 (覆盖已有分轨) + %(prog)s /mnt/z/music --force +""", + ) + ap.add_argument("source_pos", nargs="?", metavar="SOURCE", + help="源目录 (位置参数). 也可用 --source/-s") + ap.add_argument("--source", "-s", dest="source_flag", + help="源目录 (等价于位置参数 SOURCE)") + ap.add_argument("--analyze-only", action="store_true", + help="只分析生成 manifest, 不实际切割") + ap.add_argument("--force", action="store_true", + help="忽略现有 manifest, 强制重新分析并覆盖已有分轨") + ap.add_argument("--verbose", "-v", action="store_true", help="详细日志") + ap.add_argument("--shnsplit", default="shnsplit", help="shnsplit 路径 (默认从 PATH 查)") + ap.add_argument("--cuetag", default="cuetag", help="cuetag 路径") + ap.add_argument("--ffprobe", default="ffprobe", help="ffprobe 路径") + ap.add_argument("--split-timeout", type=int, default=1800, + help="shnsplit 超时秒数 (默认 1800)") + args = ap.parse_args() + + source_arg = args.source_pos or args.source_flag + if not source_arg: + ap.error("必须提供源目录 (位置参数 SOURCE 或 --source/-s)") + + logging.basicConfig( + level=logging.DEBUG if args.verbose else logging.INFO, + format="%(asctime)s %(levelname)-5s %(message)s", + datefmt="%H:%M:%S", + ) + log = logging.getLogger("split_cue") + + source = Path(source_arg).resolve() + if not source.exists(): + log.error(f"源目录不存在: {source}") + return 2 + if not source.is_dir(): + log.error(f"源路径不是目录: {source}") + return 2 + + cfg = Config( + source_root=source, + force=args.force, + analyze_only=args.analyze_only, + shnsplit=args.shnsplit, + cuetag=args.cuetag, + ffprobe=args.ffprobe, + split_timeout_sec=args.split_timeout, + ) + + deps = probe_deps(cfg) + if not _check_startup_deps(deps, log): + return 3 + + manifest_path = source / MANIFEST_NAME + manifest = Manifest(manifest_path, cfg) + manifest.load_or_init() + + log.info(f"源目录: {source}") + log.info(f"Manifest: {manifest_path}") + log.info(f"支持源格式: {sorted(deps.supported_source_exts())}") + log.info(f"模式: {'仅分析' if cfg.analyze_only else '分析 + 切割'}") + log.info(f"Resume: {'关闭 (--force)' if cfg.force else '开启 (跳过已完成)'}") + log.info("") + + try: + analyze_all(cfg, deps, manifest, log) + if cfg.analyze_only: + print_report(manifest, log) + log.info("") + log.info("仅分析完成. 查看 manifest 确认切割计划后, 再跑一次进行切割.") + return 0 + process_all(cfg, manifest, log) + print_report(manifest, log) + except KeyboardInterrupt: + log.warning("") + log.warning("用户中断 (Ctrl+C) - 保存 manifest 后退出") + try: + manifest.save() + except OSError as e: + log.error(f"保存 manifest 失败: {e}") + return 130 + + return 0 + + +if __name__ == "__main__": + sys.exit(main())