Files
atlas/synthmind/extract_audio.py

415 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
extract_audio.py - 视频音频批量提取脚本
功能:
- 扫描单个视频文件或遍历指定目录中的所有视频文件
- 维护 manifest.json 进度文件,支持断点续传
- 使用 FFmpeg 提取音频(MP3, 64k, 22050Hz, 单声道)
- 支持视频切分(处理超大文件)
- 支持含空格和中文的文件名
用法:
python extract_audio.py <视频文件或目录> [选项]
示例:
python extract_audio.py /path/to/videos/
python extract_audio.py /path/to/video.mp4
python extract_audio.py /path/to/videos/ --output /path/to/output/
python extract_audio.py /path/to/videos/ --split-size 600 # 切分为600秒片段
python extract_audio.py /path/to/videos/ --retry-failed # 重新处理失败项
python extract_audio.py /path/to/videos/ --status # 查看进度状态
"""
import argparse
import json
import os
import subprocess
import sys
import time
from pathlib import Path
# ─── 常量 ──────────────────────────────────────────────────────────────────────
MANIFEST_NAME = "manifest.json"
VIDEO_EXTS = {".mp4", ".mkv", ".avi", ".mov", ".flv", ".wmv", ".webm", ".m4v"}
# FFmpeg 音频提取参数(语音课程优化:体积小)
FFMPEG_AUDIO_OPTS = [
"-vn", # 去除视频轨道
"-acodec", "libmp3lame",
"-ab", "64k",
"-ar", "22050",
"-ac", "1", # 单声道
"-f", "mp3",
]
# 视频切分:默认超过此时长(秒)才切分
DEFAULT_SPLIT_THRESHOLD_SECS = 3600 # 1小时
DEFAULT_SPLIT_SIZE_SECS = 600 # 每片10分钟
# ─── 日志 ──────────────────────────────────────────────────────────────────────
def log(msg: str, level: str = "INFO"):
ts = time.strftime("%H:%M:%S")
prefix = {"INFO": " ", "OK": "✅", "ERR": "❌", "WARN": "⚠️ ", "STEP": "▶ "}.get(level, " ")
print(f"[{ts}] {prefix} {msg}", flush=True)
# ─── Manifest 操作 (v2 schema) ────────────────────────────────────────────────
#
# v2 schema:每个条目形如
# {
# "extraction": { "extracted": bool, "audio_path": str, "split_segments": [],
# "error": str|None, "updated_at": str },
# "transcriptions": {
# "<model>[:<suffix>]": { "transcribed": bool, "txt_path": str, ... }
# }
# }
# transcribe_audio.py 会把转录结果并入同一条视频记录的 transcriptions 里。
# ─────────────────────────────────────────────────────────────────────────────
MANIFEST_VERSION = 2
def load_manifest(manifest_path: Path) -> dict:
if not manifest_path.exists():
return {"version": MANIFEST_VERSION, "files": {}}
with open(manifest_path, encoding="utf-8") as f:
data = json.load(f)
# 旧版 manifest (v1) 直接废弃,从头开始(用户已明确接受)
if data.get("version", 1) < MANIFEST_VERSION:
log("检测到旧版 manifest.json (v1),将以 v2 schema 重新初始化 - 旧记录将被覆盖", "WARN")
return {"version": MANIFEST_VERSION, "files": {}}
return data
def save_manifest(manifest_path: Path, manifest: dict):
manifest_path.parent.mkdir(parents=True, exist_ok=True)
with open(manifest_path, "w", encoding="utf-8") as f:
json.dump(manifest, f, ensure_ascii=False, indent=2)
def get_file_entry(manifest: dict, video_path: str) -> dict:
"""获取或初始化某个视频文件的 manifest 条目 (v2 schema)"""
if video_path not in manifest["files"]:
manifest["files"][video_path] = {
"extraction": None,
"transcriptions": {},
}
entry = manifest["files"][video_path]
if "transcriptions" not in entry:
entry["transcriptions"] = {}
return entry
def mark_done(manifest: dict, manifest_path: Path, video_path: str,
audio_path: str, segments: list = None):
entry = get_file_entry(manifest, video_path)
entry["extraction"] = {
"extracted": True,
"audio_path": audio_path,
"split_segments": segments or [],
"error": None,
"updated_at": time.strftime("%Y-%m-%dT%H:%M:%S"),
}
save_manifest(manifest_path, manifest)
def mark_failed(manifest: dict, manifest_path: Path, video_path: str, error: str):
entry = get_file_entry(manifest, video_path)
prev = entry.get("extraction") or {}
entry["extraction"] = {
"extracted": False,
"audio_path": prev.get("audio_path"),
"split_segments": prev.get("split_segments") or [],
"error": error,
"updated_at": time.strftime("%Y-%m-%dT%H:%M:%S"),
}
save_manifest(manifest_path, manifest)
# ─── FFmpeg 工具函数 ────────────────────────────────────────────────────────────
def check_ffmpeg():
try:
subprocess.run(["ffmpeg", "-version"], capture_output=True, check=True)
return True
except (subprocess.CalledProcessError, FileNotFoundError):
log("未找到 ffmpeg 命令,请确认已安装:sudo apt install ffmpeg", "ERR")
return False
def get_video_duration(video_path: str) -> float:
"""用 ffprobe 获取视频时长(秒),失败返回 0"""
try:
r = subprocess.run(
["ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1", video_path],
capture_output=True, text=True
)
return float(r.stdout.strip())
except Exception:
return 0.0
def extract_audio_simple(video_path: str, mp3_path: str) -> bool:
"""
从视频提取音频到 mp3 文件。
注意:MP4 的 moov atom 可能在文件尾部,需要 seek,因此不能用 pipe:0 输入。
直接用 -i <file> 让 ffmpeg 自行 seek,同时用 -f mp3 明确输出格式。
"""
cmd = ["ffmpeg", "-y", "-i", video_path] + FFMPEG_AUDIO_OPTS + [mp3_path]
r = subprocess.run(cmd, capture_output=True)
if r.returncode != 0:
return False
return os.path.exists(mp3_path) and os.path.getsize(mp3_path) > 1024
def split_video(video_path: str, output_dir: str, segment_secs: int) -> list:
"""
将视频切分为多个片段。
返回切分后的 mp4 片段路径列表;若失败返回空列表。
"""
stem = Path(video_path).stem
ext = Path(video_path).suffix
out_pattern = os.path.join(output_dir, f"{stem}_seg%03d{ext}")
cmd = [
"ffmpeg", "-y", "-i", video_path,
"-c", "copy",
"-segment_time", str(segment_secs),
"-f", "segment",
"-reset_timestamps", "1",
out_pattern
]
r = subprocess.run(cmd, capture_output=True)
if r.returncode != 0:
return []
# 收集生成的片段文件(按名称排序)
parent = Path(output_dir)
segments = sorted(
str(p) for p in parent.glob(f"{stem}_seg*{ext}")
)
return segments
def extract_audio_from_segments(segments: list, mp3_dir: str, stem: str) -> tuple:
"""
对切分后的每个片段提取音频,返回 (segment_mp3s, all_ok)
"""
segment_mp3s = []
all_ok = True
for seg in segments:
seg_stem = Path(seg).stem
seg_mp3 = os.path.join(mp3_dir, f"{seg_stem}.mp3")
ok = extract_audio_simple(seg, seg_mp3)
if ok:
segment_mp3s.append(seg_mp3)
log(f" 片段 {Path(seg).name} → {Path(seg_mp3).name}", "OK")
else:
log(f" 片段 {Path(seg).name} 提取失败", "ERR")
all_ok = False
return segment_mp3s, all_ok
# ─── 文件扫描 ──────────────────────────────────────────────────────────────────
def scan_videos(source: str) -> list:
"""扫描单文件或目录,返回视频文件路径列表"""
p = Path(source)
if p.is_file():
if p.suffix.lower() in VIDEO_EXTS:
return [str(p.resolve())]
else:
log(f"不支持的文件格式: {p.suffix}", "ERR")
return []
elif p.is_dir():
files = []
for f in sorted(p.rglob("*")):
if f.suffix.lower() in VIDEO_EXTS:
files.append(str(f.resolve()))
return files
else:
log(f"路径不存在: {source}", "ERR")
return []
def mp3_exists(mp3_path: str) -> bool:
return os.path.exists(mp3_path) and os.path.getsize(mp3_path) > 0
# ─── 状态展示 ──────────────────────────────────────────────────────────────────
def show_status(manifest: dict, all_videos: list):
print("\n" + "=" * 60)
print("📊 提取音频进度状态")
print("=" * 60)
done, failed = 0, 0
for v in all_videos:
ext = ((manifest["files"].get(v) or {}).get("extraction")) or {}
if ext.get("extracted"):
done += 1
elif ext.get("error"):
failed += 1
pending = len(all_videos) - done - failed
print(f" 总文件数 : {len(all_videos)}")
print(f" 已完成 : {done}")
print(f" 失败 : {failed}")
print(f" 待处理 : {pending}")
print("=" * 60)
for v in all_videos:
ext = ((manifest["files"].get(v) or {}).get("extraction")) or {}
if ext.get("extracted"):
status = "✅ 完成"
elif ext.get("error"):
status = f"❌ 失败: {ext['error']}"
else:
status = "⏳ 待处理"
print(f" {status} {Path(v).name}")
print()
# ─── 主流程 ────────────────────────────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(
description="视频音频批量提取工具",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog=__doc__
)
parser.add_argument("source", help="视频文件路径或包含视频的目录")
parser.add_argument("--output", "-o", default=None,
help="MP3 输出目录(默认与视频文件同目录)")
parser.add_argument("--manifest", "-m", default=None,
help="manifest.json 路径(默认在 source 目录下)")
parser.add_argument("--split-size", type=int, default=None,
help=f"切分阈值(秒):超过此时长的视频会被切分(默认不切分)")
parser.add_argument("--split-segment", type=int, default=DEFAULT_SPLIT_SIZE_SECS,
help=f"每个切分片段的时长(秒,默认 {DEFAULT_SPLIT_SIZE_SECS})")
parser.add_argument("--retry-failed", action="store_true",
help="重新处理上次失败的文件")
parser.add_argument("--status", action="store_true",
help="仅查看进度状态,不执行提取")
args = parser.parse_args()
if not check_ffmpeg():
sys.exit(1)
# 确定 manifest 路径
source_path = Path(args.source).resolve()
base_dir = source_path if source_path.is_dir() else source_path.parent
manifest_path = Path(args.manifest) if args.manifest else base_dir / MANIFEST_NAME
manifest = load_manifest(manifest_path)
# 扫描视频文件
all_videos = scan_videos(args.source)
if not all_videos:
log("未找到任何视频文件", "WARN")
sys.exit(0)
log(f"扫描到 {len(all_videos)} 个视频文件", "INFO")
# 仅查看状态
if args.status:
show_status(manifest, all_videos)
return
# 确定待处理列表
pending = []
for v in all_videos:
entry = get_file_entry(manifest, v)
ext = entry.get("extraction") or {}
already_done = ext.get("extracted") and ext.get("audio_path") and \
mp3_exists(ext["audio_path"])
is_failed = bool(ext.get("error"))
if already_done:
continue
if is_failed and not args.retry_failed:
log(f"跳过(上次失败,用 --retry-failed 重试): {Path(v).name}", "WARN")
continue
pending.append(v)
done_count = len(all_videos) - len(pending)
log(f"📊 总体: {done_count}/{len(all_videos)} 已完成,{len(pending)} 待处理")
if not pending:
log("全部已完成 ✅", "OK")
return
success, failed_list = 0, []
for i, video in enumerate(pending, 1):
video_name = Path(video).name
stem = Path(video).stem
video_dir = Path(video).parent
# 确定输出目录
out_dir = Path(args.output).resolve() if args.output else video_dir
out_dir.mkdir(parents=True, exist_ok=True)
mp3_path = str(out_dir / f"{stem}.mp3")
log(f"\n[{i}/{len(pending)}] {video_name}", "STEP")
try:
# 判断是否需要切分
duration = get_video_duration(video)
need_split = args.split_size and duration > 0 and duration > args.split_size
if need_split:
log(f" 视频时长 {duration:.0f}s > {args.split_size}s,启动切分模式")
seg_dir = out_dir / f"{stem}_segments"
seg_dir.mkdir(exist_ok=True)
segments = split_video(video, str(seg_dir), args.split_segment)
if not segments:
raise RuntimeError("视频切分失败")
log(f" 切分为 {len(segments)} 个片段")
seg_mp3s, all_ok = extract_audio_from_segments(
segments, str(out_dir), stem
)
if not all_ok:
raise RuntimeError("部分片段音频提取失败")
# 记录切分片段到 manifest(主 mp3_path 设为第一个片段)
mark_done(manifest, manifest_path, video, seg_mp3s[0] if seg_mp3s else mp3_path,
segments=seg_mp3s)
else:
# 普通提取
log(f" 📥 提取音频中...")
if not extract_audio_simple(video, mp3_path):
raise RuntimeError("FFmpeg 音频提取失败")
size = os.path.getsize(mp3_path)
log(f" 输出: {Path(mp3_path).name} ({size // 1024} KB)", "OK")
mark_done(manifest, manifest_path, video, mp3_path)
success += 1
log(f" 进度: {done_count + success}/{len(all_videos)}", "OK")
except Exception as e:
err_msg = str(e)
log(f" {err_msg}", "ERR")
mark_failed(manifest, manifest_path, video, err_msg)
failed_list.append(video_name)
# 清理可能残留的不完整 mp3
if os.path.exists(mp3_path) and os.path.getsize(mp3_path) == 0:
os.remove(mp3_path)
print(f"\n{'='*60}")
log(f"🏁 完成: {success}/{len(pending)} 失败: {len(failed_list)}", "INFO")
if failed_list:
log(f"失败文件列表:", "ERR")
for f in failed_list:
print(f" - {f}")
print(" 运行时加 --retry-failed 可重新处理失败项")
if __name__ == "__main__":
main()