extract_audio and transcribe_audio scripts
This commit is contained in:
384
synthmind/extract_audio.py
Normal file
384
synthmind/extract_audio.py
Normal file
@@ -0,0 +1,384 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
extract_audio.py - 视频音频批量提取脚本
|
||||
功能:
|
||||
- 扫描单个视频文件或遍历指定目录中的所有视频文件
|
||||
- 维护 manifest.json 进度文件,支持断点续传
|
||||
- 使用 FFmpeg 提取音频(MP3, 64k, 22050Hz, 单声道)
|
||||
- 支持视频切分(处理超大文件)
|
||||
- 支持含空格和中文的文件名
|
||||
|
||||
用法:
|
||||
python extract_audio.py <视频文件或目录> [选项]
|
||||
|
||||
示例:
|
||||
python extract_audio.py /path/to/videos/
|
||||
python extract_audio.py /path/to/video.mp4
|
||||
python extract_audio.py /path/to/videos/ --output /path/to/output/
|
||||
python extract_audio.py /path/to/videos/ --split-size 600 # 切分为600秒片段
|
||||
python extract_audio.py /path/to/videos/ --retry-failed # 重新处理失败项
|
||||
python extract_audio.py /path/to/videos/ --status # 查看进度状态
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
# ─── 常量 ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
MANIFEST_NAME = "manifest.json"
|
||||
VIDEO_EXTS = {".mp4", ".mkv", ".avi", ".mov", ".flv", ".wmv", ".webm", ".m4v"}
|
||||
|
||||
# FFmpeg 音频提取参数(语音课程优化:体积小)
|
||||
FFMPEG_AUDIO_OPTS = [
|
||||
"-vn", # 去除视频轨道
|
||||
"-acodec", "libmp3lame",
|
||||
"-ab", "64k",
|
||||
"-ar", "22050",
|
||||
"-ac", "1", # 单声道
|
||||
"-f", "mp3",
|
||||
]
|
||||
|
||||
# 视频切分:默认超过此时长(秒)才切分
|
||||
DEFAULT_SPLIT_THRESHOLD_SECS = 3600 # 1小时
|
||||
DEFAULT_SPLIT_SIZE_SECS = 600 # 每片10分钟
|
||||
|
||||
|
||||
# ─── 日志 ──────────────────────────────────────────────────────────────────────
|
||||
|
||||
def log(msg: str, level: str = "INFO"):
|
||||
ts = time.strftime("%H:%M:%S")
|
||||
prefix = {"INFO": " ", "OK": "✅", "ERR": "❌", "WARN": "⚠️ ", "STEP": "▶ "}.get(level, " ")
|
||||
print(f"[{ts}] {prefix} {msg}", flush=True)
|
||||
|
||||
|
||||
# ─── Manifest 操作 ─────────────────────────────────────────────────────────────
|
||||
|
||||
def load_manifest(manifest_path: Path) -> dict:
|
||||
if manifest_path.exists():
|
||||
with open(manifest_path, encoding="utf-8") as f:
|
||||
return json.load(f)
|
||||
return {"version": 1, "files": {}}
|
||||
|
||||
|
||||
def save_manifest(manifest_path: Path, manifest: dict):
|
||||
manifest_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(manifest_path, "w", encoding="utf-8") as f:
|
||||
json.dump(manifest, f, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def get_file_entry(manifest: dict, video_path: str) -> dict:
|
||||
"""获取或初始化某个视频文件的 manifest 条目"""
|
||||
if video_path not in manifest["files"]:
|
||||
manifest["files"][video_path] = {
|
||||
"audio_extracted": False,
|
||||
"audio_path": None,
|
||||
"split_segments": [], # 切分后的片段列表(若有)
|
||||
"error": None,
|
||||
"updated_at": None,
|
||||
}
|
||||
return manifest["files"][video_path]
|
||||
|
||||
|
||||
def mark_done(manifest: dict, manifest_path: Path, video_path: str,
|
||||
audio_path: str, segments: list = None):
|
||||
entry = get_file_entry(manifest, video_path)
|
||||
entry["audio_extracted"] = True
|
||||
entry["audio_path"] = audio_path
|
||||
entry["split_segments"] = segments or []
|
||||
entry["error"] = None
|
||||
entry["updated_at"] = time.strftime("%Y-%m-%dT%H:%M:%S")
|
||||
save_manifest(manifest_path, manifest)
|
||||
|
||||
|
||||
def mark_failed(manifest: dict, manifest_path: Path, video_path: str, error: str):
|
||||
entry = get_file_entry(manifest, video_path)
|
||||
entry["audio_extracted"] = False
|
||||
entry["error"] = error
|
||||
entry["updated_at"] = time.strftime("%Y-%m-%dT%H:%M:%S")
|
||||
save_manifest(manifest_path, manifest)
|
||||
|
||||
|
||||
# ─── FFmpeg 工具函数 ────────────────────────────────────────────────────────────
|
||||
|
||||
def check_ffmpeg():
|
||||
try:
|
||||
subprocess.run(["ffmpeg", "-version"], capture_output=True, check=True)
|
||||
return True
|
||||
except (subprocess.CalledProcessError, FileNotFoundError):
|
||||
log("未找到 ffmpeg 命令,请确认已安装:sudo apt install ffmpeg", "ERR")
|
||||
return False
|
||||
|
||||
|
||||
def get_video_duration(video_path: str) -> float:
|
||||
"""用 ffprobe 获取视频时长(秒),失败返回 0"""
|
||||
try:
|
||||
r = subprocess.run(
|
||||
["ffprobe", "-v", "error", "-show_entries", "format=duration",
|
||||
"-of", "default=noprint_wrappers=1:nokey=1", video_path],
|
||||
capture_output=True, text=True
|
||||
)
|
||||
return float(r.stdout.strip())
|
||||
except Exception:
|
||||
return 0.0
|
||||
|
||||
|
||||
def extract_audio_simple(video_path: str, mp3_path: str) -> bool:
|
||||
"""
|
||||
从视频提取音频到 mp3 文件。
|
||||
注意:MP4 的 moov atom 可能在文件尾部,需要 seek,因此不能用 pipe:0 输入。
|
||||
直接用 -i <file> 让 ffmpeg 自行 seek,同时用 -f mp3 明确输出格式。
|
||||
"""
|
||||
cmd = ["ffmpeg", "-y", "-i", video_path] + FFMPEG_AUDIO_OPTS + [mp3_path]
|
||||
r = subprocess.run(cmd, capture_output=True)
|
||||
if r.returncode != 0:
|
||||
return False
|
||||
return os.path.exists(mp3_path) and os.path.getsize(mp3_path) > 1024
|
||||
|
||||
|
||||
def split_video(video_path: str, output_dir: str, segment_secs: int) -> list:
|
||||
"""
|
||||
将视频切分为多个片段。
|
||||
返回切分后的 mp4 片段路径列表;若失败返回空列表。
|
||||
"""
|
||||
stem = Path(video_path).stem
|
||||
ext = Path(video_path).suffix
|
||||
out_pattern = os.path.join(output_dir, f"{stem}_seg%03d{ext}")
|
||||
|
||||
cmd = [
|
||||
"ffmpeg", "-y", "-i", video_path,
|
||||
"-c", "copy",
|
||||
"-segment_time", str(segment_secs),
|
||||
"-f", "segment",
|
||||
"-reset_timestamps", "1",
|
||||
out_pattern
|
||||
]
|
||||
r = subprocess.run(cmd, capture_output=True)
|
||||
if r.returncode != 0:
|
||||
return []
|
||||
|
||||
# 收集生成的片段文件(按名称排序)
|
||||
parent = Path(output_dir)
|
||||
segments = sorted(
|
||||
str(p) for p in parent.glob(f"{stem}_seg*{ext}")
|
||||
)
|
||||
return segments
|
||||
|
||||
|
||||
def extract_audio_from_segments(segments: list, mp3_dir: str, stem: str) -> tuple:
|
||||
"""
|
||||
对切分后的每个片段提取音频,返回 (segment_mp3s, all_ok)
|
||||
"""
|
||||
segment_mp3s = []
|
||||
all_ok = True
|
||||
for seg in segments:
|
||||
seg_stem = Path(seg).stem
|
||||
seg_mp3 = os.path.join(mp3_dir, f"{seg_stem}.mp3")
|
||||
ok = extract_audio_simple(seg, seg_mp3)
|
||||
if ok:
|
||||
segment_mp3s.append(seg_mp3)
|
||||
log(f" 片段 {Path(seg).name} → {Path(seg_mp3).name}", "OK")
|
||||
else:
|
||||
log(f" 片段 {Path(seg).name} 提取失败", "ERR")
|
||||
all_ok = False
|
||||
return segment_mp3s, all_ok
|
||||
|
||||
|
||||
# ─── 文件扫描 ──────────────────────────────────────────────────────────────────
|
||||
|
||||
def scan_videos(source: str) -> list:
|
||||
"""扫描单文件或目录,返回视频文件路径列表"""
|
||||
p = Path(source)
|
||||
if p.is_file():
|
||||
if p.suffix.lower() in VIDEO_EXTS:
|
||||
return [str(p.resolve())]
|
||||
else:
|
||||
log(f"不支持的文件格式: {p.suffix}", "ERR")
|
||||
return []
|
||||
elif p.is_dir():
|
||||
files = []
|
||||
for f in sorted(p.rglob("*")):
|
||||
if f.suffix.lower() in VIDEO_EXTS:
|
||||
files.append(str(f.resolve()))
|
||||
return files
|
||||
else:
|
||||
log(f"路径不存在: {source}", "ERR")
|
||||
return []
|
||||
|
||||
|
||||
def mp3_exists(mp3_path: str) -> bool:
|
||||
return os.path.exists(mp3_path) and os.path.getsize(mp3_path) > 0
|
||||
|
||||
|
||||
# ─── 状态展示 ──────────────────────────────────────────────────────────────────
|
||||
|
||||
def show_status(manifest: dict, all_videos: list):
|
||||
print("\n" + "=" * 60)
|
||||
print("📊 提取音频进度状态")
|
||||
print("=" * 60)
|
||||
done = sum(1 for v in all_videos
|
||||
if manifest["files"].get(v, {}).get("audio_extracted"))
|
||||
failed = sum(1 for v in all_videos
|
||||
if manifest["files"].get(v, {}).get("error"))
|
||||
pending = len(all_videos) - done
|
||||
print(f" 总文件数 : {len(all_videos)}")
|
||||
print(f" 已完成 : {done}")
|
||||
print(f" 失败 : {failed}")
|
||||
print(f" 待处理 : {pending}")
|
||||
print("=" * 60)
|
||||
for v in all_videos:
|
||||
entry = manifest["files"].get(v, {})
|
||||
if entry.get("audio_extracted"):
|
||||
status = "✅ 完成"
|
||||
elif entry.get("error"):
|
||||
status = f"❌ 失败: {entry['error']}"
|
||||
else:
|
||||
status = "⏳ 待处理"
|
||||
print(f" {status} {Path(v).name}")
|
||||
print()
|
||||
|
||||
|
||||
# ─── 主流程 ────────────────────────────────────────────────────────────────────
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="视频音频批量提取工具",
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
epilog=__doc__
|
||||
)
|
||||
parser.add_argument("source", help="视频文件路径或包含视频的目录")
|
||||
parser.add_argument("--output", "-o", default=None,
|
||||
help="MP3 输出目录(默认与视频文件同目录)")
|
||||
parser.add_argument("--manifest", "-m", default=None,
|
||||
help="manifest.json 路径(默认在 source 目录下)")
|
||||
parser.add_argument("--split-size", type=int, default=None,
|
||||
help=f"切分阈值(秒):超过此时长的视频会被切分(默认不切分)")
|
||||
parser.add_argument("--split-segment", type=int, default=DEFAULT_SPLIT_SIZE_SECS,
|
||||
help=f"每个切分片段的时长(秒,默认 {DEFAULT_SPLIT_SIZE_SECS})")
|
||||
parser.add_argument("--retry-failed", action="store_true",
|
||||
help="重新处理上次失败的文件")
|
||||
parser.add_argument("--status", action="store_true",
|
||||
help="仅查看进度状态,不执行提取")
|
||||
args = parser.parse_args()
|
||||
|
||||
if not check_ffmpeg():
|
||||
sys.exit(1)
|
||||
|
||||
# 确定 manifest 路径
|
||||
source_path = Path(args.source).resolve()
|
||||
base_dir = source_path if source_path.is_dir() else source_path.parent
|
||||
manifest_path = Path(args.manifest) if args.manifest else base_dir / MANIFEST_NAME
|
||||
|
||||
manifest = load_manifest(manifest_path)
|
||||
|
||||
# 扫描视频文件
|
||||
all_videos = scan_videos(args.source)
|
||||
if not all_videos:
|
||||
log("未找到任何视频文件", "WARN")
|
||||
sys.exit(0)
|
||||
|
||||
log(f"扫描到 {len(all_videos)} 个视频文件", "INFO")
|
||||
|
||||
# 仅查看状态
|
||||
if args.status:
|
||||
show_status(manifest, all_videos)
|
||||
return
|
||||
|
||||
# 确定待处理列表
|
||||
pending = []
|
||||
for v in all_videos:
|
||||
entry = get_file_entry(manifest, v)
|
||||
already_done = entry.get("audio_extracted") and entry.get("audio_path") and \
|
||||
mp3_exists(entry["audio_path"])
|
||||
is_failed = bool(entry.get("error"))
|
||||
|
||||
if already_done:
|
||||
continue
|
||||
if is_failed and not args.retry_failed:
|
||||
log(f"跳过(上次失败,用 --retry-failed 重试): {Path(v).name}", "WARN")
|
||||
continue
|
||||
pending.append(v)
|
||||
|
||||
done_count = len(all_videos) - len(pending)
|
||||
log(f"📊 总体: {done_count}/{len(all_videos)} 已完成,{len(pending)} 待处理")
|
||||
|
||||
if not pending:
|
||||
log("全部已完成 ✅", "OK")
|
||||
return
|
||||
|
||||
success, failed_list = 0, []
|
||||
|
||||
for i, video in enumerate(pending, 1):
|
||||
video_name = Path(video).name
|
||||
stem = Path(video).stem
|
||||
video_dir = Path(video).parent
|
||||
|
||||
# 确定输出目录
|
||||
out_dir = Path(args.output).resolve() if args.output else video_dir
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
mp3_path = str(out_dir / f"{stem}.mp3")
|
||||
|
||||
log(f"\n[{i}/{len(pending)}] {video_name}", "STEP")
|
||||
|
||||
try:
|
||||
# 判断是否需要切分
|
||||
duration = get_video_duration(video)
|
||||
need_split = args.split_size and duration > 0 and duration > args.split_size
|
||||
|
||||
if need_split:
|
||||
log(f" 视频时长 {duration:.0f}s > {args.split_size}s,启动切分模式")
|
||||
seg_dir = out_dir / f"{stem}_segments"
|
||||
seg_dir.mkdir(exist_ok=True)
|
||||
|
||||
segments = split_video(video, str(seg_dir), args.split_segment)
|
||||
if not segments:
|
||||
raise RuntimeError("视频切分失败")
|
||||
log(f" 切分为 {len(segments)} 个片段")
|
||||
|
||||
seg_mp3s, all_ok = extract_audio_from_segments(
|
||||
segments, str(out_dir), stem
|
||||
)
|
||||
if not all_ok:
|
||||
raise RuntimeError("部分片段音频提取失败")
|
||||
|
||||
# 记录切分片段到 manifest(主 mp3_path 设为第一个片段)
|
||||
mark_done(manifest, manifest_path, video, seg_mp3s[0] if seg_mp3s else mp3_path,
|
||||
segments=seg_mp3s)
|
||||
else:
|
||||
# 普通提取
|
||||
log(f" 📥 提取音频中...")
|
||||
if not extract_audio_simple(video, mp3_path):
|
||||
raise RuntimeError("FFmpeg 音频提取失败")
|
||||
|
||||
size = os.path.getsize(mp3_path)
|
||||
log(f" 输出: {Path(mp3_path).name} ({size // 1024} KB)", "OK")
|
||||
mark_done(manifest, manifest_path, video, mp3_path)
|
||||
|
||||
success += 1
|
||||
log(f" 进度: {done_count + success}/{len(all_videos)}", "OK")
|
||||
|
||||
except Exception as e:
|
||||
err_msg = str(e)
|
||||
log(f" {err_msg}", "ERR")
|
||||
mark_failed(manifest, manifest_path, video, err_msg)
|
||||
failed_list.append(video_name)
|
||||
# 清理可能残留的不完整 mp3
|
||||
if os.path.exists(mp3_path) and os.path.getsize(mp3_path) == 0:
|
||||
os.remove(mp3_path)
|
||||
|
||||
print(f"\n{'='*60}")
|
||||
log(f"🏁 完成: {success}/{len(pending)} 失败: {len(failed_list)}", "INFO")
|
||||
if failed_list:
|
||||
log(f"失败文件列表:", "ERR")
|
||||
for f in failed_list:
|
||||
print(f" - {f}")
|
||||
print(" 运行时加 --retry-failed 可重新处理失败项")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user