cang-ying/video-ai-system/tools/audio_pipeline.py

244 lines
10 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""
audio_pipeline.py — 声画流水线:配音 / 字幕 / BGM / 混音
剧本有声成片的最后一段:把分镜 dialogue 变成配音+字幕+BGM混入成片。
用法:
python audio_pipeline.py voice <分镜.json> <输出目录> # Edge-TTS 配音
python audio_pipeline.py srt <分镜.json> <输出目录> # 生成 SRT 字幕
python audio_pipeline.py bgm <输出目录> <秒数> # 生成 BGMstable_audio 失败则 ffmpeg 氛围音)
python audio_pipeline.py mix <成片.mp4> <配音目录> <bgm.mp3> <输出.mp4> # 混音
"""
import os, sys, json, asyncio, subprocess, math
VOICES = {
"": "zh-CN-YunxiNeural",
"": "zh-CN-XiaoxiaoNeural",
"旁白": "zh-CN-XiaoyiNeural",
"少年": "zh-CN-YunyangNeural",
"老人": "zh-CN-YunfengNeural",
}
def load_shots(sb_json):
d = json.load(open(sb_json, encoding="utf-8"))
shots = d.get("shots", [])
return shots
def _dialogue_of(shot):
"""从分镜镜头提取台词(支持字符串或 dict"""
dl = shot.get("dialogue", "")
if not dl:
return []
if isinstance(dl, str):
return [{"text": dl, "char": shot.get("characters", [""])[0] if shot.get("characters") else "旁白"}]
if isinstance(dl, list):
out = []
for it in dl:
if isinstance(it, str):
out.append({"text": it, "char": shot.get("characters", [""])[0] if shot.get("characters") else "旁白"})
elif isinstance(it, dict):
out.append({"text": it.get("text", ""), "char": it.get("character", it.get("char", "旁白"))})
return [x for x in out if x["text"].strip()]
return []
def generate_voice(sb_json, out_dir):
"""Edge-TTS 逐条配音。返回 [(shot_no, 文件路径, 台词, 时长)]"""
shots = load_shots(sb_json)
vdir = os.path.join(out_dir, "voice")
os.makedirs(vdir, exist_ok=True)
results = []
for i, sh in enumerate(shots):
lines = _dialogue_of(sh)
if not lines:
continue
shot_no = sh.get("shot_number", f"S{i+1:02d}")
for j, ln in enumerate(lines):
voice = VOICES.get(ln["char"], VOICES.get(ln["char"][:1], "zh-CN-XiaoxiaoNeural"))
fname = f"shot_{i+1:04d}_{j:02d}.mp3"
fpath = os.path.join(vdir, fname)
if not os.path.isfile(fpath):
text = ln["text"][:500]
try:
asyncio.run(_tts(text, voice, fpath))
except Exception as e:
print(f" ⚠️ 配音失败 {shot_no}:{ln['char']}: {e}")
continue
dur = _mp3_duration(fpath)
results.append({"shot": shot_no, "char": ln["char"], "text": ln["text"],
"file": fpath, "duration": dur})
print(f" 🎙 [{shot_no}] {ln['char']}: {ln['text'][:40]} ({dur:.1f}s)")
return results
async def _tts(text, voice, fpath):
import edge_tts
c = edge_tts.Communicate(text, voice)
await c.save(fpath)
def _mp3_duration(path):
try:
r = subprocess.run(["ffprobe", "-v", "quiet", "-show_entries", "format=duration",
"-of", "csv=p=0", path], capture_output=True, text=True)
return float(r.stdout.strip() or 0)
except Exception:
return 0
def generate_srt(sb_json, out_dir, voice_results=None):
"""生成 SRT 字幕(基于配音时长分配时间轴)。"""
shots = load_shots(sb_json)
sdir = os.path.join(out_dir, "subs")
os.makedirs(sdir, exist_ok=True)
srt_path = os.path.join(sdir, "subtitles.srt")
idx = 1
t = 0.0
lines = []
for i, sh in enumerate(shots):
lines_d = _dialogue_of(sh)
if not lines_d:
continue
shot_no = sh.get("shot_number", f"S{i+1:02d}")
for ln in lines_d:
dur = 2.5
if voice_results:
for vr in voice_results:
if vr["shot"] == shot_no and vr["char"] == ln["char"]:
dur = max(1.0, vr["duration"] + 0.3)
break
st = t
t += dur
lines.append(f"{idx}\n{_fmt_srt(st)} --> {_fmt_srt(t)}\n{ln['char']}: {ln['text']}\n")
idx += 1
with open(srt_path, "w", encoding="utf-8") as f:
f.write("\n".join(lines))
print(f" 📝 字幕: {srt_path} ({idx-1} 条)")
return srt_path
def _fmt_srt(sec):
h = int(sec // 3600); m = int((sec % 3600) // 60); s = int(sec % 60); ms = int((sec - int(sec)) * 1000)
return f"{h:02d}:{m:02d}:{s:02d},{ms:03d}"
def generate_bgm(out_dir, seconds=20, mood="calm ambient background music, gentle piano", use_stable=False):
"""生成 BGM默认 ffmpeg 氛围音稳定use_stable=True 时先试 stable_audio_3。"""
bgm_path = os.path.join(out_dir, "bgm.mp3")
ok = False
if use_stable:
ok = _try_stable_audio(bgm_path, seconds, mood)
if ok:
print(f" 🎵 BGM(stable_audio_3): {bgm_path} ({seconds}s)")
if not ok:
_ffmpeg_pad(bgm_path, seconds)
print(f" 🎵 BGM(ffmpeg氛围音): {bgm_path} ({seconds}s)")
return bgm_path
def _try_stable_audio(path, seconds, mood):
"""调本地 ComfyUI stable_audio_3 生成。返回 bool。"""
try:
import json, urllib.request, time
COMFY = "http://127.0.0.1:8188"
sigmas = ", ".join(str(round(1.0 - i / 24, 4)) for i in range(25)) + ", 0.0"
wf = {
"1": {"class_type": "CheckpointLoaderSimple", "inputs": {"ckpt_name": "stable_audio_3_medium_base.safetensors"}},
"2": {"class_type": "CLIPTextEncode", "inputs": {"text": mood, "clip": ["1", 1]}},
"3": {"class_type": "CLIPTextEncode", "inputs": {"text": "vocals, singing, noisy, harsh, distorted, speech", "clip": ["1", 1]}},
"4": {"class_type": "ConditioningStableAudio", "inputs": {"positive": ["2", 0], "negative": ["3", 0], "seconds_start": 0, "seconds_total": seconds}},
"5": {"class_type": "EmptyLatentAudio", "inputs": {"seconds": seconds, "batch_size": 1}},
"6": {"class_type": "KSamplerSelect", "inputs": {"sampler_name": "euler"}},
"9": {"class_type": "ManualSigmas", "inputs": {"sigmas": sigmas}},
"10": {"class_type": "SamplerCustom", "inputs": {"model": ["1", 0], "positive": ["4", 0], "negative": ["3", 0], "cfg": 4.0, "noise_seed": 42, "add_noise": True, "sampler": ["6", 0], "sigmas": ["9", 0], "latent_image": ["5", 0]}},
"11": {"class_type": "VAEDecodeAudio", "inputs": {"samples": ["10", 0], "vae": ["1", 2]}},
"12": {"class_type": "SaveAudio", "inputs": {"audio": ["11", 0], "filename_prefix": "eed_bgm_gen"}},
}
data = json.dumps({"prompt": wf}).encode()
req = urllib.request.Request(f"{COMFY}/prompt", data=data, headers={"Content-Type": "application/json"})
pid = json.loads(urllib.request.urlopen(req, timeout=15).read())["prompt_id"]
for _ in range(15): # 30秒内不成功即放弃避免卡住
time.sleep(2)
h = json.loads(urllib.request.urlopen(f"{COMFY}/history/{pid}", timeout=8).read())
if pid in h and h[pid].get("outputs"):
aud = h[pid]["outputs"].get("12", {}).get("audio", [])
if aud:
import shutil
shutil.copy(os.path.join(os.path.expanduser("~/comfy/ComfyUI/output"), aud[0]["filename"]), path)
return True
if pid in h and h[pid].get("status", {}).get("status_str") == "error":
return False
except Exception:
return False
return False
def _ffmpeg_pad(path, seconds):
"""ffmpeg 生成柔和氛围音垫底aevalsrc 单输入多频和声,避免多 lavfi 输入偶发失败)。"""
expr = (f"0.04*sin(2*PI*220*t)+0.04*sin(2*PI*277*t)+0.04*sin(2*PI*330*t)"
f"+0.03*sin(2*PI*165*t)")
cmd = ["ffmpeg", "-y", "-f", "lavfi",
"-i", f"aevalsrc={expr}:d={seconds}:s=44100",
"-af", "lowpass=f=1500,volume=0.35", path]
try:
r = subprocess.run(cmd, capture_output=True, timeout=60)
if r.returncode != 0:
print(f" ⚠️ BGM ffmpeg 退出码 {r.returncode}: {(r.stderr or '')[-200:]}")
except Exception as e:
print(f" ⚠️ BGM ffmpeg 异常: {e}")
def mix_final(video, voice_dir, bgm, out_path, subtitles=None):
"""ffmpeg 混音:成片(无声) + 配音 + BGM(压低)。"""
v = ["-i", video]
inputs = [video]
filters = []
n = 0
vf_in = {}
if voice_dir and os.path.isdir(voice_dir):
mps = sorted([os.path.join(voice_dir, f) for f in os.listdir(voice_dir) if f.endswith(".mp3")])
for mp in mps:
inputs.append(mp); vf_in[f"v{n+1}"] = mp
n += 1
if bgm and os.path.isfile(bgm):
inputs.append(bgm); vf_in[f"v{n+1}"] = bgm
n += 1
if not inputs[1:]:
# 无任何音轨:直接复制视频
subprocess.run(["ffmpeg", "-y", "-i", video, "-c", "copy", out_path], capture_output=True)
return out_path
# 简单混音:所有输入 amixBGM 音量压低
labels = [f"[{i}:a]" for i in range(1, len(inputs))]
vol = []
for i in range(1, len(inputs)):
if inputs[i] == bgm:
vol.append(f"[{i}:a]volume=0.15[v{i}]")
else:
vol.append(f"[{i}:a][v{i}]" if False else f"[{i}:a]volume=1.0[v{i}]")
fc = ";".join(vol) + ";" + "".join(f"[v{i}]" for i in range(1, len(inputs))) + f"amix=inputs={n}:normalize=0[aout]"
cmd = ["ffmpeg", "-y"] + sum([["-i", i] for i in inputs], []) + \
["-filter_complex", fc, "-map", "0:v", "-map", "[aout]", "-c:v", "copy", "-c:a", "aac", out_path]
try:
subprocess.run(cmd, capture_output=True, timeout=120)
except Exception as e:
print(" 混音失败:", e)
return out_path
if __name__ == "__main__":
if len(sys.argv) < 2:
print(__doc__); sys.exit(0)
cmd = sys.argv[1]
if cmd == "voice" and len(sys.argv) >= 4:
generate_voice(sys.argv[2], sys.argv[3])
elif cmd == "srt" and len(sys.argv) >= 4:
generate_srt(sys.argv[2], sys.argv[3])
elif cmd == "bgm" and len(sys.argv) >= 4:
generate_bgm(sys.argv[2], int(float(sys.argv[3])), sys.argv[4] if len(sys.argv) > 4 else "calm ambient background music, gentle piano")
elif cmd == "mix" and len(sys.argv) >= 5:
mix_final(sys.argv[2], sys.argv[3], sys.argv[4], sys.argv[5])
else:
print(__doc__)