| name | voiceover |
| description | 使用 edge-tts 生成多语言配音(中文/英文)。当需要为视频生成语音旁白、基于时间线同步配音时使用。支持语速调整、多种声音选择和配音验证。 |
配音生成技能
技术选型
| 方案 | 优点 | 缺点 |
|---|
| edge-tts | 免费、音质好、多语言支持 | 需要网络 |
| Azure TTS | 更多声音选择、更稳定 | 需要付费 |
推荐: edge-tts
声音选择
中文声音列表 (zh-CN)
| 声音 ID | 性别 | 风格 | 适用场景 |
|---|
| zh-CN-XiaoxiaoNeural | 女 | 温暖亲切 | 产品介绍、教程 |
| zh-CN-YunxiNeural | 男 | 专业稳重 | 企业宣传、正式场合 |
| zh-CN-YunjianNeural | 男 | 激情活力 | 科技发布、激励视频 |
| zh-CN-XiaoyiNeural | 女 | 年轻活泼 | 创意内容、轻松主题 |
| zh-CN-YunyangNeural | 男 | 新闻播报 | 资讯类、严肃主题 |
英文声音列表 (en-US)
| 声音 ID | 性别 | 风格 | 适用场景 |
|---|
| en-US-GuyNeural | 男 | 专业稳重 | 企业宣传、产品介绍 |
| en-US-JennyNeural | 女 | 温暖友好 | 教程、客户服务 |
| en-US-AriaNeural | 女 | 清晰专业 | 新闻、正式场合 |
| en-US-DavisNeural | 男 | 年轻活力 | 科技内容、创意视频 |
| en-US-JasonNeural | 男 | 激情活力 | 发布会、激励视频 |
| en-US-SaraNeural | 女 | 年轻活泼 | 社交媒体、轻松主题 |
声音选择建议
中文视频:
| 视频类型 | 推荐声音 |
|---|
| 产品演示 | XiaoxiaoNeural (女) |
| 公司介绍 | YunxiNeural (男) 或 XiaoxiaoNeural |
| 科技历程 | YunjianNeural (男) - 有激情感 |
| 教程类 | XiaoxiaoNeural (女) |
| 发布会风格 | YunjianNeural (男) |
英文视频:
| 视频类型 | 推荐声音 |
|---|
| 产品演示 | GuyNeural (男) 或 JennyNeural (女) |
| 公司介绍 | GuyNeural (男) 或 AriaNeural (女) |
| 科技历程 | JasonNeural (男) - 有激情感 |
| 教程类 | JennyNeural (女) |
| 发布会风格 | JasonNeural (男) 或 DavisNeural (男) |
时间线计算
核心公式
DEMO_START = OPENING_DURATION + FEATURES_DURATION
final_time = recording_time + DEMO_START
图文视频时间线
对于图文展示型视频,时间线由分镜直接定义:
SCENES = {
"opening": {"start": 0, "duration": 8},
"scene_1": {"start": 8, "duration": 14},
"scene_2": {"start": 22, "duration": 16},
}
VOICEOVER_SEGMENTS = [
(0.5, 7.5, "片头配音..."),
(8.5, 21.5, "场景1配音..."),
(22.5, 37.5, "场景2配音..."),
]
配音验证机制 (重要)
自动化验证函数
def validate_voiceover(segments, total_duration):
"""
验证配音时间线
返回: (是否通过, 问题列表)
"""
issues = []
for i, seg in enumerate(segments):
actual_end = seg["start_time"] + seg["actual_duration"]
if seg["actual_duration"] > seg["target_duration"] + 0.5:
issues.append({
"type": "duration_exceeded",
"segment": i,
"message": f"片段{i}: 实际({seg['actual_duration']:.1f}s) > 目标({seg['target_duration']:.1f}s)",
"severity": "warning"
})
if i < len(segments) - 1:
next_start = segments[i+1]["start_time"]
if actual_end > next_start:
issues.append({
"type": "overlap",
"segment": i,
"message": f"片段{i}和{i+1}重叠: {actual_end:.1f}s > {next_start:.1f}s",
"severity": "error"
})
last_seg = segments[-1]
last_end = last_seg["start_time"] + last_seg["actual_duration"]
if last_end > total_duration + 1:
issues.append({
"type": "exceeds_video",
"message": f"配音结束({last_end:.1f}s) > 视频时长({total_duration}s)",
"severity": "error"
})
for i in range(len(segments) - 1):
current_end = segments[i]["start_time"] + segments[i]["actual_duration"]
next_start = segments[i+1]["start_time"]
gap = next_start - current_end
if gap > 3:
issues.append({
"type": "large_gap",
"segment": i,
"message": f"片段{i}和{i+1}之间有{gap:.1f}s空白",
"severity": "warning"
})
passed = not any(issue["severity"] == "error" for issue in issues)
return passed, issues
验证报告输出
def print_validation_report(segments, total_duration):
"""打印配音验证报告"""
passed, issues = validate_voiceover(segments, total_duration)
print("╔" + "═" * 58 + "╗")
print("║" + "配音验证报告".center(54) + "║")
print("╠" + "═" * 58 + "╣")
print("║ 片段 │ 开始 │ 目标时长 │ 实际时长 │ 状态 ║")
print("╠" + "═" * 58 + "╣")
for i, seg in enumerate(segments):
status = "✅ OK" if seg["actual_duration"] <= seg["target_duration"] + 0.5 else "⚠️ 超时"
print(f"║ {i:2d} │ {seg['start_time']:5.1f}s │ {seg['target_duration']:5.1f}s │ {seg['actual_duration']:5.1f}s │ {status:10s} ║")
print("╠" + "═" * 58 + "╣")
if passed:
print("║ ✅ 验证通过 ║")
else:
print("║ ❌ 验证失败,请检查以下问题: ║")
for issue in issues:
if issue["severity"] == "error":
print(f"║ ❌ {issue['message'][:50]:50s} ║")
print("╚" + "═" * 58 + "╝")
return passed
完整配音脚本模板 (V2 - 多语言版)
"""
配音生成脚本 V2 - 包含验证机制 + 多语言支持
"""
import asyncio
import subprocess
from pathlib import Path
import json
import re
LANGUAGE = "zh"
VOICE = "zh-CN-YunjianNeural"
OUTPUT_DIR = Path("public/audio")
TOTAL_DURATION = 85
VOICEOVER_SEGMENTS = [
(0.5, 7.5, "配音内容1"),
(8.5, 21.5, "配音内容2"),
]
def get_audio_duration(file_path):
"""获取音频时长"""
result = subprocess.run(
["ffprobe", "-v", "error", "-show_entries", "format=duration",
"-of", "default=noprint_wrappers=1:nokey=1", str(file_path)],
capture_output=True, text=True
)
return float(result.stdout.strip())
def validate_voiceover(segments, total_duration):
"""验证配音时间线"""
issues = []
for i, seg in enumerate(segments):
if seg["actual_duration"] > seg["target_duration"] + 0.5:
issues.append(f"⚠️ 片段{i}: 超时 {seg['actual_duration'] - seg['target_duration']:.1f}s")
if i < len(segments) - 1:
actual_end = seg["start_time"] + seg["actual_duration"]
next_start = segments[i+1]["start_time"]
if actual_end > next_start:
issues.append(f"❌ 片段{i}和{i+1}重叠")
return len([i for i in issues if i.startswith("❌")]) == 0, issues
def calculate_natural_duration(text, language):
"""计算文本自然朗读时长"""
if language == "zh":
char_count = len(re.sub(r'[^\u4e00-\u9fff]', '', text))
return char_count / 4.0
else:
word_count = len(text.split())
return word_count / 2.5
async def generate_segment(index, start, end, text):
"""生成单个配音片段(支持多语言)"""
import edge_tts
output_file = OUTPUT_DIR / f"vo_{index:02d}.mp3"
duration_target = end - start
natural_duration = calculate_natural_duration(text, LANGUAGE)
if natural_duration > duration_target:
rate_adjust = min(35, int((natural_duration / duration_target - 1) * 100))
rate = f"+{rate_adjust}%"
elif natural_duration < duration_target * 0.7:
rate_adjust = min(15, int((1 - natural_duration / duration_target) * 50))
rate = f"-{rate_adjust}%"
else:
rate = "+0%"
communicate = edge_tts.Communicate(text=text, voice=VOICE, rate=rate)
await communicate.save(str(output_file))
actual_duration = get_audio_duration(output_file)
return {
"index": index,
"file": output_file.name,
"start_time": start,
"target_duration": duration_target,
"actual_duration": actual_duration,
"text": text[:20] + "...",
"rate": rate,
"language": LANGUAGE,
}
def merge_audio(segments):
"""合并音频"""
filter_parts = []
inputs = []
for i, seg in enumerate(segments):
inputs.extend(["-i", str(OUTPUT_DIR / seg["file"])])
delay_ms = int(seg["start_time"] * 1000)
filter_parts.append(f"[{i}:a]adelay={delay_ms}|{delay_ms}[a{i}];")
mix_inputs = "".join([f"[a{i}]" for i in range(len(segments))])
filter_parts.append(f"{mix_inputs}amix=inputs={len(segments)}:duration=longest[out]")
output_file = OUTPUT_DIR / "synced_voiceover.mp3"
subprocess.run([