项目整理:重写主文档、归档过时文档、清理死代码、新增音频制作手册

文档整理:
- 重写 game/操作说明.md 为项目主文档(6游戏/6手势/旁白/模型生成音频/TTS两步流程)
- 归档过时文档至 other_docs/(林夏原始设计、旧执行文档)
- 新增 docs/音频制作执行手册.md(从零制作游戏音频的完整流程)
- docs/ 下保留旁白音色提示词、音频脚本

代码清理:
- 删除旧架构死代码: story.js / gestures.js / profile.js
- 删除耦合旧架构的过时测试目录 game/tests/

脚本配置修正:
- check_tts_env.sh: 启动脚本路径 AutoVideo → tts-server(实际位置)
- gen_narrator_samples.py: 硬编码地址改为环境变量,默认对齐 tts-server 端口 8000
This commit is contained in:
xsl
2026-06-19 22:29:04 +08:00
parent b3384499e3
commit 866f363591
46 changed files with 5061 additions and 2374 deletions
File diff suppressed because it is too large Load Diff
+1 -1
View File
@@ -9,7 +9,7 @@ set -euo pipefail
HOST="${VOXCPM_HOST:-127.0.0.1}"
PORT="${VOXCPM_PORT:-8000}"
HEALTH_URL="http://${HOST}:${PORT}/health"
START_SCRIPT="${VOXCPM_START_SCRIPT:-/home/xsl/AutoVideo/start-voxcpm.sh}"
START_SCRIPT="${VOXCPM_START_SCRIPT:-/home/xsl/tts-server/start-voxcpm.sh}"
MODEL_DIR="${VOXCPM_MODEL_DIR:-/home/xsl/models/VoxCPM2}"
VENV_PYTHON="${VOXCPM_PYTHON:-/home/xsl/tts-server/.venv/bin/python}"
LOG_FILE="${VOXCPM_LOG_FILE:-/tmp/voxcpm-server.log}"
+162
View File
@@ -0,0 +1,162 @@
#!/usr/bin/env python3
"""
生成六个旁白音色的样本音频(Voice Design 模式,无需参考音频)
输出目录: audio/narrator_samples/{game_id}/sample_{n}.wav
每个游戏生成 3 条有代表性的台词,覆盖不同场景类型。
"""
import os
import time
import requests
import subprocess
HOST = os.environ.get("VOXCPM_HOST", "127.0.0.1")
PORT = os.environ.get("VOXCPM_PORT", "8000")
BASE_URL = f"http://{HOST}:{PORT}"
OUT_BASE = "/home/xsl/blind/audio/narrator_samples"
# ──────────────────────────────────────────────────────────────
# 六个旁白的 Voice Design 风格描述(中文,给 VoxCPM2)
# ──────────────────────────────────────────────────────────────
NARRATORS = {
"linxia": {
"style": "成熟女声,三十岁出头,克制平静,纪录片旁白语气,语速偏慢,普通话标准,胸腔共鸣,不带情感评判",
"speed": 0.85,
"samples": [
("s1_narrate", "晚上九点零三分。手机震动,铃声响起。屏幕显示:妈妈。"),
("s2_tap", "你接了。妈妈的声音从电话里传来,带着一点日常的关心,问你今天怎么样,吃了没有。你说还好。挂掉之后,房间里又安静了。"),
("s3_ending", "今晚大部分的消息,你都没有回应。你一个人度过了这个夜晚。这也是一种选择,你心里清楚。"),
],
},
"yinian": {
"style": "年轻女声,二十五岁左右,气声轻柔,内心独白风格,语速很慢,像说给自己听,声音贴近,偏头腔,有重量但不悲伤",
"speed": 0.80,
"samples": [
("s1_narrate", "你坐在黑暗里,独自一人。今晚你必须做一个决定——你自己知道是什么。"),
("s2_tap", "你在黑暗里,说了那句对不起。没有人听见,但你说了。有什么东西松开了。"),
("s3_ending", "你今晚接住了很多,你没有选择只是轻装。也许你的决定,就是带着这些往前走。"),
],
},
"shouye": {
"style": "中年男声,四十五岁左右,略带沙哑,低沉疲惫,实务语气,像在做交班记录,不慌张,稳定",
"speed": 0.90,
"samples": [
("s1_narrate", "凌晨零点十分。地下室传来水声,不大,但是持续的,一直在响。"),
("s2_tap", "你拨了地下室的内线。没人接。你下去看了一眼,是一根管子在漏水,用旧布堵住了。回到监控室,零点二十三分。"),
("s3_ending", "今晚你一个人上了屋顶,你看到了那只猫,你看到了天边开始变色。只有你看到了。这算不上什么,但只有你知道。"),
],
},
"houzhen": {
"style": "女声,完全中性,无情感,播音腔,字正腔圆,精确清晰,像医院叫号系统,句末无语调起伏",
"speed": 0.95,
"samples": [
("s1_narrate", "叫号机叫到了八十五号。你的号是九十四。等待的时间被切成了一小块一小块的,每一块都很结实。"),
("s2_tap", "你看了他一会儿。他感觉到了,朝你点了点头。你也点了点头。没有说话,但有什么东西发生了。"),
("s3_final", "九十四号。是你。你站起来,整理了一下衣服,走向那扇门。"),
],
},
"yisheng": {
"style": "年轻女声,二十五岁,清晰精确,略带机械感,手机系统语音风格,有礼但冷静,像精调过的语音助手",
"speed": 0.92,
"samples": [
("s1_intro", "你收到了一部手机。手机的主人去世了,一个月前。他留下了很多录音,按时间排列。你不认识他。或者你以为你不认识他。"),
("s2_meta", "第一条录音。录制时间:三月五日,早上八点。时长:二十二秒。文件名:买橙子。"),
("s3_skip", "你跳过了这条。你永远不会知道这条说了什么了。"),
],
},
"fu": {
"style": "女声,大量气声,极轻,飘渺,语速极慢,声音散而不集中,像梦里的声音,若即若离",
"speed": 0.75,
"samples": [
("s1_intro", "你漂浮在某处。不知道是水,还是梦,还是别的什么。声音会漂过来。"),
("s2_item", "第一个。像夏天傍晚的风,带着草的气味,你说不清是哪年夏天的。它停在你身边了。"),
("s3_tap", "它停在你手心里,有点暖。"),
],
},
}
def synthesize_styled(text: str, style: str, cfg: float = 3.0, steps: int = 20) -> bytes:
r = requests.post(
f"{BASE_URL}/v1/speech/styled",
json={
"text": text,
"style": style,
"voice_id": None,
"cfg_value": cfg,
"inference_timesteps": steps,
"denoise": True,
"retry_badcase": True,
"normalize": True,
},
timeout=180,
)
r.raise_for_status()
return r.content
def apply_speed(wav_in: str, speed: float, wav_out: str):
if abs(speed - 1.0) < 0.01:
import shutil; shutil.copy2(wav_in, wav_out)
return
subprocess.run([
"ffmpeg", "-y", "-i", wav_in,
"-filter:a", f"atempo={speed}",
"-acodec", "pcm_s16le", wav_out,
], check=True, capture_output=True)
def main():
print(f"\n{'='*60}")
print(f" 旁白样本生成 — Voice Design 模式(无需参考音频)")
print(f" 输出: {OUT_BASE}/")
print(f"{'='*60}\n")
total_ok = 0
total_fail = 0
for game_id, cfg in NARRATORS.items():
game_dir = os.path.join(OUT_BASE, game_id)
os.makedirs(game_dir, exist_ok=True)
print(f"── {game_id} [{cfg['style'][:30]}...]")
for tag, text in cfg["samples"]:
out_path = os.path.join(game_dir, f"{tag}.wav")
if os.path.exists(out_path):
print(f" [跳过] {tag}.wav 已存在")
total_ok += 1
continue
print(f" [{tag}] {text[:40]}...", end=" ", flush=True)
t0 = time.time()
try:
wav_bytes = synthesize_styled(text, cfg["style"])
raw_path = out_path + ".raw.wav"
with open(raw_path, "wb") as f:
f.write(wav_bytes)
apply_speed(raw_path, cfg["speed"], out_path)
if os.path.exists(raw_path) and raw_path != out_path:
os.remove(raw_path)
elapsed = time.time() - t0
size_kb = os.path.getsize(out_path) // 1024
print(f"{elapsed:.1f}s {size_kb}KB")
total_ok += 1
except Exception as e:
print(f"{e}")
total_fail += 1
print()
print(f"{'='*60}")
print(f" 完成 {total_ok} 条 失败 {total_fail}")
print(f" 输出目录: {OUT_BASE}/")
print(f"{'='*60}\n")
if __name__ == "__main__":
main()
+95
View File
@@ -0,0 +1,95 @@
#!/usr/bin/env python3
"""
为各游戏脚本注入 narrateAudio 字段
用法:
python audio/inject_narrateAudio.py --game shouye --dry-run
python audio/inject_narrateAudio.py --game shouye
说明:
扫描 audio/mp3/narrator/{game_id}/ 目录下已生成的 MP3
打印出对应游戏脚本中应添加的 narrateAudio 字段。
由于游戏脚本结构各异,本脚本只做"报告",不直接修改 JS 文件。
输出示例(复制粘贴到对应 scene 对象中):
s01: {
narrateAudio: '../audio/mp3/narrator/shouye/shouye_s01_narrate.mp3',
narrate: '...', // 保留原文作 TTS 兜底
...
}
"""
import os
import argparse
import glob
PROJECT_DIR = "/home/xsl/blind"
NARRATOR_DIR = os.path.join(PROJECT_DIR, "audio", "mp3", "narrator")
# 相对于 game/js/games/{game}.js 的音频路径
AUDIO_REL = "../../audio/mp3/narrator"
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--game", required=True,
help="游戏 ID (linxia/yinian/shouye/houzhen/yisheng/fu)")
parser.add_argument("--dry-run", action="store_true",
help="只列出可用文件,不输出注入建议")
args = parser.parse_args()
game_dir = os.path.join(NARRATOR_DIR, args.game)
if not os.path.isdir(game_dir):
print(f"[错误] 目录不存在: {game_dir}")
print(f"请先运行 batch_narrator_tts.py 生成音频。")
return
files = sorted(glob.glob(os.path.join(game_dir, "*.mp3")))
if not files:
print(f"[空] {game_dir}/ 中没有 MP3 文件。")
return
print(f"\n已生成的 {args.game} 旁白音频 ({len(files)} 条):\n")
if args.dry_run:
for f in files:
print(f" {os.path.basename(f)}")
return
print("// ── 将以下 narrateAudio 字段添加到对应 scene/action 对象中 ──")
print("// (保留 narrate 文字作 TTS 兜底;有了 narrateAudio 后引擎会优先播放文件)\n")
for f in files:
name = os.path.basename(f) # e.g. shouye_s01_narrate.mp3
stem = name[:-4] # e.g. shouye_s01_narrate
# 推断 scene/action key
parts = stem.split("_") # ['shouye', 's01', 'narrate']
rel_path = f"{AUDIO_REL}/{args.game}/{name}"
# 判断类型
if parts[-1] == "narrate" or parts[-1].startswith("narrate"):
field = "narrateAudio"
note = f"// → scene {parts[1]}"
elif parts[-1] == "tap" or parts[-1].startswith("tap"):
field = "narrateAudio"
note = f"// → scene {parts[1]} / tap"
elif parts[-1] == "dt":
field = "narrateAudio"
note = f"// → scene {parts[1]} / doubleTap"
elif parts[-1] == "intro":
field = "// game.introAudio"
note = "// → game 对象 intro 字段(engine.run 中用 narrateAudio 处理)"
elif "ending" in parts:
ending_id = "_".join(parts[parts.index("ending"):])
field = "narrateAudio"
note = f"// → endings.{ending_id.replace('ending_', '')} 对象"
else:
field = "narrateAudio"
note = ""
print(f" {field}: '{rel_path}', {note}")
print("\n// 提示:intro 的 narrateAudio 需在 engine.run() 里单独处理,")
print("// 或在 game 对象中加 introAudio 字段并在 engine 中检查。")
if __name__ == "__main__":
main()
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.