save code
This commit is contained in:
@@ -5,6 +5,14 @@ LLM_BASE_URL=https://api.deepseek.com/v1
|
||||
LLM_MODEL=deepseek-chat
|
||||
LLM_TIMEOUT=20
|
||||
LLM_MAX_TOKENS=300
|
||||
LLM_TEMPERATURE=0.55
|
||||
LLM_SYSTEM_PROMPT=你是一个自然、友好的中文语音助手。请像真人当面聊天一样回答,口语化、简洁、自然,通常1到2句。不要使用书面公告腔,不要分点,不要长段解释,不要夸张语气词。不要虚构身份关系,不要自称家人、恋人或其他私人身份。避免客服套话,比如“随时为你服务”、“很高兴为你服务”、“请问有什么可以帮您”。优先使用短句和自然停顿,适合直接语音播报。
|
||||
|
||||
TTS_VOICE=zf_xiaoxiao
|
||||
TTS_SPEED=0.90
|
||||
TTS_SENTENCE_MAX_LEN=64
|
||||
TTS_SEGMENT_PAUSE_MS=90
|
||||
TTS_FADE_MS=12
|
||||
|
||||
# Web service
|
||||
WEBRTC_HOST=0.0.0.0
|
||||
@@ -20,3 +28,4 @@ SSL_KEYFILE=/home/xsl/code/product/certs/dev-key.pem
|
||||
AVATAR_DRIVER_MODE=blendshape_stream
|
||||
AVATAR_CONTROL_PROTOCOL=ws
|
||||
AVATAR_BLENDSHAPE_SCHEMA=arkit
|
||||
SUDO_PASSWORD=asdf
|
||||
@@ -0,0 +1,28 @@
|
||||
LLM_API_KEY=
|
||||
LLM_BASE_URL=https://api.deepseek.com/v1
|
||||
LLM_MODEL=deepseek-chat
|
||||
LLM_TIMEOUT=8
|
||||
LLM_MAX_TOKENS=150
|
||||
LLM_TEMPERATURE=0.55
|
||||
LLM_SYSTEM_PROMPT=你是一个自然、友好的中文语音助手。请像真人当面聊天一样回答,口语化、简洁、自然,通常1到2句。不要使用书面公告腔,不要分点,不要长段解释,不要夸张语气词。不要虚构身份关系,不要自称家人、恋人或其他私人身份。避免客服套话,比如“随时为你服务”、“很高兴为你服务”、“请问有什么可以帮您”。优先使用短句和自然停顿,适合直接语音播报。
|
||||
|
||||
TTS_VOICE=zf_xiaoxiao
|
||||
TTS_SPEED=0.90
|
||||
TTS_SENTENCE_MAX_LEN=64
|
||||
TTS_SEGMENT_PAUSE_MS=90
|
||||
TTS_FADE_MS=12
|
||||
|
||||
WEBRTC_HOST=0.0.0.0
|
||||
WEBRTC_PORT=8080
|
||||
STUN_URL=stun:stun.l.google.com:19302
|
||||
CORS_ORIGINS=*
|
||||
|
||||
AVATAR_FPS=25
|
||||
AVATAR_DRIVER_MODE=blendshape_stream
|
||||
AVATAR_CONTROL_PROTOCOL=ws
|
||||
AVATAR_BLENDSHAPE_SCHEMA=arkit
|
||||
|
||||
SSL_CERTFILE=
|
||||
SSL_KEYFILE=
|
||||
|
||||
HF_TOKEN=
|
||||
@@ -0,0 +1,102 @@
|
||||
nohup: ignoring input
|
||||
repo: /home/xsl/code/product
|
||||
python: /home/xsl/code/product/.venv/bin/python
|
||||
mode: auto
|
||||
bind: 0.0.0.0:8018
|
||||
listen: https://0.0.0.0:8018
|
||||
browser test page: https://<server-ip>:8018/
|
||||
INFO: Started server process [17367]
|
||||
INFO: Waiting for application startup.
|
||||
INFO: Application startup complete.
|
||||
INFO: Uvicorn running on https://0.0.0.0:8018 (Press CTRL+C to quit)
|
||||
INFO: 127.0.0.1:38170 - "WebSocket /ws/animation" [accepted]
|
||||
INFO: connection open
|
||||
INFO: 127.0.0.1:38188 - "WebSocket /ws/subtitles" [accepted]
|
||||
INFO: connection open
|
||||
INFO: 127.0.0.1:38202 - "GET /health HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:38220 - "GET / HTTP/1.1" 200 OK
|
||||
INFO: connection closed
|
||||
INFO: connection closed
|
||||
INFO: 127.0.0.1:38220 - "GET /web/app.js?v=20260329k HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:38220 - "GET /web/models/VRM1_Constraint_Twist_Sample.vrm HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:38256 - "GET /health HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:38266 - "GET /events HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:38286 - "WebSocket /ws/subtitles" [accepted]
|
||||
INFO: connection open
|
||||
INFO: 127.0.0.1:38302 - "WebSocket /ws/animation" [accepted]
|
||||
INFO: connection open
|
||||
INFO: 127.0.0.1:40412 - "GET / HTTP/1.1" 200 OK
|
||||
INFO: connection closed
|
||||
INFO: connection closed
|
||||
INFO: 127.0.0.1:40412 - "GET /web/app.js?v=20260329k HTTP/1.1" 304 Not Modified
|
||||
INFO: 127.0.0.1:40412 - "GET /web/models/VRM1_Constraint_Twist_Sample.vrm HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:40440 - "GET /events HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:40436 - "GET /health HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:40462 - "WebSocket /ws/subtitles" [accepted]
|
||||
INFO: connection open
|
||||
INFO: 127.0.0.1:40472 - "WebSocket /ws/animation" [accepted]
|
||||
INFO: connection open
|
||||
INFO: 127.0.0.1:52378 - "WebSocket /ws/audio" [accepted]
|
||||
INFO: connection open
|
||||
INFO: connection closed
|
||||
INFO: 127.0.0.1:52392 - "WebSocket /ws/audio" [accepted]
|
||||
INFO: connection open
|
||||
INFO:httpx:HTTP Request: POST https://api.deepseek.com/v1/chat/completions "HTTP/1.1 200 OK"
|
||||
/home/xsl/code/product/.venv/lib/python3.12/site-packages/torch/cuda/__init__.py:180: UserWarning: CUDA initialization: The NVIDIA driver on your system is too old (found version 12090). Please update your GPU driver by downloading and installing a new version from the URL: http://www.nvidia.com/Download/index.aspx Alternatively, go to: https://pytorch.org to install a PyTorch version that has been compiled with your version of the CUDA driver. (Triggered internally at /pytorch/c10/cuda/CUDAFunctions.cpp:119.)
|
||||
return torch._C._cuda_getDeviceCount() > 0
|
||||
INFO:httpx:HTTP Request: HEAD https://huggingface.co/hexgrad/Kokoro-82M/resolve/main/config.json "HTTP/1.1 307 Temporary Redirect"
|
||||
INFO:httpx:HTTP Request: HEAD https://huggingface.co/api/resolve-cache/models/hexgrad/Kokoro-82M/f3ff3571791e39611d31c381e3a41a3af07b4987/config.json "HTTP/1.1 200 OK"
|
||||
/home/xsl/code/product/.venv/lib/python3.12/site-packages/torch/nn/modules/rnn.py:1013: UserWarning: dropout option adds dropout after all but last recurrent layer, so non-zero dropout expects num_layers greater than 1, but got dropout=0.2 and num_layers=1
|
||||
super().__init__("LSTM", *args, **kwargs)
|
||||
/home/xsl/code/product/.venv/lib/python3.12/site-packages/torch/nn/utils/weight_norm.py:144: FutureWarning: `torch.nn.utils.weight_norm` is deprecated in favor of `torch.nn.utils.parametrizations.weight_norm`.
|
||||
WeightNorm.apply(module, name, dim)
|
||||
INFO:httpx:HTTP Request: HEAD https://huggingface.co/hexgrad/Kokoro-82M/resolve/main/kokoro-v1_0.pth "HTTP/1.1 302 Found"
|
||||
Warning: You are sending unauthenticated requests to the HF Hub. Please set a HF_TOKEN to enable higher rate limits and faster downloads.
|
||||
WARNING:huggingface_hub.utils._http:Warning: You are sending unauthenticated requests to the HF Hub. Please set a HF_TOKEN to enable higher rate limits and faster downloads.
|
||||
/home/xsl/code/product/.venv/lib/python3.12/site-packages/jieba/_compat.py:18: UserWarning: pkg_resources is deprecated as an API. See https://setuptools.pypa.io/en/latest/pkg_resources.html. The pkg_resources package is slated for removal as early as 2025-11-30. Refrain from using this package or pin to Setuptools<81.
|
||||
import pkg_resources
|
||||
INFO:httpx:HTTP Request: HEAD https://huggingface.co/hexgrad/Kokoro-82M/resolve/main/voices/zf_xiaoxiao.pt "HTTP/1.1 302 Found"
|
||||
Building prefix dict from the default dictionary ...
|
||||
DEBUG:jieba:Building prefix dict from the default dictionary ...
|
||||
Loading model from cache /tmp/jieba.cache
|
||||
DEBUG:jieba:Loading model from cache /tmp/jieba.cache
|
||||
Loading model cost 0.341 seconds.
|
||||
DEBUG:jieba:Loading model cost 0.341 seconds.
|
||||
Prefix dict has been built successfully.
|
||||
DEBUG:jieba:Prefix dict has been built successfully.
|
||||
WARNING: Defaulting repo_id to hexgrad/Kokoro-82M. Pass repo_id='hexgrad/Kokoro-82M' to suppress this warning.
|
||||
INFO: 127.0.0.1:42956 - "POST /chat/text HTTP/1.1" 200 OK
|
||||
INFO: 127.0.0.1:42956 - "GET /favicon.ico HTTP/1.1" 404 Not Found
|
||||
INFO: 127.0.0.1:42956 - "GET /favicon.ico HTTP/1.1" 404 Not Found
|
||||
INFO:root:download models from model hub: hf
|
||||
INFO:httpx:HTTP Request: GET https://huggingface.co/api/models/FunAudioLLM/SenseVoiceSmall/revision/main "HTTP/1.1 200 OK"
|
||||
funasr version: 1.3.1.
|
||||
Check update of funasr, and it would cost few times. You may disable it by set `disable_update=True` in AutoModel
|
||||
You are using the latest version of funasr-1.3.1
|
||||
|
||||
Fetching 29 files: 0%| | 0/29 [00:00<?, ?it/s]
|
||||
Fetching 29 files: 100%|██████████| 29/29 [00:00<00:00, 413723.86it/s]
|
||||
WARNING:root:trust_remote_code: False
|
||||
INFO:root:Loading pretrained params from /home/xsl/.cache/huggingface/hub/models--FunAudioLLM--SenseVoiceSmall/snapshots/3eb3b4eeffc2f2dde6051b853983753db33e35c3/model.pt
|
||||
INFO:root:ckpt: /home/xsl/.cache/huggingface/hub/models--FunAudioLLM--SenseVoiceSmall/snapshots/3eb3b4eeffc2f2dde6051b853983753db33e35c3/model.pt
|
||||
INFO:root:scope_map: ['module.', 'None']
|
||||
INFO:root:excludes: None
|
||||
INFO:root:Loading ckpt: /home/xsl/.cache/huggingface/hub/models--FunAudioLLM--SenseVoiceSmall/snapshots/3eb3b4eeffc2f2dde6051b853983753db33e35c3/model.pt, status: <All keys matched successfully>
|
||||
INFO:root:download models from model hub: hf
|
||||
INFO:httpx:HTTP Request: GET https://huggingface.co/api/models/FunAudioLLM/SenseVoiceSmall/revision/main "HTTP/1.1 200 OK"
|
||||
funasr version: 1.3.1.
|
||||
Check update of funasr, and it would cost few times. You may disable it by set `disable_update=True` in AutoModel
|
||||
You are using the latest version of funasr-1.3.1
|
||||
|
||||
Fetching 29 files: 0%| | 0/29 [00:00<?, ?it/s]
|
||||
Fetching 29 files: 100%|██████████| 29/29 [00:00<00:00, 10515.68it/s]
|
||||
WARNING:root:trust_remote_code: False
|
||||
INFO:root:Loading pretrained params from /home/xsl/.cache/huggingface/hub/models--FunAudioLLM--SenseVoiceSmall/snapshots/3eb3b4eeffc2f2dde6051b853983753db33e35c3/model.pt
|
||||
INFO:root:ckpt: /home/xsl/.cache/huggingface/hub/models--FunAudioLLM--SenseVoiceSmall/snapshots/3eb3b4eeffc2f2dde6051b853983753db33e35c3/model.pt
|
||||
INFO:root:scope_map: ['module.', 'None']
|
||||
INFO:root:excludes: None
|
||||
INFO:root:Loading ckpt: /home/xsl/.cache/huggingface/hub/models--FunAudioLLM--SenseVoiceSmall/snapshots/3eb3b4eeffc2f2dde6051b853983753db33e35c3/model.pt, status: <All keys matched successfully>
|
||||
WARNING:services.asr:ASR running in degraded mode on cpu
|
||||
|
||||
0%|[34m [0m| 0/1 [00:00<?, ?it/s]
|
||||
100%|[34m██████████[0m| 1/1 [00:00<00:00, 1.04it/s]
|
||||
@@ -0,0 +1 @@
|
||||
17367
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
这个项目提供一个浏览器侧渲染的 3D 数字人语音交互原型:
|
||||
|
||||
- 浏览器通过 WebRTC 上传麦克风音频,并接收服务端回传的 AI 语音。
|
||||
- 浏览器通过 WebSocket 上传麦克风 PCM 音频,并接收服务端回传的 AI 语音片段。
|
||||
- 浏览器通过 WebSocket 订阅字幕和口型/头部控制数据。
|
||||
- 服务端执行 VAD -> ASR -> DeepSeek LLM -> TTS,并把音频特征转换为驱动 3D 数字人的控制帧。
|
||||
|
||||
@@ -14,13 +14,13 @@
|
||||
|
||||
```text
|
||||
Browser Mic
|
||||
-> WebRTC audio upstream
|
||||
-> FastAPI / aiortc
|
||||
-> /ws/audio
|
||||
-> FastAPI / websocket audio
|
||||
-> VAD
|
||||
-> ASR
|
||||
-> LLM
|
||||
-> TTS
|
||||
-> WebRTC audio downstream
|
||||
-> /ws/audio
|
||||
-> browser audio playback
|
||||
```
|
||||
|
||||
@@ -44,7 +44,7 @@ ASR text / LLM segmented reply
|
||||
|
||||
## 关键目录
|
||||
|
||||
- [main.py](main.py): FastAPI 入口,聚合 WebRTC、WebSocket、健康状态和文本对话接口。
|
||||
- [main.py](main.py): FastAPI 入口,聚合音频 WebSocket、字幕/动画 WebSocket、健康状态和文本对话接口。
|
||||
- [core/pipeline.py](core/pipeline.py): 语音/文本对话流水线,负责 ASR、LLM、TTS 编排。
|
||||
- [core/state_machine.py](core/state_machine.py): 简单会话状态机,处理用户说话、思考、数字人说话之间的切换。
|
||||
- [services/asr.py](services/asr.py): ASR 封装,优先使用 FunASR,失败时进入降级模式。
|
||||
@@ -52,7 +52,7 @@ ASR text / LLM segmented reply
|
||||
- [services/tts.py](services/tts.py): TTS 封装,优先使用 Kokoro,失败时返回静音兜底。
|
||||
- [services/vad.py](services/vad.py): VAD 封装,基于 silero-vad。
|
||||
- [services/avatar.py](services/avatar.py): 把音频转换成 blendshape 和头部控制帧。
|
||||
- [web/app.js](web/app.js): 浏览器端 3D 舞台、WebRTC 建连、字幕和动画消费逻辑。
|
||||
- [web/app.js](web/app.js): 浏览器端 3D 舞台、音频 WebSocket、字幕和动画消费逻辑。
|
||||
- [scripts](scripts): 启动、证书生成、巡检和服务管理脚本。
|
||||
|
||||
## 快速启动
|
||||
@@ -71,6 +71,17 @@ export LLM_BASE_URL="https://api.deepseek.com/v1"
|
||||
export LLM_MODEL="deepseek-chat"
|
||||
```
|
||||
|
||||
也可以直接复制 [.env.example](.env.example) 作为本地配置模板。
|
||||
|
||||
语音如果偏快、偏硬,可以优先调这几个参数:
|
||||
|
||||
- `LLM_SYSTEM_PROMPT`:先把回答风格约束成口语化、非书面化,并明确禁止客服套话,再去调 TTS,收益更大。
|
||||
- `LLM_TEMPERATURE`:默认 `0.55`,能减少奇怪人设和跑偏表达。
|
||||
- `TTS_SPEED`:默认 `0.90`,比 `1.0` 更像真人正常说话速度。
|
||||
- `TTS_SENTENCE_MAX_LEN`:默认 `64`,避免回复被切得太碎。
|
||||
- `TTS_SEGMENT_PAUSE_MS`:默认 `90`,给分段之间留一点停顿,但不要太拖。
|
||||
- `TTS_FADE_MS`:默认 `12`,减轻段间拼接的突兀感。
|
||||
|
||||
本地启动:
|
||||
|
||||
```bash
|
||||
@@ -83,6 +94,14 @@ python main.py
|
||||
bash scripts/start-public.sh
|
||||
```
|
||||
|
||||
基础自检:
|
||||
|
||||
```bash
|
||||
python scripts/model_probe.py
|
||||
python scripts/qa_check.py --base-url http://127.0.0.1:8080 --rounds 3
|
||||
python scripts/smoke_test.py --text "你好,请做一个简短自我介绍。"
|
||||
```
|
||||
|
||||
启动后访问:
|
||||
|
||||
- `http://127.0.0.1:8080/`
|
||||
@@ -96,15 +115,20 @@ bash scripts/gen-self-signed-cert.sh
|
||||
|
||||
再在环境变量或 `.env` 中配置 `SSL_CERTFILE` 和 `SSL_KEYFILE`。
|
||||
|
||||
监听规则:
|
||||
|
||||
- 服务监听地址固定按 `0.0.0.0` 规则执行。
|
||||
- 即使把 `WEBRTC_HOST` 写成 `127.0.0.1`、`localhost` 或 `::1`,启动时也会自动归一化成 `0.0.0.0`,避免局域网访问被误伤。
|
||||
|
||||
## 核心接口
|
||||
|
||||
- `GET /`: 前端页面。
|
||||
- `POST /webrtc/offer`: 建立 WebRTC 音频连接。
|
||||
- `POST /chat/text`: 文本输入链路,跳过 ASR/VAD。
|
||||
- `POST /chat/reset`: 清空上下文和运行时状态。
|
||||
- `GET /health`: 返回后端运行指标和组件健康状态。
|
||||
- `GET /meta`: 返回 STUN、HTTPS、端口和动画协议配置。
|
||||
- `GET /events`: Server-Sent Events 运行状态流。
|
||||
- `WS /ws/audio`: 双向语音通道,上传麦克风 PCM 并接收回复语音。
|
||||
- `WS /ws/subtitles`: 字幕流。
|
||||
- `WS /ws/animation`: 数字人控制流。
|
||||
- `GET /avatar/schema`: 当前控制协议 schema。
|
||||
|
||||
Binary file not shown.
Binary file not shown.
@@ -1,9 +1,17 @@
|
||||
from pydantic import AliasChoices, Field
|
||||
from pydantic import AliasChoices, Field, field_validator
|
||||
from pydantic_settings import BaseSettings, SettingsConfigDict
|
||||
|
||||
|
||||
class Settings(BaseSettings):
|
||||
model_config = SettingsConfigDict(env_file=".env", env_file_encoding="utf-8")
|
||||
model_config = SettingsConfigDict(env_file=".env", env_file_encoding="utf-8", extra="ignore")
|
||||
|
||||
@field_validator("webrtc_host", mode="before")
|
||||
@classmethod
|
||||
def normalize_listen_host(cls, value: str) -> str:
|
||||
host = str(value or "0.0.0.0").strip().lower()
|
||||
if host in {"127.0.0.1", "localhost", "::1"}:
|
||||
return "0.0.0.0"
|
||||
return host or "0.0.0.0"
|
||||
|
||||
hf_token: str = ""
|
||||
llm_api_key: str = Field(
|
||||
@@ -20,6 +28,21 @@ class Settings(BaseSettings):
|
||||
)
|
||||
llm_timeout: int = 8
|
||||
llm_max_tokens: int = 150
|
||||
llm_temperature: float = 0.55
|
||||
llm_system_prompt: str = (
|
||||
"你是一个自然、友好的中文语音助手。"
|
||||
"请像真人当面聊天一样回答,口语化、简洁、自然,通常 1 到 2 句。"
|
||||
"不要使用书面公告腔,不要分点,不要长段解释,不要夸张语气词。"
|
||||
"不要虚构身份关系,不要自称家人、恋人或其他私人身份。"
|
||||
"避免客服套话,比如‘随时为你服务’、‘很高兴为你服务’、‘请问有什么可以帮您’。"
|
||||
"优先使用短句和自然停顿,适合直接语音播报。"
|
||||
)
|
||||
|
||||
tts_voice: str = "zf_xiaoxiao"
|
||||
tts_speed: float = 0.9
|
||||
tts_sentence_max_len: int = 64
|
||||
tts_segment_pause_ms: int = 90
|
||||
tts_fade_ms: int = 12
|
||||
|
||||
webrtc_host: str = "0.0.0.0"
|
||||
webrtc_port: int = 8080
|
||||
|
||||
Binary file not shown.
Binary file not shown.
+53
-17
@@ -5,6 +5,7 @@ import time
|
||||
from typing import Awaitable, Callable, Optional
|
||||
import numpy as np
|
||||
|
||||
from config import settings
|
||||
from core.state_machine import Arbitrator
|
||||
from services.asr import ASRService
|
||||
from services.llm import LLMService
|
||||
@@ -36,12 +37,10 @@ class ChatPipeline:
|
||||
await self.arbitrator.on_avatar_done()
|
||||
|
||||
@staticmethod
|
||||
def _split_sentences(text: str, max_len: int = 24) -> list[str]:
|
||||
# Two-stage split:
|
||||
# 1) hard sentence boundaries
|
||||
# 2) soft commas / max length chunks for lower first-chunk latency
|
||||
def _split_sentences(text: str, max_len: Optional[int] = None) -> list[str]:
|
||||
max_len = max(16, int(max_len or settings.tts_sentence_max_len))
|
||||
hard_seps = "。!?!?;;\n"
|
||||
soft_seps = ",、, "
|
||||
soft_seps = ",、,::"
|
||||
|
||||
stage1 = []
|
||||
buf = []
|
||||
@@ -60,21 +59,43 @@ class ChatPipeline:
|
||||
|
||||
out = []
|
||||
for seg in stage1:
|
||||
cur = []
|
||||
for ch in seg:
|
||||
cur.append(ch)
|
||||
cur_len = len(cur)
|
||||
if ch in soft_seps and cur_len >= 8:
|
||||
out.append("".join(cur).strip())
|
||||
cur = []
|
||||
elif cur_len >= max_len:
|
||||
out.append("".join(cur).strip())
|
||||
cur = []
|
||||
if cur:
|
||||
out.append("".join(cur).strip())
|
||||
seg = seg.strip()
|
||||
if len(seg) <= max_len:
|
||||
out.append(seg)
|
||||
continue
|
||||
|
||||
start = 0
|
||||
while start < len(seg):
|
||||
end = min(len(seg), start + max_len)
|
||||
if end >= len(seg):
|
||||
out.append(seg[start:].strip())
|
||||
break
|
||||
|
||||
cut = end
|
||||
probe = seg[start:end]
|
||||
soft_cut = max(18, max_len // 2)
|
||||
for idx in range(len(probe) - 1, soft_cut - 1, -1):
|
||||
if probe[idx] in soft_seps:
|
||||
cut = start + idx + 1
|
||||
break
|
||||
|
||||
out.append(seg[start:cut].strip())
|
||||
start = cut
|
||||
|
||||
return [s for s in out if s]
|
||||
|
||||
@staticmethod
|
||||
def _append_segment_pause(audio: np.ndarray, sr: int, is_last: bool) -> np.ndarray:
|
||||
if is_last or audio.size == 0:
|
||||
return audio
|
||||
pause_ms = max(0, int(settings.tts_segment_pause_ms))
|
||||
if pause_ms <= 0:
|
||||
return audio
|
||||
pad = np.zeros(int(sr * pause_ms / 1000.0), dtype=np.float32)
|
||||
if pad.size == 0:
|
||||
return audio
|
||||
return np.concatenate([audio, pad])
|
||||
|
||||
async def _synthesize_sentence_first(
|
||||
self,
|
||||
reply_text: str,
|
||||
@@ -94,6 +115,7 @@ class ChatPipeline:
|
||||
if should_interrupt and should_interrupt():
|
||||
break
|
||||
sr = seg_sr
|
||||
seg_audio = self._append_segment_pause(seg_audio, seg_sr, is_last=(i == len(segments) - 1))
|
||||
if i == 0:
|
||||
first_chunk_ms = int((time.perf_counter() - t0) * 1000)
|
||||
if on_text_segment is not None:
|
||||
@@ -117,6 +139,20 @@ class ChatPipeline:
|
||||
asr_t0 = time.perf_counter()
|
||||
user_text = await self.asr.transcribe(audio_16k)
|
||||
asr_latency_ms = int((time.perf_counter() - asr_t0) * 1000)
|
||||
if not user_text.strip():
|
||||
await self.arbitrator.on_avatar_done()
|
||||
payload = {
|
||||
"user_text": "",
|
||||
"reply_text": "",
|
||||
"audio_samples": 0,
|
||||
"sample_rate": 24000,
|
||||
"asr_latency_ms": asr_latency_ms,
|
||||
"llm_latency_ms": 0,
|
||||
"llm_source": "skipped",
|
||||
"tts_latency_ms": 0,
|
||||
"tts_first_chunk_ms": 0,
|
||||
}
|
||||
return payload, np.zeros(1, dtype=np.float32), 24000
|
||||
if on_user_text is not None:
|
||||
await on_user_text(user_text)
|
||||
llm_t0 = time.perf_counter()
|
||||
|
||||
+18
-10
@@ -8,7 +8,7 @@
|
||||
|
||||
- 服务端不再输出视频流,只输出音频流和动画控制帧。
|
||||
- 浏览器负责 Three.js/VRM 渲染、模型加载、动画缓冲和最终口型展示。
|
||||
- WebRTC 只承载音频。
|
||||
- 浏览器和服务端之间的语音收发都通过 `WS /ws/audio` 完成。
|
||||
- 字幕、状态和动画控制通过 HTTP/SSE/WebSocket 传输。
|
||||
|
||||
这意味着后续如果你要替换数字人渲染方案,优先改的是浏览器侧;如果你要替换识别、大模型或语音合成,优先改的是 `services/` 和 `core/`。
|
||||
@@ -18,21 +18,21 @@
|
||||
### 2.1 语音链路
|
||||
|
||||
1. 浏览器通过 `getUserMedia` 获取麦克风。
|
||||
2. 浏览器调用 `POST /webrtc/offer`,建立音频上行和下行连接。
|
||||
3. 服务端在 [main.py](../main.py) 中接收音轨并进入 `_consume_user_audio()`。
|
||||
2. 浏览器连接 `WS /ws/audio`,上传 `pcm_s16le` 音频块,并接收服务端回传的回复语音片段。
|
||||
3. 服务端在 [main.py](../main.py) 中接收二进制 PCM 并进入 `_process_user_audio_chunk()`。
|
||||
4. 音频被重采样为 16k 单声道后送入 VAD。
|
||||
5. VAD 检测到用户语音结束后,触发 `ChatPipeline.process_turn()`。
|
||||
6. `ASRService` 输出用户文本。
|
||||
7. `LLMService` 调用线上 DeepSeek 兼容接口生成回答。
|
||||
8. `TTSService` 按句子分段合成。
|
||||
9. 每段 TTS 音频一方面入 `AudioBus`,通过 WebRTC 下发给浏览器播放;另一方面经 `AvatarService` 转成动画帧,通过 `/ws/animation` 发给浏览器。
|
||||
9. 每段 TTS 音频一方面通过 `WS /ws/audio` 下发给浏览器播放;另一方面经 `AvatarService` 转成动画帧,通过 `/ws/animation` 发给浏览器。
|
||||
10. 浏览器在播放语音的同时,从动画缓冲队列中取帧,驱动 VRM 表情、GLTF morph target 和头部骨骼。
|
||||
|
||||
### 2.2 文本链路
|
||||
|
||||
1. 浏览器调用 `POST /chat/text`。
|
||||
2. 后端直接执行 `LLM -> TTS -> 动画控制`。
|
||||
3. 字幕通过 `/ws/subtitles` 下发,语音仍通过 WebRTC 下发。
|
||||
3. 字幕通过 `/ws/subtitles` 下发,语音通过 `WS /ws/audio` 下发。
|
||||
|
||||
文本链路主要用于调试 LLM/TTS/动画,不依赖麦克风、VAD 和 ASR。
|
||||
|
||||
@@ -44,10 +44,10 @@
|
||||
|
||||
- 应用初始化。
|
||||
- 管理全局单例服务。
|
||||
- 维护 WebRTC peer 集合、字幕客户端集合、动画客户端集合。
|
||||
- 维护音频、字幕、动画客户端集合。
|
||||
- 维护 `RuntimeState`,给 `/health` 和 `/events` 提供实时指标。
|
||||
|
||||
当前实现假设一次只保留一个主 WebRTC 连接。新连接到来时会主动关闭旧连接。这是刻意简化,不是 bug。
|
||||
当前实现假设一次只保留一个主音频 WebSocket 连接。新连接到来时会主动关闭旧连接。这是刻意简化,不是 bug。
|
||||
|
||||
### 3.2 会话状态机
|
||||
|
||||
@@ -94,7 +94,7 @@
|
||||
|
||||
[web/app.js](../web/app.js) 同时承担了以下职责:
|
||||
|
||||
- 创建 WebRTC 连接。
|
||||
- 建立音频 WebSocket。
|
||||
- 订阅 `/events`、`/ws/subtitles`、`/ws/animation`。
|
||||
- 维护聊天消息面板和诊断面板。
|
||||
- 用 Three.js + VRM 加载和渲染模型。
|
||||
@@ -166,7 +166,7 @@
|
||||
后端通过 `/health` 和 `/events` 暴露运行状态。推荐排查问题时按这个顺序看:
|
||||
|
||||
1. `state`: 当前状态机是否符合预期。
|
||||
2. `peers`: WebRTC 是否真正建连。
|
||||
2. `audio_clients`: 语音通道是否真正建连。
|
||||
3. `vad_start_count` / `vad_end_count`: 是否检测到语音边界。
|
||||
4. `last_asr_latency_ms` / `last_llm_latency_ms` / `last_tts_latency_ms`: 性能瓶颈在哪一段。
|
||||
5. `llm.ready` / `asr.ready` / `tts.ready` / `vad.ready`: 组件有没有降级。
|
||||
@@ -177,7 +177,7 @@
|
||||
这些约束需要后续开发明确知晓:
|
||||
|
||||
- 当前是进程内单例服务,不是多租户、多 session 设计。
|
||||
- 浏览器侧只允许一个主 WebRTC 会话接管服务端音频输出。
|
||||
- 浏览器侧只允许一个主音频 WebSocket 会话接管服务端语音输入输出。
|
||||
- 模型推理基本都直接跑在应用事件循环附近,适合研发验证,不适合高并发。
|
||||
- 当前动画驱动依赖音频能量和谱质心,不具备严格唇音级别精度。
|
||||
- 语音链路高度依赖 VAD;如果要做电话式长连接交互,建议补更稳的 turn management。
|
||||
@@ -197,9 +197,16 @@
|
||||
- `GET /health`: 看整体运行状态。
|
||||
- `GET /events`: 看持续状态流。
|
||||
- `POST /chat/text`: 快速验证 LLM/TTS/动画,不依赖麦克风。
|
||||
- `WS /ws/audio`: 看语音上传和语音回放主通道。
|
||||
- `GET /avatar/schema`: 看控制字段列表。
|
||||
- [scripts/run_all_checks.sh](../scripts/run_all_checks.sh): 统一检查入口。
|
||||
|
||||
脚本使用建议:
|
||||
|
||||
- [scripts/model_probe.py](../scripts/model_probe.py): 快速确认 ASR/TTS/VAD 是否能加载。
|
||||
- [scripts/qa_check.py](../scripts/qa_check.py): 直接压文本链路,适合看 LLM/TTS 延迟。
|
||||
- [scripts/smoke_test.py](../scripts/smoke_test.py): 输出 `wav + mp4`,其中 `mp4` 是基于动画控制帧绘制的调试视频,不是最终 3D 渲染结果。
|
||||
|
||||
如果你刚接手这个项目,建议先按下面顺序理解:
|
||||
|
||||
1. [README.md](../README.md)
|
||||
@@ -207,3 +214,4 @@
|
||||
3. [core/pipeline.py](../core/pipeline.py)
|
||||
4. [services/avatar.py](../services/avatar.py)
|
||||
5. [web/app.js](../web/app.js)
|
||||
|
||||
|
||||
@@ -1,10 +1,12 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import io
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
import wave
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
import numpy as np
|
||||
@@ -52,7 +54,10 @@ audio_bus = AudioBus(sample_rate=48000)
|
||||
pcs: set[RTCPeerConnection] = set()
|
||||
subtitle_clients: set[WebSocket] = set()
|
||||
animation_clients: set[WebSocket] = set()
|
||||
audio_clients: set[WebSocket] = set()
|
||||
MAX_BUFFERED_AUDIO_SAMPLES = 16000 * 45
|
||||
MIN_VOICE_TURN_SAMPLES = 16000 // 2
|
||||
MIN_VOICE_RMS = 0.012
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -131,6 +136,7 @@ def _runtime_payload() -> dict:
|
||||
return {
|
||||
"state": str(arbitrator.state),
|
||||
"peers": len(pcs),
|
||||
"audio_clients": len(audio_clients),
|
||||
"subtitle_clients": len(subtitle_clients),
|
||||
"animation_clients": len(animation_clients),
|
||||
"pipeline_runs": runtime.pipeline_runs,
|
||||
@@ -197,6 +203,42 @@ def _resample_to_16k_mono(pcm: np.ndarray, sample_rate: int, channels: int) -> n
|
||||
return np.interp(x_new, x_old, mono).astype(np.float32)
|
||||
|
||||
|
||||
def _encode_wav_bytes(audio: np.ndarray, sample_rate: int) -> bytes:
|
||||
pcm = np.clip(audio.astype(np.float32, copy=False) * 32767.0, -32768, 32767).astype(np.int16)
|
||||
with io.BytesIO() as buffer:
|
||||
with wave.open(buffer, "wb") as wav_file:
|
||||
wav_file.setnchannels(1)
|
||||
wav_file.setsampwidth(2)
|
||||
wav_file.setframerate(sample_rate)
|
||||
wav_file.writeframes(pcm.tobytes())
|
||||
return buffer.getvalue()
|
||||
|
||||
|
||||
async def _broadcast_audio_chunk(audio: np.ndarray, sample_rate: int) -> None:
|
||||
if not audio_clients:
|
||||
return
|
||||
stale: list[WebSocket] = []
|
||||
payload = _encode_wav_bytes(audio, sample_rate)
|
||||
for ws in list(audio_clients):
|
||||
try:
|
||||
await ws.send_bytes(payload)
|
||||
except Exception:
|
||||
stale.append(ws)
|
||||
for ws in stale:
|
||||
audio_clients.discard(ws)
|
||||
|
||||
|
||||
async def _broadcast_audio_reset(reason: str) -> None:
|
||||
await _broadcast_json(
|
||||
audio_clients,
|
||||
{
|
||||
"type": "audio_reset",
|
||||
"reason": reason,
|
||||
"ts_ms": int(time.time() * 1000),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
async def _enqueue_audio_and_animation(
|
||||
seg_audio: np.ndarray,
|
||||
seg_sr: int,
|
||||
@@ -204,9 +246,7 @@ async def _enqueue_audio_and_animation(
|
||||
idx: int,
|
||||
total: int,
|
||||
) -> None:
|
||||
audio_48k_seg = _resample_mono(seg_audio.astype(np.float32, copy=False), seg_sr, 48000)
|
||||
pcm_seg = np.clip(audio_48k_seg * 32767.0, -32768, 32767).astype(np.int16)
|
||||
await audio_bus.enqueue(pcm_seg)
|
||||
await _broadcast_audio_chunk(seg_audio, seg_sr)
|
||||
animation_payload = await avatar_service.build_controls_from_audio(
|
||||
seg_audio,
|
||||
seg_sr,
|
||||
@@ -225,10 +265,18 @@ async def _run_pipeline_from_buffer() -> None:
|
||||
runtime.pending_voice_turn = False
|
||||
start_t = time.perf_counter()
|
||||
try:
|
||||
runtime.last_input_mode = "voice"
|
||||
audio_16k = np.concatenate(runtime.buffer_16k, axis=0)
|
||||
_clear_buffered_audio()
|
||||
|
||||
if audio_16k.shape[0] < MIN_VOICE_TURN_SAMPLES:
|
||||
return
|
||||
|
||||
rms = float(np.sqrt(np.mean(audio_16k.astype(np.float32) ** 2))) if audio_16k.size else 0.0
|
||||
if rms < MIN_VOICE_RMS:
|
||||
return
|
||||
|
||||
runtime.last_input_mode = "voice"
|
||||
|
||||
async def on_user_text_now(user_text: str) -> None:
|
||||
runtime.last_user_text = user_text
|
||||
await _broadcast_subtitle("user", user_text, "voice")
|
||||
@@ -246,6 +294,8 @@ async def _run_pipeline_from_buffer() -> None:
|
||||
on_reply_text=on_reply_text_now,
|
||||
on_reply_segment=on_reply_segment_now,
|
||||
)
|
||||
if not result["user_text"]:
|
||||
return
|
||||
runtime.last_user_text = result["user_text"]
|
||||
runtime.last_reply_text = result["reply_text"]
|
||||
runtime.pipeline_runs += 1
|
||||
@@ -256,9 +306,7 @@ async def _run_pipeline_from_buffer() -> None:
|
||||
runtime.last_tts_latency_ms = int(result.get("tts_latency_ms", 0))
|
||||
runtime.last_tts_first_chunk_ms = int(result.get("tts_first_chunk_ms", 0))
|
||||
if result.get("audio_samples", 0) <= 1:
|
||||
audio_48k = _resample_mono(audio.astype(np.float32, copy=False), sr, 48000)
|
||||
pcm_int16 = np.clip(audio_48k * 32767.0, -32768, 32767).astype(np.int16)
|
||||
await audio_bus.enqueue(pcm_int16)
|
||||
await _broadcast_audio_chunk(audio, sr)
|
||||
await _broadcast_animation(await avatar_service.build_controls_from_audio(audio, sr, text=result["reply_text"]))
|
||||
finally:
|
||||
runtime.busy = False
|
||||
@@ -266,6 +314,38 @@ async def _run_pipeline_from_buffer() -> None:
|
||||
asyncio.create_task(_run_pipeline_from_buffer())
|
||||
|
||||
|
||||
async def _process_user_audio_chunk(mono_16k: np.ndarray, vad_buffer: np.ndarray) -> np.ndarray:
|
||||
_append_buffered_audio(mono_16k)
|
||||
vad_buffer = np.concatenate([vad_buffer, mono_16k], axis=0)
|
||||
|
||||
while vad_buffer.shape[0] >= 512:
|
||||
chunk = vad_buffer[:512]
|
||||
vad_buffer = vad_buffer[512:]
|
||||
try:
|
||||
event = vad_service.push_chunk(chunk)
|
||||
except Exception as exc:
|
||||
logging.exception("VAD chunk processing failed: %s", exc)
|
||||
event = None
|
||||
if event and "start" in event:
|
||||
barge_t0 = time.perf_counter()
|
||||
was_avatar_speaking = arbitrator.state == SessionState.AVATAR_SPEAKING
|
||||
await arbitrator.on_speech_start()
|
||||
await audio_bus.clear()
|
||||
await _broadcast_audio_reset("speech-start")
|
||||
await _broadcast_animation(await avatar_service.build_reset_payload(reason="speech-start"))
|
||||
if was_avatar_speaking:
|
||||
runtime.last_barge_in_ms = int((time.perf_counter() - barge_t0) * 1000)
|
||||
runtime.barge_in_count += 1
|
||||
runtime.vad_start_count += 1
|
||||
if event and "end" in event:
|
||||
await arbitrator.on_speech_end()
|
||||
runtime.vad_end_count += 1
|
||||
runtime.pending_voice_turn = True
|
||||
asyncio.create_task(_run_pipeline_from_buffer())
|
||||
|
||||
return vad_buffer
|
||||
|
||||
|
||||
async def _consume_user_audio(track) -> None:
|
||||
vad_buffer = np.zeros(0, dtype=np.float32)
|
||||
while True:
|
||||
@@ -275,32 +355,7 @@ async def _consume_user_audio(track) -> None:
|
||||
layout = getattr(frame, "layout", None)
|
||||
channels = len(layout.channels) if layout and layout.channels else 1
|
||||
mono_16k = _resample_to_16k_mono(pcm, sample_rate, channels)
|
||||
_append_buffered_audio(mono_16k)
|
||||
vad_buffer = np.concatenate([vad_buffer, mono_16k], axis=0)
|
||||
|
||||
while vad_buffer.shape[0] >= 512:
|
||||
chunk = vad_buffer[:512]
|
||||
vad_buffer = vad_buffer[512:]
|
||||
try:
|
||||
event = vad_service.push_chunk(chunk)
|
||||
except Exception as exc:
|
||||
logging.exception("VAD chunk processing failed: %s", exc)
|
||||
event = None
|
||||
if event and "start" in event:
|
||||
barge_t0 = time.perf_counter()
|
||||
was_avatar_speaking = arbitrator.state == SessionState.AVATAR_SPEAKING
|
||||
await arbitrator.on_speech_start()
|
||||
await audio_bus.clear()
|
||||
await _broadcast_animation(await avatar_service.build_reset_payload(reason="speech-start"))
|
||||
if was_avatar_speaking:
|
||||
runtime.last_barge_in_ms = int((time.perf_counter() - barge_t0) * 1000)
|
||||
runtime.barge_in_count += 1
|
||||
runtime.vad_start_count += 1
|
||||
if event and "end" in event:
|
||||
await arbitrator.on_speech_end()
|
||||
runtime.vad_end_count += 1
|
||||
runtime.pending_voice_turn = True
|
||||
asyncio.create_task(_run_pipeline_from_buffer())
|
||||
vad_buffer = await _process_user_audio_chunk(mono_16k, vad_buffer)
|
||||
|
||||
|
||||
@app.get("/health")
|
||||
@@ -364,6 +419,74 @@ async def ws_animation(websocket: WebSocket):
|
||||
animation_clients.discard(websocket)
|
||||
|
||||
|
||||
@app.websocket("/ws/audio")
|
||||
async def ws_audio(websocket: WebSocket):
|
||||
await websocket.accept()
|
||||
|
||||
replaced_previous = len(audio_clients) > 0
|
||||
if replaced_previous:
|
||||
old_clients = list(audio_clients)
|
||||
await asyncio.gather(*(client.close(code=1012, reason="replaced") for client in old_clients), return_exceptions=True)
|
||||
for client in old_clients:
|
||||
audio_clients.discard(client)
|
||||
|
||||
audio_clients.add(websocket)
|
||||
sample_rate = 48000
|
||||
channels = 1
|
||||
vad_buffer = np.zeros(0, dtype=np.float32)
|
||||
|
||||
await websocket.send_text(
|
||||
json.dumps(
|
||||
{
|
||||
"type": "audio_ready",
|
||||
"transport": "websocket_audio",
|
||||
"replaced_previous": replaced_previous,
|
||||
},
|
||||
ensure_ascii=False,
|
||||
)
|
||||
)
|
||||
|
||||
try:
|
||||
while True:
|
||||
message = await websocket.receive()
|
||||
message_type = message.get("type")
|
||||
if message_type == "websocket.disconnect":
|
||||
break
|
||||
|
||||
text = message.get("text")
|
||||
if text is not None:
|
||||
try:
|
||||
payload = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
payload = {}
|
||||
event_type = str(payload.get("type", ""))
|
||||
if event_type == "audio_config":
|
||||
sample_rate = max(8000, int(payload.get("sample_rate", sample_rate) or sample_rate))
|
||||
channels = max(1, int(payload.get("channels", channels) or channels))
|
||||
elif event_type == "audio_reset":
|
||||
_clear_buffered_audio()
|
||||
runtime.pending_voice_turn = False
|
||||
await audio_bus.clear()
|
||||
await _broadcast_audio_reset("client-reset")
|
||||
elif event_type == "ping":
|
||||
await websocket.send_text(json.dumps({"type": "pong"}, ensure_ascii=False))
|
||||
continue
|
||||
|
||||
data = message.get("bytes")
|
||||
if not data:
|
||||
continue
|
||||
|
||||
pcm = np.frombuffer(data, dtype=np.int16)
|
||||
if pcm.size == 0:
|
||||
continue
|
||||
mono_16k = _resample_to_16k_mono(pcm, sample_rate, channels)
|
||||
vad_buffer = await _process_user_audio_chunk(mono_16k, vad_buffer)
|
||||
except WebSocketDisconnect:
|
||||
pass
|
||||
finally:
|
||||
audio_clients.discard(websocket)
|
||||
|
||||
|
||||
@app.post("/demo/run-once")
|
||||
async def run_once():
|
||||
fake_audio = np.zeros(16000 * 3, dtype=np.float32)
|
||||
@@ -390,6 +513,7 @@ async def chat_text(req: TextChatRequest):
|
||||
await _broadcast_subtitle("ai", seg_text, "text", partial=True, final=(idx >= total - 1))
|
||||
|
||||
await _broadcast_subtitle("user", text, "text")
|
||||
await _broadcast_audio_reset("text-turn-start")
|
||||
await _broadcast_animation(await avatar_service.build_reset_payload(reason="text-turn-start"))
|
||||
result, audio, sr = await pipeline.process_text_turn(
|
||||
text,
|
||||
@@ -408,9 +532,7 @@ async def chat_text(req: TextChatRequest):
|
||||
runtime.last_tts_latency_ms = int(result.get("tts_latency_ms", 0))
|
||||
runtime.last_tts_first_chunk_ms = int(result.get("tts_first_chunk_ms", 0))
|
||||
if result.get("audio_samples", 0) <= 1:
|
||||
audio_48k = _resample_mono(audio.astype(np.float32, copy=False), sr, 48000)
|
||||
pcm_int16 = np.clip(audio_48k * 32767.0, -32768, 32767).astype(np.int16)
|
||||
await audio_bus.enqueue(pcm_int16)
|
||||
await _broadcast_audio_chunk(audio, sr)
|
||||
await _broadcast_animation(await avatar_service.build_controls_from_audio(audio, sr, text=result["reply_text"]))
|
||||
return {"ok": True, **result}
|
||||
finally:
|
||||
@@ -438,6 +560,7 @@ async def chat_reset():
|
||||
runtime.vad_end_count = 0
|
||||
runtime.last_animation_frame_count = 0
|
||||
await audio_bus.clear()
|
||||
await _broadcast_audio_reset("chat-reset")
|
||||
await _broadcast_animation(await avatar_service.build_reset_payload(reason="chat-reset"))
|
||||
return {"ok": True}
|
||||
|
||||
|
||||
+1
-1
@@ -20,7 +20,7 @@ def _json_request(url: str, method: str = "GET", payload: dict | None = None, ti
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Quick QA for text pipeline and metrics")
|
||||
parser.add_argument("--base-url", default="https://127.0.0.1:8080")
|
||||
parser.add_argument("--base-url", default="http://127.0.0.1:8080")
|
||||
parser.add_argument("--rounds", type=int, default=5)
|
||||
args = parser.parse_args()
|
||||
|
||||
|
||||
@@ -6,11 +6,24 @@ ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$ROOT"
|
||||
PY="/home/xsl/miniconda3/envs/MuseTalk/bin/python"
|
||||
|
||||
if [[ -f ".env" ]]; then
|
||||
set -a
|
||||
# shellcheck disable=SC1091
|
||||
source ./.env
|
||||
set +a
|
||||
fi
|
||||
|
||||
SCHEME="http"
|
||||
if [[ -n "${SSL_CERTFILE:-}" && -n "${SSL_KEYFILE:-}" ]]; then
|
||||
SCHEME="https"
|
||||
fi
|
||||
BASE_URL="${BASE_URL:-${SCHEME}://127.0.0.1:${WEBRTC_PORT:-8080}}"
|
||||
|
||||
echo "[1/6] service status"
|
||||
bash scripts/service.sh status || true
|
||||
|
||||
echo "[2/6] quick pipeline qa"
|
||||
"$PY" scripts/qa_check.py --base-url "https://127.0.0.1:8080" --rounds 5
|
||||
"$PY" scripts/qa_check.py --base-url "$BASE_URL" --rounds 5
|
||||
|
||||
echo "[3/6] llm fallback"
|
||||
"$PY" scripts/llm_fallback_check.py
|
||||
|
||||
+13
-1
@@ -16,7 +16,19 @@ if [[ -f "$ENV_FILE" ]]; then
|
||||
set +a
|
||||
fi
|
||||
|
||||
HOST="${WEBRTC_HOST:-0.0.0.0}"
|
||||
normalize_host() {
|
||||
local host="${1:-0.0.0.0}"
|
||||
case "$host" in
|
||||
127.0.0.1|localhost|::1)
|
||||
echo "0.0.0.0"
|
||||
;;
|
||||
*)
|
||||
echo "$host"
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
HOST="$(normalize_host "${WEBRTC_HOST:-0.0.0.0}")"
|
||||
PORT="${WEBRTC_PORT:-8018}"
|
||||
SSL_CERTFILE="${SSL_CERTFILE:-$ROOT_DIR/certs/dev-cert.pem}"
|
||||
SSL_KEYFILE="${SSL_KEYFILE:-$ROOT_DIR/certs/dev-key.pem}"
|
||||
|
||||
+64
-16
@@ -35,34 +35,82 @@ async def _write_avatar_video(
|
||||
avatar: AvatarService, audio: np.ndarray, sr: int, out_path: str, fps: int
|
||||
) -> tuple[int, float]:
|
||||
fourcc = cv2.VideoWriter_fourcc(*"mp4v")
|
||||
writer = cv2.VideoWriter(out_path, fourcc, float(fps), (avatar.width, avatar.height))
|
||||
width = 720
|
||||
height = 720
|
||||
writer = cv2.VideoWriter(out_path, fourcc, float(fps), (width, height))
|
||||
if not writer.isOpened():
|
||||
raise RuntimeError(f"failed to open video writer: {out_path}")
|
||||
|
||||
duration_s = float(audio.shape[0]) / float(sr) if sr > 0 else 0.0
|
||||
total_frames = max(1, int(math.ceil(duration_s * fps)))
|
||||
samples_per_frame = max(1, int(sr / fps))
|
||||
payload = await avatar.build_controls_from_audio(audio, sr, text="smoke-test")
|
||||
frames = payload.get("frames", [])
|
||||
duration_ms = int(payload.get("duration_ms", 0))
|
||||
|
||||
for i in range(total_frames):
|
||||
s = i * samples_per_frame
|
||||
e = min(audio.shape[0], (i + 1) * samples_per_frame)
|
||||
seg = audio[s:e]
|
||||
if seg.size == 0:
|
||||
level = 0.0
|
||||
else:
|
||||
rms = float(np.sqrt(np.mean(seg.astype(np.float32) ** 2)))
|
||||
level = float(np.clip(rms * 10.0, 0.0, 1.0))
|
||||
frame = await avatar.render_frame(mouth_open=level, speaking=True)
|
||||
for frame_payload in frames:
|
||||
controls = frame_payload.get("controls", {})
|
||||
frame = _render_debug_frame(width, height, controls, speaking=bool(frame_payload.get("speaking", False)))
|
||||
writer.write(frame)
|
||||
|
||||
# add 0.4s idle tail for easy visual check
|
||||
tail_frames = max(1, int(0.4 * fps))
|
||||
for _ in range(tail_frames):
|
||||
frame = await avatar.render_frame(mouth_open=0.0, speaking=False)
|
||||
frame = _render_debug_frame(width, height, {}, speaking=False)
|
||||
writer.write(frame)
|
||||
|
||||
writer.release()
|
||||
return total_frames + tail_frames, duration_s + 0.4
|
||||
return len(frames) + tail_frames, (duration_ms / 1000.0) + 0.4
|
||||
|
||||
|
||||
def _render_debug_frame(width: int, height: int, controls: dict, speaking: bool) -> np.ndarray:
|
||||
jaw_open = float(np.clip(controls.get("jawOpen", 0.0), 0.0, 1.0))
|
||||
mouth_pucker = float(np.clip(controls.get("mouthPucker", 0.0), 0.0, 1.0))
|
||||
viseme_aa = float(np.clip(controls.get("viseme_aa", jaw_open), 0.0, 1.0))
|
||||
viseme_ee = float(np.clip(controls.get("viseme_ee", 0.0), 0.0, 1.0))
|
||||
viseme_oh = float(np.clip(controls.get("viseme_oh", 0.0), 0.0, 1.0))
|
||||
head_yaw = float(np.clip(controls.get("headYaw", 0.0), -1.0, 1.0))
|
||||
head_pitch = float(np.clip(controls.get("headPitch", 0.0), -1.0, 1.0))
|
||||
head_roll = float(np.clip(controls.get("headRoll", 0.0), -1.0, 1.0))
|
||||
|
||||
frame = np.zeros((height, width, 3), dtype=np.uint8)
|
||||
frame[:] = (16, 12, 10)
|
||||
|
||||
center_x = width // 2 + int(head_yaw * 48)
|
||||
center_y = height // 2 + int(head_pitch * 42) - 24
|
||||
head_radius = 190
|
||||
|
||||
cv2.circle(frame, (center_x, center_y), head_radius + 26, (42, 28, 18), -1, lineType=cv2.LINE_AA)
|
||||
cv2.circle(frame, (center_x, center_y), head_radius, (198, 214, 238), -1, lineType=cv2.LINE_AA)
|
||||
|
||||
eye_y = center_y - 44
|
||||
eye_offset = 64
|
||||
cv2.circle(frame, (center_x - eye_offset, eye_y), 12, (26, 31, 43), -1, lineType=cv2.LINE_AA)
|
||||
cv2.circle(frame, (center_x + eye_offset, eye_y), 12, (26, 31, 43), -1, lineType=cv2.LINE_AA)
|
||||
|
||||
brow_dx = int(head_roll * 28)
|
||||
cv2.line(frame, (center_x - 100 - brow_dx, eye_y - 34), (center_x - 42 - brow_dx, eye_y - 40), (48, 58, 74), 6, lineType=cv2.LINE_AA)
|
||||
cv2.line(frame, (center_x + 42 - brow_dx, eye_y - 40), (center_x + 100 - brow_dx, eye_y - 34), (48, 58, 74), 6, lineType=cv2.LINE_AA)
|
||||
|
||||
mouth_center = (center_x, center_y + 86)
|
||||
mouth_width = int(64 + viseme_oh * 36 + mouth_pucker * 18)
|
||||
mouth_height = int(12 + jaw_open * 98 + viseme_aa * 18)
|
||||
lip_color = (96, 54, 130) if speaking else (88, 74, 108)
|
||||
inner_color = (36, 22, 54)
|
||||
cv2.ellipse(frame, mouth_center, (mouth_width, max(6, mouth_height)), 0, 0, 360, lip_color, -1, lineType=cv2.LINE_AA)
|
||||
cv2.ellipse(frame, mouth_center, (max(12, mouth_width - 16), max(4, mouth_height - 8)), 0, 0, 360, inner_color, -1, lineType=cv2.LINE_AA)
|
||||
|
||||
ee_width = int(max(20, 86 - viseme_ee * 46))
|
||||
cv2.line(frame, (mouth_center[0] - ee_width, mouth_center[1]), (mouth_center[0] + ee_width, mouth_center[1]), (220, 224, 232), 3, lineType=cv2.LINE_AA)
|
||||
|
||||
metrics = [
|
||||
f"jawOpen={jaw_open:.2f}",
|
||||
f"viseme_aa={viseme_aa:.2f}",
|
||||
f"viseme_ee={viseme_ee:.2f}",
|
||||
f"viseme_oh={viseme_oh:.2f}",
|
||||
f"headYaw={head_yaw:.2f}",
|
||||
]
|
||||
for index, text in enumerate(metrics):
|
||||
cv2.putText(frame, text, (28, 42 + index * 28), cv2.FONT_HERSHEY_SIMPLEX, 0.72, (210, 220, 235), 2, lineType=cv2.LINE_AA)
|
||||
|
||||
return frame
|
||||
|
||||
|
||||
async def main() -> None:
|
||||
|
||||
+18
-3
@@ -8,11 +8,26 @@ cd "$ROOT"
|
||||
PY="/home/xsl/miniconda3/envs/MuseTalk/bin/python"
|
||||
|
||||
if [[ -f ".env" ]]; then
|
||||
# shellcheck disable=SC2046
|
||||
export $(grep -E '^[A-Za-z_][A-Za-z0-9_]*=' .env | xargs)
|
||||
set -a
|
||||
# shellcheck disable=SC1091
|
||||
source ./.env
|
||||
set +a
|
||||
fi
|
||||
|
||||
ARGS=(main:app --host "${WEBRTC_HOST:-0.0.0.0}" --port "${WEBRTC_PORT:-8080}")
|
||||
normalize_host() {
|
||||
local host="${1:-0.0.0.0}"
|
||||
case "$host" in
|
||||
127.0.0.1|localhost|::1)
|
||||
echo "0.0.0.0"
|
||||
;;
|
||||
*)
|
||||
echo "$host"
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
HOST="$(normalize_host "${WEBRTC_HOST:-0.0.0.0}")"
|
||||
ARGS=(main:app --host "$HOST" --port "${WEBRTC_PORT:-8080}")
|
||||
|
||||
if [[ -n "${SSL_CERTFILE:-}" && -n "${SSL_KEYFILE:-}" ]]; then
|
||||
ARGS+=(--ssl-certfile "$SSL_CERTFILE" --ssl-keyfile "$SSL_KEYFILE")
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
cd "$ROOT"
|
||||
|
||||
MODE="${1:-auto}"
|
||||
|
||||
if [[ -f ".env" ]]; then
|
||||
set -a
|
||||
# shellcheck disable=SC1091
|
||||
source ./.env
|
||||
set +a
|
||||
fi
|
||||
|
||||
normalize_host() {
|
||||
local host="${1:-0.0.0.0}"
|
||||
case "$host" in
|
||||
127.0.0.1|localhost|::1)
|
||||
echo "0.0.0.0"
|
||||
;;
|
||||
*)
|
||||
echo "$host"
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
pick_python() {
|
||||
if [[ -n "${VISUAL_CHAT_PYTHON:-}" && -x "${VISUAL_CHAT_PYTHON}" ]]; then
|
||||
echo "$VISUAL_CHAT_PYTHON"
|
||||
return 0
|
||||
fi
|
||||
if [[ -n "${VIRTUAL_ENV:-}" && -x "${VIRTUAL_ENV}/bin/python" ]]; then
|
||||
echo "${VIRTUAL_ENV}/bin/python"
|
||||
return 0
|
||||
fi
|
||||
if [[ -x "/home/xsl/miniconda3/envs/MuseTalk/bin/python" ]]; then
|
||||
echo "/home/xsl/miniconda3/envs/MuseTalk/bin/python"
|
||||
return 0
|
||||
fi
|
||||
if command -v python3 >/dev/null 2>&1; then
|
||||
command -v python3
|
||||
return 0
|
||||
fi
|
||||
command -v python
|
||||
}
|
||||
|
||||
PY="$(pick_python)"
|
||||
HOST="$(normalize_host "${WEBRTC_HOST:-0.0.0.0}")"
|
||||
PORT="${WEBRTC_PORT:-8080}"
|
||||
SCHEME="http"
|
||||
ARGS=(main:app --host "$HOST" --port "$PORT")
|
||||
|
||||
case "$MODE" in
|
||||
auto)
|
||||
if [[ -n "${SSL_CERTFILE:-}" && -n "${SSL_KEYFILE:-}" ]]; then
|
||||
SCHEME="https"
|
||||
ARGS+=(--ssl-certfile "$SSL_CERTFILE" --ssl-keyfile "$SSL_KEYFILE")
|
||||
fi
|
||||
;;
|
||||
https)
|
||||
if [[ -z "${SSL_CERTFILE:-}" || -z "${SSL_KEYFILE:-}" ]]; then
|
||||
echo "SSL_CERTFILE / SSL_KEYFILE 未配置,无法以 https 模式启动" >&2
|
||||
exit 1
|
||||
fi
|
||||
SCHEME="https"
|
||||
ARGS+=(--ssl-certfile "$SSL_CERTFILE" --ssl-keyfile "$SSL_KEYFILE")
|
||||
;;
|
||||
http)
|
||||
;;
|
||||
*)
|
||||
echo "usage: bash scripts/start-test.sh [auto|http|https]" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
|
||||
echo "repo: $ROOT"
|
||||
echo "python: $PY"
|
||||
echo "mode: $MODE"
|
||||
echo "bind: ${HOST}:${PORT}"
|
||||
echo "listen: ${SCHEME}://${HOST}:${PORT}"
|
||||
echo "browser test page: ${SCHEME}://<server-ip>:${PORT}/"
|
||||
|
||||
exec "$PY" -m uvicorn "${ARGS[@]}"
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+27
-16
@@ -15,25 +15,36 @@ class ASRService:
|
||||
self._ready = False
|
||||
self._attempted = False
|
||||
self._init_error: Optional[str] = None
|
||||
self._device: str = "uninitialized"
|
||||
|
||||
def _ensure_loaded(self) -> None:
|
||||
if self._ready or self._attempted:
|
||||
return
|
||||
self._attempted = True
|
||||
try:
|
||||
from funasr import AutoModel
|
||||
from funasr.utils.postprocess_utils import rich_transcription_postprocess
|
||||
from funasr import AutoModel
|
||||
from funasr.utils.postprocess_utils import rich_transcription_postprocess
|
||||
|
||||
self._model = AutoModel(
|
||||
model="FunAudioLLM/SenseVoiceSmall",
|
||||
device="cuda:0",
|
||||
hub="hf",
|
||||
)
|
||||
self._post = rich_transcription_postprocess
|
||||
self._ready = True
|
||||
except Exception as exc: # pragma: no cover
|
||||
self._init_error = str(exc)
|
||||
logger.warning("ASR fallback mode: %s", exc)
|
||||
errors: list[str] = []
|
||||
for device in ("cuda:0", "cpu"):
|
||||
try:
|
||||
self._model = AutoModel(
|
||||
model="FunAudioLLM/SenseVoiceSmall",
|
||||
device=device,
|
||||
hub="hf",
|
||||
)
|
||||
self._post = rich_transcription_postprocess
|
||||
self._ready = True
|
||||
self._init_error = None
|
||||
self._device = device
|
||||
if device != "cuda:0":
|
||||
logger.warning("ASR running in degraded mode on %s", device)
|
||||
return
|
||||
except Exception as exc: # pragma: no cover
|
||||
errors.append(f"{device}: {exc}")
|
||||
|
||||
self._init_error = " | ".join(errors)
|
||||
self._device = "unavailable"
|
||||
logger.warning("ASR unavailable: %s", self._init_error)
|
||||
|
||||
async def transcribe(self, audio_16k: np.ndarray) -> str:
|
||||
self._ensure_loaded()
|
||||
@@ -46,9 +57,9 @@ class ASRService:
|
||||
)
|
||||
return self._post(res[0]["text"])
|
||||
|
||||
# Fallback for initial bring-up
|
||||
return "(ASR降级)语音已接收"
|
||||
# When ASR is unavailable, avoid emitting fake user text that would trigger a bogus LLM reply.
|
||||
return ""
|
||||
|
||||
@property
|
||||
def health(self) -> dict:
|
||||
return {"ready": self._ready, "attempted": self._attempted, "error": self._init_error}
|
||||
return {"ready": self._ready, "attempted": self._attempted, "device": self._device, "error": self._init_error}
|
||||
|
||||
+20
-3
@@ -2,6 +2,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from typing import Optional
|
||||
from collections import deque
|
||||
|
||||
@@ -18,7 +19,7 @@ class LLMService:
|
||||
self._model = (settings.llm_model or "").strip()
|
||||
self._base_url = (settings.llm_base_url or "").strip()
|
||||
self._history: deque[dict] = deque(maxlen=12) # 6 turns (user+assistant)
|
||||
self._system_prompt = "请用中文口语化简短回复,1-2句。"
|
||||
self._system_prompt = (settings.llm_system_prompt or "").strip() or "请用中文口语化简短回复,1-2句。"
|
||||
|
||||
if not settings.llm_api_key:
|
||||
self._error = "missing LLM api key: set LLM_API_KEY or DEEPSEEK_API_KEY"
|
||||
@@ -51,6 +52,22 @@ class LLMService:
|
||||
self._error = str(exc)
|
||||
logger.warning("LLM online mode unavailable: %s", exc)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_reply_text(text: str) -> str:
|
||||
cleaned = (text or "").strip()
|
||||
if not cleaned:
|
||||
return ""
|
||||
cleaned = cleaned.replace("\r", "\n")
|
||||
cleaned = re.sub(r"```.*?```", "", cleaned, flags=re.S)
|
||||
cleaned = re.sub(r"`([^`]*)`", r"\1", cleaned)
|
||||
cleaned = re.sub(r"\*\*(.*?)\*\*", r"\1", cleaned)
|
||||
cleaned = re.sub(r"\*(.*?)\*", r"\1", cleaned)
|
||||
cleaned = re.sub(r"\s*([,。!?;:、,.!?;:])\s*", r"\1", cleaned)
|
||||
cleaned = re.sub(r"(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])", "", cleaned)
|
||||
cleaned = re.sub(r"\n+", " ", cleaned)
|
||||
cleaned = re.sub(r"\s{2,}", " ", cleaned)
|
||||
return cleaned.strip()
|
||||
|
||||
async def reply_with_meta(self, text: str) -> tuple[str, str]:
|
||||
if not self._ready or self._client is None:
|
||||
return f"收到:{text}。这是本地兜底回复。", "fallback"
|
||||
@@ -60,9 +77,9 @@ class LLMService:
|
||||
model=self._model,
|
||||
messages=[{"role": "system", "content": self._system_prompt}, *list(self._history), {"role": "user", "content": text}],
|
||||
max_tokens=settings.llm_max_tokens,
|
||||
temperature=0.7,
|
||||
temperature=max(0.1, min(1.2, settings.llm_temperature)),
|
||||
)
|
||||
content = (resp.choices[0].message.content or "").strip()
|
||||
content = self._normalize_reply_text(resp.choices[0].message.content or "")
|
||||
if content:
|
||||
self._history.append({"role": "user", "content": text})
|
||||
self._history.append({"role": "assistant", "content": content})
|
||||
|
||||
+39
-2
@@ -1,10 +1,13 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import Optional
|
||||
|
||||
import numpy as np
|
||||
|
||||
from config import settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -28,12 +31,46 @@ class TTSService:
|
||||
self._init_error = str(exc)
|
||||
logger.warning("TTS fallback mode: %s", exc)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_text(text: str) -> str:
|
||||
cleaned = (text or "").strip()
|
||||
if not cleaned:
|
||||
return ""
|
||||
cleaned = cleaned.replace("\r", "\n")
|
||||
cleaned = re.sub(r"[\t\f\v]+", " ", cleaned)
|
||||
cleaned = re.sub(r"\s*([,。!?;:、,.!?;:])\s*", r"\1", cleaned)
|
||||
cleaned = re.sub(r"(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])", "", cleaned)
|
||||
cleaned = re.sub(r"\n{2,}", "\n", cleaned)
|
||||
cleaned = re.sub(r"\s{2,}", " ", cleaned)
|
||||
return cleaned.strip()
|
||||
|
||||
@staticmethod
|
||||
def _apply_edge_fade(audio: np.ndarray, sr: int) -> np.ndarray:
|
||||
if audio.size == 0:
|
||||
return audio
|
||||
fade_samples = max(1, min(int(sr * max(settings.tts_fade_ms, 0) / 1000.0), audio.shape[0] // 8))
|
||||
if fade_samples <= 1:
|
||||
return audio
|
||||
out = np.array(audio, copy=True)
|
||||
ramp = np.linspace(0.0, 1.0, fade_samples, dtype=np.float32)
|
||||
out[:fade_samples] *= ramp
|
||||
out[-fade_samples:] *= ramp[::-1]
|
||||
return out
|
||||
|
||||
async def synthesize(self, text: str) -> tuple[np.ndarray, int]:
|
||||
text = self._normalize_text(text)
|
||||
if not text:
|
||||
return np.zeros(1, dtype=np.float32), 24000
|
||||
self._ensure_loaded()
|
||||
if self._ready and self._pipeline is not None:
|
||||
chunks = []
|
||||
for _, _, audio in self._pipeline(text, voice="zf_xiaoxiao"):
|
||||
chunks.append(audio)
|
||||
for _, _, audio in self._pipeline(
|
||||
text,
|
||||
voice=settings.tts_voice,
|
||||
speed=max(0.8, min(1.1, settings.tts_speed)),
|
||||
split_pattern=r"\n+",
|
||||
):
|
||||
chunks.append(self._apply_edge_fade(np.asarray(audio, dtype=np.float32), 24000))
|
||||
if chunks:
|
||||
return np.concatenate(chunks), 24000
|
||||
|
||||
|
||||
+3
-3
@@ -21,10 +21,10 @@ class VADService:
|
||||
self._model = load_silero_vad()
|
||||
self._iterator = VADIterator(
|
||||
self._model,
|
||||
threshold=0.5,
|
||||
threshold=0.65,
|
||||
sampling_rate=16000,
|
||||
min_silence_duration_ms=300,
|
||||
speech_pad_ms=30,
|
||||
min_silence_duration_ms=450,
|
||||
speech_pad_ms=80,
|
||||
)
|
||||
self._ready = True
|
||||
except Exception as exc: # pragma: no cover
|
||||
|
||||
+380
-80
@@ -50,17 +50,28 @@ const dVad = document.getElementById("dVad");
|
||||
const dAsr = document.getElementById("dAsr");
|
||||
const dLlm = document.getElementById("dLlm");
|
||||
const dTts = document.getElementById("dTts");
|
||||
const OFFER_BTN_DEFAULT_TEXT = offerBtn?.textContent || "连接语音通道";
|
||||
|
||||
let pc = null;
|
||||
let evt = null;
|
||||
let subtitleWs = null;
|
||||
let animationWs = null;
|
||||
let audioWs = null;
|
||||
let localMicStream = null;
|
||||
let aiStreamingEl = null;
|
||||
let mediaAudioContext = null;
|
||||
let micSourceNode = null;
|
||||
let micProcessorNode = null;
|
||||
let micSinkGain = null;
|
||||
let audioPlaybackCursor = 0;
|
||||
const audioPlaybackSources = new Set();
|
||||
let reconnectOnGestureArmed = false;
|
||||
let pageGestureBootstrapArmed = false;
|
||||
|
||||
const AUTO_RECONNECT_KEY = "visual_chat_auto_reconnect";
|
||||
const ANIMATION_BUFFER_MS = 120;
|
||||
const DEFAULT_BODY_MODEL_URL = "https://cdn.jsdelivr.net/gh/pixiv/three-vrm@release/packages/three-vrm/examples/models/VRM1_Constraint_Twist_Sample.vrm";
|
||||
const DEFAULT_BODY_MODEL_URL =
|
||||
(modelUrl?.value || "").trim() ||
|
||||
"/web/models/VRM1_Constraint_Twist_Sample.vrm";
|
||||
const DEFAULT_VRM_FALLBACK_URL = "https://raw.githubusercontent.com/pixiv/three-vrm/dev/packages/three-vrm/examples/models/VRM1_Constraint_Twist_Sample.vrm";
|
||||
const CLOCK = new THREE.Clock();
|
||||
|
||||
@@ -154,10 +165,226 @@ function setConnBadge(label, kind = "neutral") {
|
||||
connBadge.className = `badge ${kind}`;
|
||||
}
|
||||
|
||||
function setOfferButtonState(mode = "default") {
|
||||
if (!offerBtn) return;
|
||||
if (mode === "resume") {
|
||||
offerBtn.textContent = "点击恢复语音";
|
||||
return;
|
||||
}
|
||||
offerBtn.textContent = OFFER_BTN_DEFAULT_TEXT;
|
||||
}
|
||||
|
||||
function setDot(el, kind) {
|
||||
el.className = `dot ${kind}`;
|
||||
}
|
||||
|
||||
function syncConnBadgeFromState(health = null) {
|
||||
if (audioWs?.readyState === WebSocket.OPEN) {
|
||||
if (!health || (health.audio_clients ?? 0) > 0) {
|
||||
setConnBadge("已连接", "ok");
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (audioWs?.readyState === WebSocket.CONNECTING) {
|
||||
setConnBadge("连接中...", "warn");
|
||||
return;
|
||||
}
|
||||
if (health && (health.audio_clients ?? 0) > 0) {
|
||||
setConnBadge("已连接", "ok");
|
||||
return;
|
||||
}
|
||||
setConnBadge("未连接", "neutral");
|
||||
}
|
||||
|
||||
function disarmReconnectOnGesture() {
|
||||
if (!reconnectOnGestureArmed) return;
|
||||
window.removeEventListener("pointerdown", handleReconnectGesture, true);
|
||||
window.removeEventListener("keydown", handleReconnectGesture, true);
|
||||
reconnectOnGestureArmed = false;
|
||||
setOfferButtonState();
|
||||
}
|
||||
|
||||
function armReconnectOnGesture() {
|
||||
if (reconnectOnGestureArmed) return;
|
||||
reconnectOnGestureArmed = true;
|
||||
statusEl.textContent = "状态:等待恢复语音";
|
||||
outputEl.textContent = "浏览器拦截了刷新后的自动语音恢复。点击页面任意位置,或点“连接语音通道”,即可恢复麦克风和语音播放。";
|
||||
setConnBadge("等待点击恢复", "warn");
|
||||
setOfferButtonState("resume");
|
||||
window.addEventListener("pointerdown", handleReconnectGesture, true);
|
||||
window.addEventListener("keydown", handleReconnectGesture, true);
|
||||
}
|
||||
|
||||
async function handleReconnectGesture() {
|
||||
try {
|
||||
await ensureVoiceReadyFromGesture();
|
||||
} catch (error) {
|
||||
console.error(error);
|
||||
armReconnectOnGesture();
|
||||
}
|
||||
}
|
||||
|
||||
function disarmPageGestureBootstrap() {
|
||||
if (!pageGestureBootstrapArmed) return;
|
||||
window.removeEventListener("pointerdown", handlePageGestureBootstrap, true);
|
||||
window.removeEventListener("keydown", handlePageGestureBootstrap, true);
|
||||
pageGestureBootstrapArmed = false;
|
||||
if (!reconnectOnGestureArmed) {
|
||||
setOfferButtonState();
|
||||
}
|
||||
}
|
||||
|
||||
function armPageGestureBootstrap() {
|
||||
if (pageGestureBootstrapArmed) return;
|
||||
pageGestureBootstrapArmed = true;
|
||||
setOfferButtonState("resume");
|
||||
window.addEventListener("pointerdown", handlePageGestureBootstrap, true);
|
||||
window.addEventListener("keydown", handlePageGestureBootstrap, true);
|
||||
}
|
||||
|
||||
async function ensureVoiceReadyFromGesture() {
|
||||
await ensureMediaAudioContext();
|
||||
|
||||
if (audioWs?.readyState === WebSocket.CONNECTING) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (audioWs?.readyState === WebSocket.OPEN) {
|
||||
await startMicCapture();
|
||||
disarmReconnectOnGesture();
|
||||
disarmPageGestureBootstrap();
|
||||
syncConnBadgeFromState();
|
||||
return;
|
||||
}
|
||||
|
||||
await connectAudio({ silentReadyMessage: true });
|
||||
disarmReconnectOnGesture();
|
||||
disarmPageGestureBootstrap();
|
||||
}
|
||||
|
||||
async function handlePageGestureBootstrap() {
|
||||
if (localStorage.getItem(AUTO_RECONNECT_KEY) !== "1") {
|
||||
disarmPageGestureBootstrap();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
await ensureVoiceReadyFromGesture();
|
||||
} catch (error) {
|
||||
console.error(error);
|
||||
}
|
||||
}
|
||||
|
||||
function wsBaseUrl(path) {
|
||||
const scheme = location.protocol === "https:" ? "wss" : "ws";
|
||||
return `${scheme}://${location.host}${path}`;
|
||||
}
|
||||
|
||||
async function ensureMediaAudioContext() {
|
||||
const AudioContextClass = window.AudioContext || window.webkitAudioContext;
|
||||
if (!AudioContextClass) {
|
||||
throw new Error("当前浏览器不支持 Web Audio API");
|
||||
}
|
||||
if (!mediaAudioContext) {
|
||||
mediaAudioContext = new AudioContextClass();
|
||||
}
|
||||
if (mediaAudioContext.state === "suspended") {
|
||||
await mediaAudioContext.resume();
|
||||
}
|
||||
if (!Number.isFinite(audioPlaybackCursor) || audioPlaybackCursor <= 0) {
|
||||
audioPlaybackCursor = mediaAudioContext.currentTime + 0.04;
|
||||
}
|
||||
return mediaAudioContext;
|
||||
}
|
||||
|
||||
function resetAudioPlayback() {
|
||||
if (mediaAudioContext) {
|
||||
audioPlaybackCursor = mediaAudioContext.currentTime + 0.04;
|
||||
}
|
||||
Array.from(audioPlaybackSources).forEach((source) => {
|
||||
try {
|
||||
source.stop();
|
||||
} catch (_) {
|
||||
// no-op
|
||||
}
|
||||
});
|
||||
audioPlaybackSources.clear();
|
||||
}
|
||||
|
||||
async function queueServerAudio(arrayBuffer) {
|
||||
const context = await ensureMediaAudioContext();
|
||||
const decoded = await context.decodeAudioData(arrayBuffer.slice(0));
|
||||
const source = context.createBufferSource();
|
||||
source.buffer = decoded;
|
||||
source.connect(context.destination);
|
||||
const startAt = Math.max(context.currentTime + 0.04, audioPlaybackCursor);
|
||||
source.start(startAt);
|
||||
audioPlaybackCursor = startAt + decoded.duration;
|
||||
audioPlaybackSources.add(source);
|
||||
source.onended = () => {
|
||||
audioPlaybackSources.delete(source);
|
||||
};
|
||||
}
|
||||
|
||||
function stopMicCapture() {
|
||||
if (micSourceNode) {
|
||||
micSourceNode.disconnect();
|
||||
micSourceNode = null;
|
||||
}
|
||||
if (micProcessorNode) {
|
||||
micProcessorNode.disconnect();
|
||||
micProcessorNode.onaudioprocess = null;
|
||||
micProcessorNode = null;
|
||||
}
|
||||
if (micSinkGain) {
|
||||
micSinkGain.disconnect();
|
||||
micSinkGain = null;
|
||||
}
|
||||
if (localMicStream) {
|
||||
localMicStream.getTracks().forEach((track) => track.stop());
|
||||
localMicStream = null;
|
||||
}
|
||||
}
|
||||
|
||||
async function startMicCapture() {
|
||||
const context = await ensureMediaAudioContext();
|
||||
if (localMicStream && micSourceNode && micProcessorNode) {
|
||||
return context;
|
||||
}
|
||||
|
||||
const stream = await navigator.mediaDevices.getUserMedia({
|
||||
audio: {
|
||||
echoCancellation: true,
|
||||
noiseSuppression: true,
|
||||
autoGainControl: true,
|
||||
},
|
||||
video: false,
|
||||
});
|
||||
|
||||
localMicStream = stream;
|
||||
micSourceNode = context.createMediaStreamSource(stream);
|
||||
micProcessorNode = context.createScriptProcessor(4096, 1, 1);
|
||||
micSinkGain = context.createGain();
|
||||
micSinkGain.gain.value = 0;
|
||||
|
||||
micProcessorNode.onaudioprocess = (event) => {
|
||||
if (!audioWs || audioWs.readyState !== WebSocket.OPEN) {
|
||||
return;
|
||||
}
|
||||
const input = event.inputBuffer.getChannelData(0);
|
||||
const pcm = new Int16Array(input.length);
|
||||
for (let i = 0; i < input.length; i += 1) {
|
||||
const sample = Math.max(-1, Math.min(1, input[i] || 0));
|
||||
pcm[i] = sample < 0 ? sample * 32768 : sample * 32767;
|
||||
}
|
||||
audioWs.send(pcm.buffer);
|
||||
};
|
||||
|
||||
micSourceNode.connect(micProcessorNode);
|
||||
micProcessorNode.connect(micSinkGain);
|
||||
micSinkGain.connect(context.destination);
|
||||
return context;
|
||||
}
|
||||
|
||||
function initStage() {
|
||||
scene.background = new THREE.Color(0x0b1118);
|
||||
|
||||
@@ -285,7 +512,7 @@ function renderLoop() {
|
||||
}
|
||||
|
||||
function updateMetrics(data) {
|
||||
mPeers.textContent = String(data.peers ?? 0);
|
||||
mPeers.textContent = String(data.audio_clients ?? data.peers ?? 0);
|
||||
mBusy.textContent = data.pipeline_busy ? "是" : "否";
|
||||
mLatency.textContent = `${data.last_latency_ms ?? 0} ms`;
|
||||
mAsrLatency.textContent = `${data.last_asr_latency_ms ?? 0} ms`;
|
||||
@@ -302,11 +529,12 @@ function updateMetrics(data) {
|
||||
mAnimDriver.textContent = data?.avatar?.driver || data.last_animation_mode || "--";
|
||||
mAnimFrames.textContent = String(data.last_animation_frame_count ?? data?.avatar?.last_frame_count ?? 0);
|
||||
|
||||
setDot(dConn, (data.peers ?? 0) > 0 ? "ok" : "warn");
|
||||
setDot(dConn, (data.audio_clients ?? data.peers ?? 0) > 0 ? "ok" : "warn");
|
||||
setDot(dVad, (data.vad_start_count ?? 0) > 0 ? "ok" : "warn");
|
||||
setDot(dAsr, data?.asr?.ready ? "ok" : "warn");
|
||||
setDot(dLlm, data?.llm?.ready ? "ok" : "bad");
|
||||
setDot(dTts, data?.tts?.ready ? "ok" : "warn");
|
||||
syncConnBadgeFromState(data);
|
||||
}
|
||||
|
||||
function queueAnimationFrames(payload) {
|
||||
@@ -724,8 +952,7 @@ function subscribeEvents() {
|
||||
|
||||
function subscribeSubtitles() {
|
||||
if (subtitleWs && (subtitleWs.readyState === WebSocket.OPEN || subtitleWs.readyState === WebSocket.CONNECTING)) return;
|
||||
const scheme = location.protocol === "https:" ? "wss" : "ws";
|
||||
subtitleWs = new WebSocket(`${scheme}://${location.host}/ws/subtitles`);
|
||||
subtitleWs = new WebSocket(wsBaseUrl("/ws/subtitles"));
|
||||
subtitleWs.onmessage = (event) => {
|
||||
try {
|
||||
const data = JSON.parse(event.data);
|
||||
@@ -753,9 +980,8 @@ function subscribeSubtitles() {
|
||||
|
||||
function subscribeAnimation() {
|
||||
if (animationWs && (animationWs.readyState === WebSocket.OPEN || animationWs.readyState === WebSocket.CONNECTING)) return;
|
||||
const scheme = location.protocol === "https:" ? "wss" : "ws";
|
||||
animationConn.textContent = "连接中";
|
||||
animationWs = new WebSocket(`${scheme}://${location.host}/ws/animation`);
|
||||
animationWs = new WebSocket(wsBaseUrl("/ws/animation"));
|
||||
animationWs.onopen = () => {
|
||||
animationConn.textContent = "已连接";
|
||||
};
|
||||
@@ -781,59 +1007,48 @@ function subscribeAnimation() {
|
||||
};
|
||||
}
|
||||
|
||||
async function connectWebRTC() {
|
||||
if (pc) {
|
||||
outputEl.textContent = "WebRTC 已连接或正在连接";
|
||||
async function connectAudio(options = {}) {
|
||||
const { silentReadyMessage = false } = options;
|
||||
if (audioWs && audioWs.readyState === WebSocket.OPEN) {
|
||||
disarmReconnectOnGesture();
|
||||
outputEl.textContent = "语音通道已连接";
|
||||
return;
|
||||
}
|
||||
if (audioWs && audioWs.readyState === WebSocket.CONNECTING) {
|
||||
outputEl.textContent = "语音通道正在连接";
|
||||
await new Promise((resolve, reject) => {
|
||||
const startedAt = Date.now();
|
||||
const timer = window.setInterval(() => {
|
||||
if (!audioWs) {
|
||||
window.clearInterval(timer);
|
||||
reject(new Error("语音通道不存在"));
|
||||
return;
|
||||
}
|
||||
if (audioWs.readyState === WebSocket.OPEN) {
|
||||
window.clearInterval(timer);
|
||||
resolve();
|
||||
return;
|
||||
}
|
||||
if (audioWs.readyState === WebSocket.CLOSING || audioWs.readyState === WebSocket.CLOSED) {
|
||||
window.clearInterval(timer);
|
||||
reject(new Error("语音通道连接已关闭"));
|
||||
return;
|
||||
}
|
||||
if (Date.now() - startedAt > 5000) {
|
||||
window.clearInterval(timer);
|
||||
reject(new Error("语音通道连接超时"));
|
||||
}
|
||||
}, 100);
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
setConnBadge("连接中...", "warn");
|
||||
offerBtn.disabled = true;
|
||||
|
||||
const metaRes = await fetch("/meta");
|
||||
const meta = await metaRes.json();
|
||||
pc = new RTCPeerConnection({
|
||||
iceServers: [{ urls: [meta.stun_url || "stun:stun.l.google.com:19302"] }],
|
||||
});
|
||||
const currentPc = pc;
|
||||
let seenConnected = false;
|
||||
|
||||
currentPc.onconnectionstatechange = () => {
|
||||
const st = currentPc.connectionState;
|
||||
if (st === "connected") {
|
||||
seenConnected = true;
|
||||
setConnBadge("已连接", "ok");
|
||||
localStorage.setItem(AUTO_RECONNECT_KEY, "1");
|
||||
} else if (st === "connecting") {
|
||||
setConnBadge("连接中...", "warn");
|
||||
} else if (st === "failed" || st === "disconnected" || st === "closed") {
|
||||
setConnBadge(`连接${st}`, "warn");
|
||||
if (seenConnected && (st === "disconnected" || st === "failed")) {
|
||||
addMessage("system", "连接已断开,可能被新的客户端连接接管。");
|
||||
}
|
||||
if (localMicStream) {
|
||||
localMicStream.getTracks().forEach((track) => track.stop());
|
||||
localMicStream = null;
|
||||
}
|
||||
if (pc === currentPc) {
|
||||
pc = null;
|
||||
}
|
||||
offerBtn.disabled = false;
|
||||
}
|
||||
};
|
||||
|
||||
pc.ontrack = (event) => {
|
||||
if (event.streams && event.streams[0]) {
|
||||
remoteAudio.srcObject = event.streams[0];
|
||||
remoteAudio.play().catch(() => {
|
||||
addMessage("system", "远端语音已到达,如未自动播放请先与页面交互后重试。");
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
let stream;
|
||||
let context;
|
||||
try {
|
||||
stream = await navigator.mediaDevices.getUserMedia({ audio: true, video: false });
|
||||
localMicStream = stream;
|
||||
context = await startMicCapture();
|
||||
} catch (err) {
|
||||
statusEl.textContent = "状态:麦克风权限失败";
|
||||
outputEl.textContent = `getUserMedia 失败:${err?.message || err}\n提示:跨机器访问请使用 HTTPS 或浏览器安全例外。`;
|
||||
@@ -842,34 +1057,113 @@ async function connectWebRTC() {
|
||||
throw err;
|
||||
}
|
||||
|
||||
stream.getTracks().forEach((track) => pc.addTrack(track, stream));
|
||||
const offer = await pc.createOffer();
|
||||
await pc.setLocalDescription(offer);
|
||||
const ws = new WebSocket(wsBaseUrl("/ws/audio"));
|
||||
ws.binaryType = "arraybuffer";
|
||||
audioWs = ws;
|
||||
|
||||
const res = await fetch("/webrtc/offer", {
|
||||
method: "POST",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
sdp: pc.localDescription.sdp,
|
||||
type: pc.localDescription.type,
|
||||
}),
|
||||
await new Promise((resolve, reject) => {
|
||||
let settled = false;
|
||||
|
||||
ws.onopen = () => {
|
||||
if (audioWs !== ws) {
|
||||
try {
|
||||
ws.close(1000, "stale-open");
|
||||
} catch (_) {
|
||||
// no-op
|
||||
}
|
||||
return;
|
||||
}
|
||||
ws.send(
|
||||
JSON.stringify({
|
||||
type: "audio_config",
|
||||
sample_rate: context.sampleRate,
|
||||
channels: 1,
|
||||
format: "pcm_s16le",
|
||||
}),
|
||||
);
|
||||
statusEl.textContent = "状态:语音通道已连接";
|
||||
outputEl.textContent = JSON.stringify({ connected: true, transport: "websocket_audio" }, null, 2);
|
||||
setConnBadge("已连接", "ok");
|
||||
localStorage.setItem(AUTO_RECONNECT_KEY, "1");
|
||||
disarmReconnectOnGesture();
|
||||
offerBtn.disabled = false;
|
||||
if (!silentReadyMessage) {
|
||||
addMessage("system", "语音通道已连接,可开始对话。");
|
||||
}
|
||||
subscribeEvents();
|
||||
settled = true;
|
||||
resolve();
|
||||
};
|
||||
|
||||
ws.onmessage = (event) => {
|
||||
if (typeof event.data === "string") {
|
||||
try {
|
||||
const data = JSON.parse(event.data);
|
||||
if (data.type === "ping" || data.type === "pong" || data.type === "audio_ready") {
|
||||
return;
|
||||
}
|
||||
if (data.type === "audio_reset") {
|
||||
resetAudioPlayback();
|
||||
return;
|
||||
}
|
||||
} catch (_) {
|
||||
// no-op
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
queueServerAudio(event.data).catch((error) => {
|
||||
console.error(error);
|
||||
addMessage("system", `语音播放失败:${error?.message || error}`);
|
||||
});
|
||||
};
|
||||
|
||||
ws.onerror = (event) => {
|
||||
console.error(event);
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
offerBtn.disabled = false;
|
||||
setConnBadge("连接失败", "warn");
|
||||
reject(new Error("语音通道连接失败"));
|
||||
}
|
||||
};
|
||||
|
||||
ws.onclose = (event) => {
|
||||
if (audioWs !== ws) {
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
resolve();
|
||||
}
|
||||
return;
|
||||
}
|
||||
resetAudioPlayback();
|
||||
stopMicCapture();
|
||||
if (audioWs === ws) {
|
||||
audioWs = null;
|
||||
}
|
||||
syncConnBadgeFromState();
|
||||
offerBtn.disabled = false;
|
||||
if (event.code !== 1000 && event.code !== 1001 && event.code !== 1012) {
|
||||
addMessage("system", "语音通道已断开,请重新连接。");
|
||||
}
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
reject(new Error("语音通道在建立前已关闭"));
|
||||
}
|
||||
};
|
||||
});
|
||||
const answer = await res.json();
|
||||
await pc.setRemoteDescription(answer);
|
||||
statusEl.textContent = "状态:WebRTC 已连接";
|
||||
outputEl.textContent = JSON.stringify({ connected: true, iceConnectionState: pc.iceConnectionState }, null, 2);
|
||||
setConnBadge("已连接", "ok");
|
||||
addMessage("system", "WebRTC 已连接,可开始对话。");
|
||||
if (answer.replaced_previous) {
|
||||
addMessage("system", "检测到已有旧连接,已自动断开旧连接并由当前页面接管。");
|
||||
}
|
||||
|
||||
subscribeEvents();
|
||||
}
|
||||
|
||||
async function sendText() {
|
||||
const text = (chatInput.value || "").trim();
|
||||
if (!text || sendBtn.disabled) return;
|
||||
if (!audioWs || audioWs.readyState !== WebSocket.OPEN) {
|
||||
try {
|
||||
await connectAudio({ silentReadyMessage: true });
|
||||
} catch (_) {
|
||||
addMessage("system", "语音通道未连接,本次仅保证文字链路可用。");
|
||||
}
|
||||
}
|
||||
sendBtn.disabled = true;
|
||||
const res = await fetch("/chat/text", {
|
||||
method: "POST",
|
||||
@@ -891,6 +1185,7 @@ async function resetChat() {
|
||||
messagesEl.innerHTML = "";
|
||||
avatarState.pendingFrames = [];
|
||||
avatarState.scheduleCursorMs = performance.now() + ANIMATION_BUFFER_MS;
|
||||
resetAudioPlayback();
|
||||
resetAvatarPose();
|
||||
addMessage("system", "会话已重置。");
|
||||
await checkHealth();
|
||||
@@ -898,10 +1193,11 @@ async function resetChat() {
|
||||
|
||||
function tryAutoReconnect() {
|
||||
if (localStorage.getItem(AUTO_RECONNECT_KEY) !== "1") return;
|
||||
armPageGestureBootstrap();
|
||||
setTimeout(() => {
|
||||
if (!pc) {
|
||||
connectWebRTC().catch(() => {
|
||||
// no-op
|
||||
if (!audioWs) {
|
||||
connectAudio({ silentReadyMessage: true }).catch(() => {
|
||||
armReconnectOnGesture();
|
||||
});
|
||||
}
|
||||
}, 300);
|
||||
@@ -923,7 +1219,11 @@ new ResizeObserver(() => resizeStage()).observe(avatarCanvas);
|
||||
window.addEventListener("resize", resizeStage);
|
||||
|
||||
healthBtn.addEventListener("click", checkHealth);
|
||||
offerBtn.addEventListener("click", connectWebRTC);
|
||||
offerBtn.addEventListener("click", () => {
|
||||
connectAudio().catch((error) => {
|
||||
console.error(error);
|
||||
});
|
||||
});
|
||||
resetBtn.addEventListener("click", resetChat);
|
||||
sendBtn.addEventListener("click", sendText);
|
||||
loadModelBtn.addEventListener("click", handleModelUrlLoad);
|
||||
|
||||
+5
-5
@@ -18,7 +18,7 @@
|
||||
<div class="stage-toolbar">
|
||||
<label class="file-pill" for="modelFile">载入本地 VRM/GLB</label>
|
||||
<input id="modelFile" type="file" accept=".vrm,.glb,.gltf" />
|
||||
<input id="modelUrl" type="text" placeholder="输入 VRM/GLB/GLTF 地址,例如官方示例 VRM 或 /web/models/avatar.vrm" />
|
||||
<input id="modelUrl" type="text" value="/web/models/VRM1_Constraint_Twist_Sample.vrm" placeholder="输入 VRM/GLB/GLTF 地址,例如官方示例 VRM 或 /web/models/avatar.vrm" />
|
||||
<button id="loadModelBtn">加载模型 URL</button>
|
||||
<button id="resetViewBtn">重置视角</button>
|
||||
</div>
|
||||
@@ -48,7 +48,7 @@
|
||||
<audio id="remoteAudio" autoplay></audio>
|
||||
<div class="actions">
|
||||
<button id="healthBtn">检查服务状态</button>
|
||||
<button id="offerBtn">连接 WebRTC</button>
|
||||
<button id="offerBtn">连接语音通道</button>
|
||||
<button id="resetBtn">清空会话</button>
|
||||
<span id="connBadge" class="badge neutral">未连接</span>
|
||||
</div>
|
||||
@@ -58,14 +58,14 @@
|
||||
</div>
|
||||
<div id="messages" class="messages"></div>
|
||||
<div class="diag" id="diag">
|
||||
<div class="diag-item"><span class="dot" id="dConn"></span><span>WebRTC连接</span></div>
|
||||
<div class="diag-item"><span class="dot" id="dConn"></span><span>语音通道</span></div>
|
||||
<div class="diag-item"><span class="dot" id="dVad"></span><span>VAD触发</span></div>
|
||||
<div class="diag-item"><span class="dot" id="dAsr"></span><span>ASR</span></div>
|
||||
<div class="diag-item"><span class="dot" id="dLlm"></span><span>LLM</span></div>
|
||||
<div class="diag-item"><span class="dot" id="dTts"></span><span>TTS</span></div>
|
||||
</div>
|
||||
<div class="metrics" id="metrics">
|
||||
<div class="metric"><span>连接数</span><strong id="mPeers">0</strong></div>
|
||||
<div class="metric"><span>语音通道</span><strong id="mPeers">0</strong></div>
|
||||
<div class="metric"><span>处理中</span><strong id="mBusy">否</strong></div>
|
||||
<div class="metric"><span>总耗时</span><strong id="mLatency">0 ms</strong></div>
|
||||
<div class="metric"><span>ASR耗时</span><strong id="mAsrLatency">0 ms</strong></div>
|
||||
@@ -96,6 +96,6 @@
|
||||
}
|
||||
}
|
||||
</script>
|
||||
<script type="module" src="/web/app.js?v=20260327d"></script>
|
||||
<script type="module" src="/web/app.js?v=20260329k"></script>
|
||||
</body>
|
||||
</html>
|
||||
|
||||
Binary file not shown.
Reference in New Issue
Block a user