fix: 优化小米语音

This commit is contained in:
hp0912 2026-09-06 14:12:31 +08:00
parent 999da6083c
commit 153e90cb89
2 changed files with 116 additions and 19 deletions

View File

@ -107,7 +107,7 @@ description: "文本转语音与语音消息发送技能。当用户想让我说
6. 语速、音高、音量、方言有明确要求时优先填 `speaking_rate`、`pitch`、`volume`、`dialect`;复杂演绎要求放入 `style_prompt`。 6. 语速、音高、音量、方言有明确要求时优先填 `speaking_rate`、`pitch`、`volume`、`dialect`;复杂演绎要求放入 `style_prompt`。
7. `audio_tags` 仅用于用户明确要求唱歌、方言、笑声、停顿、深呼吸等标签化控制时;如果用户已把标签写在 `content` 中,不要重复添加。 7. `audio_tags` 仅用于用户明确要求唱歌、方言、笑声、停顿、深呼吸等标签化控制时;如果用户已把标签写在 `content` 中,不要重复添加。
8. `context_texts` 适合表达上下文、场景、人物状态和补充播报要求。 8. `context_texts` 适合表达上下文、场景、人物状态和补充播报要求。
9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID` 自动判断并下载引用语音作为复刻样本。 9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID`(数据库 `messages.id`)自动判断并下载引用语音作为复刻样本。
10. `content` 超过 260 个字符时,不应该调用本技能。 10. `content` 超过 260 个字符时,不应该调用本技能。
## 音频标签控制 ## 音频标签控制
@ -172,6 +172,8 @@ description: "文本转语音与语音消息发送技能。当用户想让我说
- 只有`mimo-v2.5-tts`模型支持唱歌模式 - 只有`mimo-v2.5-tts`模型支持唱歌模式
- 唱歌请求使用预置音色模型;音色设计、音色复刻均不支持唱歌。用户同时要求引用语音克隆和唱歌时,说明该组合不受支持。
- 如需体验更佳的唱歌风格,必须在目标文本最开头添加 `(唱歌)` 标签,格式为:`(唱歌)歌词`。歌词 建议采用中文,可获得更优合成效果。标签内标识支持以下取值,效果等效:`唱歌`、`sing`、`singing` - 如需体验更佳的唱歌风格,必须在目标文本最开头添加 `(唱歌)` 标签,格式为:`(唱歌)歌词`。歌词 建议采用中文,可获得更优合成效果。标签内标识支持以下取值,效果等效:`唱歌`、`sing`、`singing`
## 执行步骤 ## 执行步骤
@ -190,10 +192,46 @@ python3 scripts/voice_message.py --content '这是一条语音消息' --emotion
- Doubao:`content` 写入文本字段;支持的 `emotion` 写入音频情绪参数;`voice` 可覆盖 speaker;其他风格控制会合并到 `context_texts` 辅助信息。 - Doubao:`content` 写入文本字段;支持的 `emotion` 写入音频情绪参数;`voice` 可覆盖 speaker;其他风格控制会合并到 `context_texts` 辅助信息。
- MiMo V2.5:`content` 写入 `assistant` 消息;`style_prompt`、`voice_prompt`、`context_texts`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect` 会合并为 `user` 风格/音色控制;`audio_tags` 会作为整体标签加到要合成的文本前。 - MiMo V2.5:`content` 写入 `assistant` 消息;`style_prompt`、`voice_prompt`、`context_texts`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect` 会合并为 `user` 风格/音色控制;`audio_tags` 会作为整体标签加到要合成的文本前。
- MiMo 会默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 流式兼容模式并在脚本内封装为 `wav`。 - MiMo 默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 并在脚本内封装为 24 kHz、单声道 `wav`。普通朗读支持低延迟流式;音色设计、复刻当前在推理完成后以流式格式返回结果。
- MiMo 在 `auto_model` 未关闭时,会根据 `voice_prompt` 自动选择 `mimo-v2.5-tts-voicedesign`;如果 `ROBOT_REF_MESSAGE_ID` 指向数据库中 `messages.type = 34` 的语音消息,则脚本会调用客户端接口下载该语音 wav,并自动选择 `mimo-v2.5-tts-voiceclone`。 - 引用语音消息时,按 `messages.id` 查询引用消息并检查 `type = 34`,下载 wav 后选择 `mimo-v2.5-tts-voiceclone`。例如引用一条语音并要求「用这个声音说:晚上好」。也可在后台配置固定的 `voice_clone_audio` 样本。
- 上下文明确指定预置 `voice` 时使用普通朗读模型;`auto_model` 开启时,上下文的 `voice_prompt` 选择音色设计模型。这些明确要求优先于配置中的默认复刻样本和音色描述。
- 引用消息下载接口为 `GET http://127.0.0.1:{ROBOT_WECHAT_CLIENT_PORT}/api/v1/robot/chat/voice/download?message_id={ROBOT_REF_MESSAGE_ID}`,返回 wav 后由脚本封装为 MiMo 需要的 `data:audio/wav;base64,...`。 - 引用消息下载接口为 `GET http://127.0.0.1:{ROBOT_WECHAT_CLIENT_PORT}/api/v1/robot/chat/voice/download?message_id={ROBOT_REF_MESSAGE_ID}`,返回 wav 后由脚本封装为 MiMo 需要的 `data:audio/wav;base64,...`。
## MiMo 配置
管理后台文本转语音配置中的 `mimo` 提供以下参数,不需要填写 `model`。模型由脚本根据上下文提取的音色描述、复刻音频等信息自动选择。
```json
{
"base_url": "https://api.xiaomimimo.com/v1",
"api_key": "",
"voice": "mimo_default",
"audio_format": "wav",
"stream": false,
"timeout": 300,
"auto_model": true,
"voice_prompt": "",
"style_prompt": [],
"context_texts": [],
"audio_tags": [],
"emotion": "",
"speaking_rate": "",
"pitch": "",
"volume": "",
"dialect": "",
"voice_clone_audio": "",
"voice_clone_mime_type": "audio/mpeg"
}
```
- `base_url`、`api_key` 用于语音合成请求;留空时沿用聊天接口的对应配置。`timeout` 是请求超时秒数,必须大于 0。
- `voice`、`voice_prompt`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect`、`audio_tags` 是默认语音控制值,上下文明确指定时优先使用上下文参数;`style_prompt` 和 `context_texts` 会与上下文提供的内容合并。
- `audio_format` 控制非流式输出格式;`stream: true` 时使用 `pcm16` 并封装为 `wav`。PCM 采样率按接口协议处理。
- `voice_clone_audio` 支持 MP3/WAV 的 Base64 或音频 data URL,Base64 内容不能超过 10 MB。仅填写 Base64 时,使用 `voice_clone_mime_type` 指定类型:`audio/mpeg`、`audio/mp3` 或 `audio/wav`。引用语音优先于配置样本,格式错误、数据为空或超限时会在调用 MiMo 前报错。
- `auto_model` 默认开启,优先使用上下文的音色要求,再按配置样本、配置音色描述、普通朗读的顺序选择;关闭时使用普通朗读模型。引用语音始终选择音色复刻;唱歌使用普通朗读模型,与引用语音克隆冲突时会明确报错。
协议依据:[小米 MiMo V2.5 语音合成官方文档](https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5)。
## 依赖安装 ## 依赖安装
- 脚本首次运行时会自动创建虚拟环境并安装依赖,无需手动执行。 - 脚本首次运行时会自动创建虚拟环境并安装依赖,无需手动执行。

View File

@ -6,7 +6,9 @@ import argparse
import base64 import base64
import gzip import gzip
import json import json
import math
import os import os
import re
import subprocess import subprocess
import sys import sys
import tempfile import tempfile
@ -54,6 +56,8 @@ MIMO_STREAM_AUDIO_FORMAT = "pcm16"
MIMO_PCM_SAMPLE_RATE = 24000 MIMO_PCM_SAMPLE_RATE = 24000
MIMO_VOICE_DESIGN_MODEL = "mimo-v2.5-tts-voicedesign" MIMO_VOICE_DESIGN_MODEL = "mimo-v2.5-tts-voicedesign"
MIMO_VOICE_CLONE_MODEL = "mimo-v2.5-tts-voiceclone" MIMO_VOICE_CLONE_MODEL = "mimo-v2.5-tts-voiceclone"
MIMO_VOICE_CLONE_MAX_BASE64_SIZE = 10_000_000
MIMO_VOICE_CLONE_MIME_TYPES = {"audio/mpeg", "audio/mp3", "audio/wav"}
WECHAT_VOICE_MESSAGE_TYPE = 34 WECHAT_VOICE_MESSAGE_TYPE = 34
MAX_CONTENT_LENGTH = 260 MAX_CONTENT_LENGTH = 260
STREAM_END_CODE = 20000000 STREAM_END_CODE = 20000000
@ -263,7 +267,8 @@ def _download_referenced_voice_clone(message_id: str) -> str:
) )
try: try:
with urllib.request.urlopen(req, timeout=60) as response: with urllib.request.urlopen(req, timeout=60) as response:
wav_data = response.read() max_audio_size = MIMO_VOICE_CLONE_MAX_BASE64_SIZE // 4 * 3
wav_data = response.read(max_audio_size + 1)
except urllib.error.HTTPError as exc: except urllib.error.HTTPError as exc:
error_body = exc.read().decode("utf-8", errors="replace") error_body = exc.read().decode("utf-8", errors="replace")
raise RuntimeError(f"下载引用语音失败,状态码 {exc.code}: {error_body}") from exc raise RuntimeError(f"下载引用语音失败,状态码 {exc.code}: {error_body}") from exc
@ -272,6 +277,8 @@ def _download_referenced_voice_clone(message_id: str) -> str:
if not wav_data: if not wav_data:
raise RuntimeError("下载引用语音失败: 响应为空") raise RuntimeError("下载引用语音失败: 响应为空")
if len(wav_data) > max_audio_size:
raise RuntimeError("引用语音过大,音色复刻样本的 Base64 不能超过 10 MB")
audio_b64 = base64.b64encode(wav_data).decode("utf-8") audio_b64 = base64.b64encode(wav_data).decode("utf-8")
return f"data:audio/wav;base64,{audio_b64}" return f"data:audio/wav;base64,{audio_b64}"
@ -282,7 +289,14 @@ def _load_referenced_voice_clone(conn) -> str:
if not ref_message_id: if not ref_message_id:
return "" return ""
message = _query_one(conn, "SELECT * FROM messages WHERE msg_id = %s LIMIT 1", (ref_message_id,)) try:
message_id = int(ref_message_id)
except ValueError:
return ""
if message_id <= 0:
return ""
message = _query_one(conn, "SELECT id, type FROM messages WHERE id = %s LIMIT 1", (message_id,))
if not message: if not message:
return "" return ""
@ -294,7 +308,7 @@ def _load_referenced_voice_clone(conn) -> str:
if message_type != WECHAT_VOICE_MESSAGE_TYPE: if message_type != WECHAT_VOICE_MESSAGE_TYPE:
return "" return ""
return _download_referenced_voice_clone(ref_message_id) return _download_referenced_voice_clone(str(message["id"]))
def _parse_cli_params(argv: list[str]) -> dict: def _parse_cli_params(argv: list[str]) -> dict:
@ -520,18 +534,34 @@ def _config_texts(config: dict, key: str) -> list[str]:
return [text] if text else [] return [text] if text else []
def _mimo_singing_requested(config: dict, params: dict) -> bool:
tags = list(params.get("audio_tags") or _config_texts(config, "audio_tags"))
leading_tags = re.match(
r"^(?:(?:\([^)]*\)|([^)]*)|\[[^\]]*\])\s*)+",
_clean_text(params.get("content")),
)
if leading_tags:
tags.append(leading_tags.group())
return any(re.search(r"唱歌|\bsing(?:ing)?\b", tag, re.IGNORECASE) for tag in tags)
def _resolve_mimo_model(config: dict, params: dict) -> str: def _resolve_mimo_model(config: dict, params: dict) -> str:
configured_model = _clean_text(config.get("model")) if _mimo_singing_requested(config, params):
if _clean_text(params.get("voice_clone_audio")):
raise RuntimeError("MiMo 音色复刻不支持唱歌,请改为朗读,或取消引用语音后使用预置音色唱歌")
return DEFAULT_MIMO_MODEL
if _clean_text(params.get("voice_clone_audio")): if _clean_text(params.get("voice_clone_audio")):
return MIMO_VOICE_CLONE_MODEL return MIMO_VOICE_CLONE_MODEL
if _clean_text(params.get("voice")):
return DEFAULT_MIMO_MODEL
auto_model = _coerce_bool(config.get("auto_model"), True) auto_model = _coerce_bool(config.get("auto_model"), True)
if auto_model and _clean_text(params.get("voice_prompt")):
return MIMO_VOICE_DESIGN_MODEL
if auto_model and _clean_text(config.get("voice_clone_audio")): if auto_model and _clean_text(config.get("voice_clone_audio")):
return MIMO_VOICE_CLONE_MODEL return MIMO_VOICE_CLONE_MODEL
if auto_model and (_clean_text(params.get("voice_prompt")) or _clean_text(config.get("voice_prompt"))): if auto_model and _clean_text(config.get("voice_prompt")):
return MIMO_VOICE_DESIGN_MODEL return MIMO_VOICE_DESIGN_MODEL
if configured_model:
return configured_model
return DEFAULT_MIMO_MODEL return DEFAULT_MIMO_MODEL
@ -542,9 +572,9 @@ def _format_mimo_audio_tags(tags: list[str]) -> str:
return f"({' '.join(cleaned_tags)})" return f"({' '.join(cleaned_tags)})"
def _build_mimo_assistant_content(params: dict) -> str: def _build_mimo_assistant_content(config: dict, params: dict) -> str:
content = _clean_text(params.get("content")) content = _clean_text(params.get("content"))
tags = _format_mimo_audio_tags(params.get("audio_tags") or []) tags = _format_mimo_audio_tags(params.get("audio_tags") or _config_texts(config, "audio_tags"))
return f"{tags}{content}" if tags else content return f"{tags}{content}" if tags else content
@ -580,6 +610,30 @@ def _build_mimo_user_content(config: dict, params: dict, model: str) -> str:
return "\n".join(parts) return "\n".join(parts)
def _mimo_voice_clone_data_url(audio: str, mime_type: str) -> str:
encoded = audio
if audio.lower().startswith("data:"):
header, separator, encoded = audio.partition(",")
if not separator or not header.lower().endswith(";base64"):
raise RuntimeError("音色复刻样本必须使用 data:audio/...;base64,... 格式")
mime_type = header[5:-7]
mime_type = mime_type.strip().lower()
if mime_type not in MIMO_VOICE_CLONE_MIME_TYPES:
raise RuntimeError("MiMo 音色复刻仅支持 MP3/WAV 样本,MIME 类型须为 audio/mpeg、audio/mp3 或 audio/wav")
encoded = "".join(encoded.split())
if len(encoded) > MIMO_VOICE_CLONE_MAX_BASE64_SIZE:
raise RuntimeError("音色复刻样本的 Base64 不能超过 10 MB")
try:
audio_bytes = base64.b64decode(encoded, validate=True)
except ValueError as exc:
raise RuntimeError("音色复刻样本不是有效的 Base64 音频数据") from exc
if not audio_bytes:
raise RuntimeError("音色复刻样本不能为空")
return f"data:{mime_type};base64,{encoded}"
def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str: def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str:
if model == MIMO_VOICE_DESIGN_MODEL: if model == MIMO_VOICE_DESIGN_MODEL:
return "" return ""
@ -588,14 +642,12 @@ def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str:
voice_clone_audio = _clean_text(params.get("voice_clone_audio")) or _clean_text(config.get("voice_clone_audio")) voice_clone_audio = _clean_text(params.get("voice_clone_audio")) or _clean_text(config.get("voice_clone_audio"))
if not voice_clone_audio: if not voice_clone_audio:
raise RuntimeError("mimo 音色复刻模型需要引用一条语音消息或配置 voice_clone_audio") raise RuntimeError("mimo 音色复刻模型需要引用一条语音消息或配置 voice_clone_audio")
if voice_clone_audio.startswith("data:"):
return voice_clone_audio
mime_type = ( mime_type = (
_clean_text(params.get("voice_clone_mime_type")) _clean_text(params.get("voice_clone_mime_type"))
or _clean_text(config.get("voice_clone_mime_type")) or _clean_text(config.get("voice_clone_mime_type"))
or "audio/mpeg" or "audio/mpeg"
) )
return f"data:{mime_type};base64,{voice_clone_audio}" return _mimo_voice_clone_data_url(voice_clone_audio, mime_type)
return _clean_text(params.get("voice")) or _clean_text(config.get("voice")) or DEFAULT_MIMO_VOICE return _clean_text(params.get("voice")) or _clean_text(config.get("voice")) or DEFAULT_MIMO_VOICE
@ -604,14 +656,14 @@ def _build_mimo_payload(config: dict, params: dict) -> tuple[dict, str, bool]:
model = _resolve_mimo_model(config, params) model = _resolve_mimo_model(config, params)
stream = _coerce_bool(config.get("stream"), False) stream = _coerce_bool(config.get("stream"), False)
audio_format = MIMO_STREAM_AUDIO_FORMAT if stream else ( audio_format = MIMO_STREAM_AUDIO_FORMAT if stream else (
_clean_text(config.get("audio_format")) or _clean_text(config.get("format")) or DEFAULT_MIMO_AUDIO_FORMAT _clean_text(config.get("audio_format")) or DEFAULT_MIMO_AUDIO_FORMAT
) )
messages = [] messages = []
user_content = _build_mimo_user_content(config, params, model) user_content = _build_mimo_user_content(config, params, model)
if user_content or model == MIMO_VOICE_CLONE_MODEL: if user_content or model == MIMO_VOICE_CLONE_MODEL:
messages.append({"role": "user", "content": user_content}) messages.append({"role": "user", "content": user_content})
messages.append({"role": "assistant", "content": _build_mimo_assistant_content(params)}) messages.append({"role": "assistant", "content": _build_mimo_assistant_content(config, params)})
audio = {"format": audio_format} audio = {"format": audio_format}
voice = _resolve_mimo_voice(config, params, model) voice = _resolve_mimo_voice(config, params, model)
@ -739,6 +791,13 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]:
if not api_key: if not api_key:
raise RuntimeError("mimo api_key 不能为空") raise RuntimeError("mimo api_key 不能为空")
try:
timeout = float(config.get("timeout", 300))
except (TypeError, ValueError) as exc:
raise RuntimeError("mimo timeout 必须是大于 0 的秒数") from exc
if not math.isfinite(timeout) or timeout <= 0:
raise RuntimeError("mimo timeout 必须是大于 0 的秒数")
# 兼容用户把 base_url 配成不带 /v1 的根地址(如 New API / OneAPI 等网关), # 兼容用户把 base_url 配成不带 /v1 的根地址(如 New API / OneAPI 等网关),
# 避免请求被前端 SPA 兜底返回 index.html。 # 避免请求被前端 SPA 兜底返回 index.html。
parsed_base = urllib.parse.urlsplit(base_url) parsed_base = urllib.parse.urlsplit(base_url)
@ -763,7 +822,7 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]:
) )
try: try:
response = urllib.request.urlopen(req, timeout=300) response = urllib.request.urlopen(req, timeout=timeout)
except urllib.error.HTTPError as exc: except urllib.error.HTTPError as exc:
try: try:
error_body = _read_response_text(exc) error_body = _read_response_text(exc)
@ -903,7 +962,7 @@ def main() -> int:
return 1 return 1
try: try:
if tts_model == "mimo": if enabled and tts_model == "mimo":
voice_clone_audio = _load_referenced_voice_clone(conn) voice_clone_audio = _load_referenced_voice_clone(conn)
if voice_clone_audio: if voice_clone_audio:
params = dict(params) params = dict(params)