fix: 优化小米语音

This commit is contained in:
hp0912 2026-09-06 14:12:31 +08:00
parent 999da6083c
commit 153e90cb89
2 changed files with 116 additions and 19 deletions

View File

@ -107,7 +107,7 @@ description: "文本转语音与语音消息发送技能。当用户想让我说
6. 语速、音高、音量、方言有明确要求时优先填 `speaking_rate`、`pitch`、`volume`、`dialect`;复杂演绎要求放入 `style_prompt`。
7. `audio_tags` 仅用于用户明确要求唱歌、方言、笑声、停顿、深呼吸等标签化控制时;如果用户已把标签写在 `content` 中,不要重复添加。
8. `context_texts` 适合表达上下文、场景、人物状态和补充播报要求。
9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID` 自动判断并下载引用语音作为复刻样本。
9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID`(数据库 `messages.id`)自动判断并下载引用语音作为复刻样本。
10. `content` 超过 260 个字符时,不应该调用本技能。
## 音频标签控制
@ -172,6 +172,8 @@ description: "文本转语音与语音消息发送技能。当用户想让我说
- 只有`mimo-v2.5-tts`模型支持唱歌模式
- 唱歌请求使用预置音色模型;音色设计、音色复刻均不支持唱歌。用户同时要求引用语音克隆和唱歌时,说明该组合不受支持。
- 如需体验更佳的唱歌风格,必须在目标文本最开头添加 `(唱歌)` 标签,格式为:`(唱歌)歌词`。歌词 建议采用中文,可获得更优合成效果。标签内标识支持以下取值,效果等效:`唱歌`、`sing`、`singing`
## 执行步骤
@ -190,10 +192,46 @@ python3 scripts/voice_message.py --content '这是一条语音消息' --emotion
- Doubao:`content` 写入文本字段;支持的 `emotion` 写入音频情绪参数;`voice` 可覆盖 speaker;其他风格控制会合并到 `context_texts` 辅助信息。
- MiMo V2.5:`content` 写入 `assistant` 消息;`style_prompt`、`voice_prompt`、`context_texts`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect` 会合并为 `user` 风格/音色控制;`audio_tags` 会作为整体标签加到要合成的文本前。
- MiMo 会默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 流式兼容模式并在脚本内封装为 `wav`。
- MiMo 在 `auto_model` 未关闭时,会根据 `voice_prompt` 自动选择 `mimo-v2.5-tts-voicedesign`;如果 `ROBOT_REF_MESSAGE_ID` 指向数据库中 `messages.type = 34` 的语音消息,则脚本会调用客户端接口下载该语音 wav,并自动选择 `mimo-v2.5-tts-voiceclone`。
- MiMo 默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 并在脚本内封装为 24 kHz、单声道 `wav`。普通朗读支持低延迟流式;音色设计、复刻当前在推理完成后以流式格式返回结果。
- 引用语音消息时,按 `messages.id` 查询引用消息并检查 `type = 34`,下载 wav 后选择 `mimo-v2.5-tts-voiceclone`。例如引用一条语音并要求「用这个声音说:晚上好」。也可在后台配置固定的 `voice_clone_audio` 样本。
- 上下文明确指定预置 `voice` 时使用普通朗读模型;`auto_model` 开启时,上下文的 `voice_prompt` 选择音色设计模型。这些明确要求优先于配置中的默认复刻样本和音色描述。
- 引用消息下载接口为 `GET http://127.0.0.1:{ROBOT_WECHAT_CLIENT_PORT}/api/v1/robot/chat/voice/download?message_id={ROBOT_REF_MESSAGE_ID}`,返回 wav 后由脚本封装为 MiMo 需要的 `data:audio/wav;base64,...`。
## MiMo 配置
管理后台文本转语音配置中的 `mimo` 提供以下参数,不需要填写 `model`。模型由脚本根据上下文提取的音色描述、复刻音频等信息自动选择。
```json
{
"base_url": "https://api.xiaomimimo.com/v1",
"api_key": "",
"voice": "mimo_default",
"audio_format": "wav",
"stream": false,
"timeout": 300,
"auto_model": true,
"voice_prompt": "",
"style_prompt": [],
"context_texts": [],
"audio_tags": [],
"emotion": "",
"speaking_rate": "",
"pitch": "",
"volume": "",
"dialect": "",
"voice_clone_audio": "",
"voice_clone_mime_type": "audio/mpeg"
}
```
- `base_url`、`api_key` 用于语音合成请求;留空时沿用聊天接口的对应配置。`timeout` 是请求超时秒数,必须大于 0。
- `voice`、`voice_prompt`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect`、`audio_tags` 是默认语音控制值,上下文明确指定时优先使用上下文参数;`style_prompt` 和 `context_texts` 会与上下文提供的内容合并。
- `audio_format` 控制非流式输出格式;`stream: true` 时使用 `pcm16` 并封装为 `wav`。PCM 采样率按接口协议处理。
- `voice_clone_audio` 支持 MP3/WAV 的 Base64 或音频 data URL,Base64 内容不能超过 10 MB。仅填写 Base64 时,使用 `voice_clone_mime_type` 指定类型:`audio/mpeg`、`audio/mp3` 或 `audio/wav`。引用语音优先于配置样本,格式错误、数据为空或超限时会在调用 MiMo 前报错。
- `auto_model` 默认开启,优先使用上下文的音色要求,再按配置样本、配置音色描述、普通朗读的顺序选择;关闭时使用普通朗读模型。引用语音始终选择音色复刻;唱歌使用普通朗读模型,与引用语音克隆冲突时会明确报错。
协议依据:[小米 MiMo V2.5 语音合成官方文档](https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5)。
## 依赖安装
- 脚本首次运行时会自动创建虚拟环境并安装依赖,无需手动执行。

View File

@ -6,7 +6,9 @@ import argparse
import base64
import gzip
import json
import math
import os
import re
import subprocess
import sys
import tempfile
@ -54,6 +56,8 @@ MIMO_STREAM_AUDIO_FORMAT = "pcm16"
MIMO_PCM_SAMPLE_RATE = 24000
MIMO_VOICE_DESIGN_MODEL = "mimo-v2.5-tts-voicedesign"
MIMO_VOICE_CLONE_MODEL = "mimo-v2.5-tts-voiceclone"
MIMO_VOICE_CLONE_MAX_BASE64_SIZE = 10_000_000
MIMO_VOICE_CLONE_MIME_TYPES = {"audio/mpeg", "audio/mp3", "audio/wav"}
WECHAT_VOICE_MESSAGE_TYPE = 34
MAX_CONTENT_LENGTH = 260
STREAM_END_CODE = 20000000
@ -263,7 +267,8 @@ def _download_referenced_voice_clone(message_id: str) -> str:
)
try:
with urllib.request.urlopen(req, timeout=60) as response:
wav_data = response.read()
max_audio_size = MIMO_VOICE_CLONE_MAX_BASE64_SIZE // 4 * 3
wav_data = response.read(max_audio_size + 1)
except urllib.error.HTTPError as exc:
error_body = exc.read().decode("utf-8", errors="replace")
raise RuntimeError(f"下载引用语音失败,状态码 {exc.code}: {error_body}") from exc
@ -272,6 +277,8 @@ def _download_referenced_voice_clone(message_id: str) -> str:
if not wav_data:
raise RuntimeError("下载引用语音失败: 响应为空")
if len(wav_data) > max_audio_size:
raise RuntimeError("引用语音过大,音色复刻样本的 Base64 不能超过 10 MB")
audio_b64 = base64.b64encode(wav_data).decode("utf-8")
return f"data:audio/wav;base64,{audio_b64}"
@ -282,7 +289,14 @@ def _load_referenced_voice_clone(conn) -> str:
if not ref_message_id:
return ""
message = _query_one(conn, "SELECT * FROM messages WHERE msg_id = %s LIMIT 1", (ref_message_id,))
try:
message_id = int(ref_message_id)
except ValueError:
return ""
if message_id <= 0:
return ""
message = _query_one(conn, "SELECT id, type FROM messages WHERE id = %s LIMIT 1", (message_id,))
if not message:
return ""
@ -294,7 +308,7 @@ def _load_referenced_voice_clone(conn) -> str:
if message_type != WECHAT_VOICE_MESSAGE_TYPE:
return ""
return _download_referenced_voice_clone(ref_message_id)
return _download_referenced_voice_clone(str(message["id"]))
def _parse_cli_params(argv: list[str]) -> dict:
@ -520,18 +534,34 @@ def _config_texts(config: dict, key: str) -> list[str]:
return [text] if text else []
def _mimo_singing_requested(config: dict, params: dict) -> bool:
tags = list(params.get("audio_tags") or _config_texts(config, "audio_tags"))
leading_tags = re.match(
r"^(?:(?:\([^)]*\)|([^)]*)|\[[^\]]*\])\s*)+",
_clean_text(params.get("content")),
)
if leading_tags:
tags.append(leading_tags.group())
return any(re.search(r"唱歌|\bsing(?:ing)?\b", tag, re.IGNORECASE) for tag in tags)
def _resolve_mimo_model(config: dict, params: dict) -> str:
configured_model = _clean_text(config.get("model"))
if _mimo_singing_requested(config, params):
if _clean_text(params.get("voice_clone_audio")):
raise RuntimeError("MiMo 音色复刻不支持唱歌,请改为朗读,或取消引用语音后使用预置音色唱歌")
return DEFAULT_MIMO_MODEL
if _clean_text(params.get("voice_clone_audio")):
return MIMO_VOICE_CLONE_MODEL
if _clean_text(params.get("voice")):
return DEFAULT_MIMO_MODEL
auto_model = _coerce_bool(config.get("auto_model"), True)
if auto_model and _clean_text(params.get("voice_prompt")):
return MIMO_VOICE_DESIGN_MODEL
if auto_model and _clean_text(config.get("voice_clone_audio")):
return MIMO_VOICE_CLONE_MODEL
if auto_model and (_clean_text(params.get("voice_prompt")) or _clean_text(config.get("voice_prompt"))):
if auto_model and _clean_text(config.get("voice_prompt")):
return MIMO_VOICE_DESIGN_MODEL
if configured_model:
return configured_model
return DEFAULT_MIMO_MODEL
@ -542,9 +572,9 @@ def _format_mimo_audio_tags(tags: list[str]) -> str:
return f"({' '.join(cleaned_tags)})"
def _build_mimo_assistant_content(params: dict) -> str:
def _build_mimo_assistant_content(config: dict, params: dict) -> str:
content = _clean_text(params.get("content"))
tags = _format_mimo_audio_tags(params.get("audio_tags") or [])
tags = _format_mimo_audio_tags(params.get("audio_tags") or _config_texts(config, "audio_tags"))
return f"{tags}{content}" if tags else content
@ -580,6 +610,30 @@ def _build_mimo_user_content(config: dict, params: dict, model: str) -> str:
return "\n".join(parts)
def _mimo_voice_clone_data_url(audio: str, mime_type: str) -> str:
encoded = audio
if audio.lower().startswith("data:"):
header, separator, encoded = audio.partition(",")
if not separator or not header.lower().endswith(";base64"):
raise RuntimeError("音色复刻样本必须使用 data:audio/...;base64,... 格式")
mime_type = header[5:-7]
mime_type = mime_type.strip().lower()
if mime_type not in MIMO_VOICE_CLONE_MIME_TYPES:
raise RuntimeError("MiMo 音色复刻仅支持 MP3/WAV 样本,MIME 类型须为 audio/mpeg、audio/mp3 或 audio/wav")
encoded = "".join(encoded.split())
if len(encoded) > MIMO_VOICE_CLONE_MAX_BASE64_SIZE:
raise RuntimeError("音色复刻样本的 Base64 不能超过 10 MB")
try:
audio_bytes = base64.b64decode(encoded, validate=True)
except ValueError as exc:
raise RuntimeError("音色复刻样本不是有效的 Base64 音频数据") from exc
if not audio_bytes:
raise RuntimeError("音色复刻样本不能为空")
return f"data:{mime_type};base64,{encoded}"
def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str:
if model == MIMO_VOICE_DESIGN_MODEL:
return ""
@ -588,14 +642,12 @@ def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str:
voice_clone_audio = _clean_text(params.get("voice_clone_audio")) or _clean_text(config.get("voice_clone_audio"))
if not voice_clone_audio:
raise RuntimeError("mimo 音色复刻模型需要引用一条语音消息或配置 voice_clone_audio")
if voice_clone_audio.startswith("data:"):
return voice_clone_audio
mime_type = (
_clean_text(params.get("voice_clone_mime_type"))
or _clean_text(config.get("voice_clone_mime_type"))
or "audio/mpeg"
)
return f"data:{mime_type};base64,{voice_clone_audio}"
return _mimo_voice_clone_data_url(voice_clone_audio, mime_type)
return _clean_text(params.get("voice")) or _clean_text(config.get("voice")) or DEFAULT_MIMO_VOICE
@ -604,14 +656,14 @@ def _build_mimo_payload(config: dict, params: dict) -> tuple[dict, str, bool]:
model = _resolve_mimo_model(config, params)
stream = _coerce_bool(config.get("stream"), False)
audio_format = MIMO_STREAM_AUDIO_FORMAT if stream else (
_clean_text(config.get("audio_format")) or _clean_text(config.get("format")) or DEFAULT_MIMO_AUDIO_FORMAT
_clean_text(config.get("audio_format")) or DEFAULT_MIMO_AUDIO_FORMAT
)
messages = []
user_content = _build_mimo_user_content(config, params, model)
if user_content or model == MIMO_VOICE_CLONE_MODEL:
messages.append({"role": "user", "content": user_content})
messages.append({"role": "assistant", "content": _build_mimo_assistant_content(params)})
messages.append({"role": "assistant", "content": _build_mimo_assistant_content(config, params)})
audio = {"format": audio_format}
voice = _resolve_mimo_voice(config, params, model)
@ -739,6 +791,13 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]:
if not api_key:
raise RuntimeError("mimo api_key 不能为空")
try:
timeout = float(config.get("timeout", 300))
except (TypeError, ValueError) as exc:
raise RuntimeError("mimo timeout 必须是大于 0 的秒数") from exc
if not math.isfinite(timeout) or timeout <= 0:
raise RuntimeError("mimo timeout 必须是大于 0 的秒数")
# 兼容用户把 base_url 配成不带 /v1 的根地址(如 New API / OneAPI 等网关),
# 避免请求被前端 SPA 兜底返回 index.html。
parsed_base = urllib.parse.urlsplit(base_url)
@ -763,7 +822,7 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]:
)
try:
response = urllib.request.urlopen(req, timeout=300)
response = urllib.request.urlopen(req, timeout=timeout)
except urllib.error.HTTPError as exc:
try:
error_body = _read_response_text(exc)
@ -903,7 +962,7 @@ def main() -> int:
return 1
try:
if tts_model == "mimo":
if enabled and tts_model == "mimo":
voice_clone_audio = _load_referenced_voice_clone(conn)
if voice_clone_audio:
params = dict(params)