diff --git a/skills/voice-message/SKILL.md b/skills/voice-message/SKILL.md index 7d3a6d2..c4b838e 100644 --- a/skills/voice-message/SKILL.md +++ b/skills/voice-message/SKILL.md @@ -107,7 +107,7 @@ description: "文本转语音与语音消息发送技能。当用户想让我说 6. 语速、音高、音量、方言有明确要求时优先填 `speaking_rate`、`pitch`、`volume`、`dialect`;复杂演绎要求放入 `style_prompt`。 7. `audio_tags` 仅用于用户明确要求唱歌、方言、笑声、停顿、深呼吸等标签化控制时;如果用户已把标签写在 `content` 中,不要重复添加。 8. `context_texts` 适合表达上下文、场景、人物状态和补充播报要求。 -9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID` 自动判断并下载引用语音作为复刻样本。 +9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID`(数据库 `messages.id`)自动判断并下载引用语音作为复刻样本。 10. `content` 超过 260 个字符时,不应该调用本技能。 ## 音频标签控制 @@ -172,6 +172,8 @@ description: "文本转语音与语音消息发送技能。当用户想让我说 - 只有`mimo-v2.5-tts`模型支持唱歌模式 +- 唱歌请求使用预置音色模型;音色设计、音色复刻均不支持唱歌。用户同时要求引用语音克隆和唱歌时,说明该组合不受支持。 + - 如需体验更佳的唱歌风格,必须在目标文本最开头添加 `(唱歌)` 标签,格式为:`(唱歌)歌词`。歌词 建议采用中文,可获得更优合成效果。标签内标识支持以下取值,效果等效:`唱歌`、`sing`、`singing` ## 执行步骤 @@ -190,10 +192,46 @@ python3 scripts/voice_message.py --content '这是一条语音消息' --emotion - Doubao:`content` 写入文本字段;支持的 `emotion` 写入音频情绪参数;`voice` 可覆盖 speaker;其他风格控制会合并到 `context_texts` 辅助信息。 - MiMo V2.5:`content` 写入 `assistant` 消息;`style_prompt`、`voice_prompt`、`context_texts`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect` 会合并为 `user` 风格/音色控制;`audio_tags` 会作为整体标签加到要合成的文本前。 -- MiMo 会默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 流式兼容模式并在脚本内封装为 `wav`。 -- MiMo 在 `auto_model` 未关闭时,会根据 `voice_prompt` 自动选择 `mimo-v2.5-tts-voicedesign`;如果 `ROBOT_REF_MESSAGE_ID` 指向数据库中 `messages.type = 34` 的语音消息,则脚本会调用客户端接口下载该语音 wav,并自动选择 `mimo-v2.5-tts-voiceclone`。 +- MiMo 默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 并在脚本内封装为 24 kHz、单声道 `wav`。普通朗读支持低延迟流式;音色设计、复刻当前在推理完成后以流式格式返回结果。 +- 引用语音消息时,按 `messages.id` 查询引用消息并检查 `type = 34`,下载 wav 后选择 `mimo-v2.5-tts-voiceclone`。例如引用一条语音并要求「用这个声音说:晚上好」。也可在后台配置固定的 `voice_clone_audio` 样本。 +- 上下文明确指定预置 `voice` 时使用普通朗读模型;`auto_model` 开启时,上下文的 `voice_prompt` 选择音色设计模型。这些明确要求优先于配置中的默认复刻样本和音色描述。 - 引用消息下载接口为 `GET http://127.0.0.1:{ROBOT_WECHAT_CLIENT_PORT}/api/v1/robot/chat/voice/download?message_id={ROBOT_REF_MESSAGE_ID}`,返回 wav 后由脚本封装为 MiMo 需要的 `data:audio/wav;base64,...`。 +## MiMo 配置 + +管理后台文本转语音配置中的 `mimo` 提供以下参数,不需要填写 `model`。模型由脚本根据上下文提取的音色描述、复刻音频等信息自动选择。 + +```json +{ + "base_url": "https://api.xiaomimimo.com/v1", + "api_key": "", + "voice": "mimo_default", + "audio_format": "wav", + "stream": false, + "timeout": 300, + "auto_model": true, + "voice_prompt": "", + "style_prompt": [], + "context_texts": [], + "audio_tags": [], + "emotion": "", + "speaking_rate": "", + "pitch": "", + "volume": "", + "dialect": "", + "voice_clone_audio": "", + "voice_clone_mime_type": "audio/mpeg" +} +``` + +- `base_url`、`api_key` 用于语音合成请求;留空时沿用聊天接口的对应配置。`timeout` 是请求超时秒数,必须大于 0。 +- `voice`、`voice_prompt`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect`、`audio_tags` 是默认语音控制值,上下文明确指定时优先使用上下文参数;`style_prompt` 和 `context_texts` 会与上下文提供的内容合并。 +- `audio_format` 控制非流式输出格式;`stream: true` 时使用 `pcm16` 并封装为 `wav`。PCM 采样率按接口协议处理。 +- `voice_clone_audio` 支持 MP3/WAV 的 Base64 或音频 data URL,Base64 内容不能超过 10 MB。仅填写 Base64 时,使用 `voice_clone_mime_type` 指定类型:`audio/mpeg`、`audio/mp3` 或 `audio/wav`。引用语音优先于配置样本,格式错误、数据为空或超限时会在调用 MiMo 前报错。 +- `auto_model` 默认开启,优先使用上下文的音色要求,再按配置样本、配置音色描述、普通朗读的顺序选择;关闭时使用普通朗读模型。引用语音始终选择音色复刻;唱歌使用普通朗读模型,与引用语音克隆冲突时会明确报错。 + +协议依据:[小米 MiMo V2.5 语音合成官方文档](https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5)。 + ## 依赖安装 - 脚本首次运行时会自动创建虚拟环境并安装依赖,无需手动执行。 diff --git a/skills/voice-message/scripts/voice_message.py b/skills/voice-message/scripts/voice_message.py index c178b39..7cf0e80 100644 --- a/skills/voice-message/scripts/voice_message.py +++ b/skills/voice-message/scripts/voice_message.py @@ -6,7 +6,9 @@ import argparse import base64 import gzip import json +import math import os +import re import subprocess import sys import tempfile @@ -54,6 +56,8 @@ MIMO_STREAM_AUDIO_FORMAT = "pcm16" MIMO_PCM_SAMPLE_RATE = 24000 MIMO_VOICE_DESIGN_MODEL = "mimo-v2.5-tts-voicedesign" MIMO_VOICE_CLONE_MODEL = "mimo-v2.5-tts-voiceclone" +MIMO_VOICE_CLONE_MAX_BASE64_SIZE = 10_000_000 +MIMO_VOICE_CLONE_MIME_TYPES = {"audio/mpeg", "audio/mp3", "audio/wav"} WECHAT_VOICE_MESSAGE_TYPE = 34 MAX_CONTENT_LENGTH = 260 STREAM_END_CODE = 20000000 @@ -263,7 +267,8 @@ def _download_referenced_voice_clone(message_id: str) -> str: ) try: with urllib.request.urlopen(req, timeout=60) as response: - wav_data = response.read() + max_audio_size = MIMO_VOICE_CLONE_MAX_BASE64_SIZE // 4 * 3 + wav_data = response.read(max_audio_size + 1) except urllib.error.HTTPError as exc: error_body = exc.read().decode("utf-8", errors="replace") raise RuntimeError(f"下载引用语音失败,状态码 {exc.code}: {error_body}") from exc @@ -272,6 +277,8 @@ def _download_referenced_voice_clone(message_id: str) -> str: if not wav_data: raise RuntimeError("下载引用语音失败: 响应为空") + if len(wav_data) > max_audio_size: + raise RuntimeError("引用语音过大,音色复刻样本的 Base64 不能超过 10 MB") audio_b64 = base64.b64encode(wav_data).decode("utf-8") return f"data:audio/wav;base64,{audio_b64}" @@ -282,7 +289,14 @@ def _load_referenced_voice_clone(conn) -> str: if not ref_message_id: return "" - message = _query_one(conn, "SELECT * FROM messages WHERE msg_id = %s LIMIT 1", (ref_message_id,)) + try: + message_id = int(ref_message_id) + except ValueError: + return "" + if message_id <= 0: + return "" + + message = _query_one(conn, "SELECT id, type FROM messages WHERE id = %s LIMIT 1", (message_id,)) if not message: return "" @@ -294,7 +308,7 @@ def _load_referenced_voice_clone(conn) -> str: if message_type != WECHAT_VOICE_MESSAGE_TYPE: return "" - return _download_referenced_voice_clone(ref_message_id) + return _download_referenced_voice_clone(str(message["id"])) def _parse_cli_params(argv: list[str]) -> dict: @@ -520,18 +534,34 @@ def _config_texts(config: dict, key: str) -> list[str]: return [text] if text else [] +def _mimo_singing_requested(config: dict, params: dict) -> bool: + tags = list(params.get("audio_tags") or _config_texts(config, "audio_tags")) + leading_tags = re.match( + r"^(?:(?:\([^)]*\)|([^)]*)|\[[^\]]*\])\s*)+", + _clean_text(params.get("content")), + ) + if leading_tags: + tags.append(leading_tags.group()) + return any(re.search(r"唱歌|\bsing(?:ing)?\b", tag, re.IGNORECASE) for tag in tags) + + def _resolve_mimo_model(config: dict, params: dict) -> str: - configured_model = _clean_text(config.get("model")) + if _mimo_singing_requested(config, params): + if _clean_text(params.get("voice_clone_audio")): + raise RuntimeError("MiMo 音色复刻不支持唱歌,请改为朗读,或取消引用语音后使用预置音色唱歌") + return DEFAULT_MIMO_MODEL if _clean_text(params.get("voice_clone_audio")): return MIMO_VOICE_CLONE_MODEL + if _clean_text(params.get("voice")): + return DEFAULT_MIMO_MODEL auto_model = _coerce_bool(config.get("auto_model"), True) + if auto_model and _clean_text(params.get("voice_prompt")): + return MIMO_VOICE_DESIGN_MODEL if auto_model and _clean_text(config.get("voice_clone_audio")): return MIMO_VOICE_CLONE_MODEL - if auto_model and (_clean_text(params.get("voice_prompt")) or _clean_text(config.get("voice_prompt"))): + if auto_model and _clean_text(config.get("voice_prompt")): return MIMO_VOICE_DESIGN_MODEL - if configured_model: - return configured_model return DEFAULT_MIMO_MODEL @@ -542,9 +572,9 @@ def _format_mimo_audio_tags(tags: list[str]) -> str: return f"({' '.join(cleaned_tags)})" -def _build_mimo_assistant_content(params: dict) -> str: +def _build_mimo_assistant_content(config: dict, params: dict) -> str: content = _clean_text(params.get("content")) - tags = _format_mimo_audio_tags(params.get("audio_tags") or []) + tags = _format_mimo_audio_tags(params.get("audio_tags") or _config_texts(config, "audio_tags")) return f"{tags}{content}" if tags else content @@ -580,6 +610,30 @@ def _build_mimo_user_content(config: dict, params: dict, model: str) -> str: return "\n".join(parts) +def _mimo_voice_clone_data_url(audio: str, mime_type: str) -> str: + encoded = audio + if audio.lower().startswith("data:"): + header, separator, encoded = audio.partition(",") + if not separator or not header.lower().endswith(";base64"): + raise RuntimeError("音色复刻样本必须使用 data:audio/...;base64,... 格式") + mime_type = header[5:-7] + + mime_type = mime_type.strip().lower() + if mime_type not in MIMO_VOICE_CLONE_MIME_TYPES: + raise RuntimeError("MiMo 音色复刻仅支持 MP3/WAV 样本,MIME 类型须为 audio/mpeg、audio/mp3 或 audio/wav") + + encoded = "".join(encoded.split()) + if len(encoded) > MIMO_VOICE_CLONE_MAX_BASE64_SIZE: + raise RuntimeError("音色复刻样本的 Base64 不能超过 10 MB") + try: + audio_bytes = base64.b64decode(encoded, validate=True) + except ValueError as exc: + raise RuntimeError("音色复刻样本不是有效的 Base64 音频数据") from exc + if not audio_bytes: + raise RuntimeError("音色复刻样本不能为空") + return f"data:{mime_type};base64,{encoded}" + + def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str: if model == MIMO_VOICE_DESIGN_MODEL: return "" @@ -588,14 +642,12 @@ def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str: voice_clone_audio = _clean_text(params.get("voice_clone_audio")) or _clean_text(config.get("voice_clone_audio")) if not voice_clone_audio: raise RuntimeError("mimo 音色复刻模型需要引用一条语音消息或配置 voice_clone_audio") - if voice_clone_audio.startswith("data:"): - return voice_clone_audio mime_type = ( _clean_text(params.get("voice_clone_mime_type")) or _clean_text(config.get("voice_clone_mime_type")) or "audio/mpeg" ) - return f"data:{mime_type};base64,{voice_clone_audio}" + return _mimo_voice_clone_data_url(voice_clone_audio, mime_type) return _clean_text(params.get("voice")) or _clean_text(config.get("voice")) or DEFAULT_MIMO_VOICE @@ -604,14 +656,14 @@ def _build_mimo_payload(config: dict, params: dict) -> tuple[dict, str, bool]: model = _resolve_mimo_model(config, params) stream = _coerce_bool(config.get("stream"), False) audio_format = MIMO_STREAM_AUDIO_FORMAT if stream else ( - _clean_text(config.get("audio_format")) or _clean_text(config.get("format")) or DEFAULT_MIMO_AUDIO_FORMAT + _clean_text(config.get("audio_format")) or DEFAULT_MIMO_AUDIO_FORMAT ) messages = [] user_content = _build_mimo_user_content(config, params, model) if user_content or model == MIMO_VOICE_CLONE_MODEL: messages.append({"role": "user", "content": user_content}) - messages.append({"role": "assistant", "content": _build_mimo_assistant_content(params)}) + messages.append({"role": "assistant", "content": _build_mimo_assistant_content(config, params)}) audio = {"format": audio_format} voice = _resolve_mimo_voice(config, params, model) @@ -739,6 +791,13 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]: if not api_key: raise RuntimeError("mimo api_key 不能为空") + try: + timeout = float(config.get("timeout", 300)) + except (TypeError, ValueError) as exc: + raise RuntimeError("mimo timeout 必须是大于 0 的秒数") from exc + if not math.isfinite(timeout) or timeout <= 0: + raise RuntimeError("mimo timeout 必须是大于 0 的秒数") + # 兼容用户把 base_url 配成不带 /v1 的根地址(如 New API / OneAPI 等网关), # 避免请求被前端 SPA 兜底返回 index.html。 parsed_base = urllib.parse.urlsplit(base_url) @@ -763,7 +822,7 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]: ) try: - response = urllib.request.urlopen(req, timeout=300) + response = urllib.request.urlopen(req, timeout=timeout) except urllib.error.HTTPError as exc: try: error_body = _read_response_text(exc) @@ -903,7 +962,7 @@ def main() -> int: return 1 try: - if tts_model == "mimo": + if enabled and tts_model == "mimo": voice_clone_audio = _load_referenced_voice_clone(conn) if voice_clone_audio: params = dict(params)