fix: 优化小米语音
This commit is contained in:
parent
999da6083c
commit
153e90cb89
@ -107,7 +107,7 @@ description: "文本转语音与语音消息发送技能。当用户想让我说
|
|||||||
6. 语速、音高、音量、方言有明确要求时优先填 `speaking_rate`、`pitch`、`volume`、`dialect`;复杂演绎要求放入 `style_prompt`。
|
6. 语速、音高、音量、方言有明确要求时优先填 `speaking_rate`、`pitch`、`volume`、`dialect`;复杂演绎要求放入 `style_prompt`。
|
||||||
7. `audio_tags` 仅用于用户明确要求唱歌、方言、笑声、停顿、深呼吸等标签化控制时;如果用户已把标签写在 `content` 中,不要重复添加。
|
7. `audio_tags` 仅用于用户明确要求唱歌、方言、笑声、停顿、深呼吸等标签化控制时;如果用户已把标签写在 `content` 中,不要重复添加。
|
||||||
8. `context_texts` 适合表达上下文、场景、人物状态和补充播报要求。
|
8. `context_texts` 适合表达上下文、场景、人物状态和补充播报要求。
|
||||||
9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID` 自动判断并下载引用语音作为复刻样本。
|
9. 不要传递音色复刻音频参数。若当前消息引用了一条语音消息,脚本会通过 `ROBOT_REF_MESSAGE_ID`(数据库 `messages.id`)自动判断并下载引用语音作为复刻样本。
|
||||||
10. `content` 超过 260 个字符时,不应该调用本技能。
|
10. `content` 超过 260 个字符时,不应该调用本技能。
|
||||||
|
|
||||||
## 音频标签控制
|
## 音频标签控制
|
||||||
@ -172,6 +172,8 @@ description: "文本转语音与语音消息发送技能。当用户想让我说
|
|||||||
|
|
||||||
- 只有`mimo-v2.5-tts`模型支持唱歌模式
|
- 只有`mimo-v2.5-tts`模型支持唱歌模式
|
||||||
|
|
||||||
|
- 唱歌请求使用预置音色模型;音色设计、音色复刻均不支持唱歌。用户同时要求引用语音克隆和唱歌时,说明该组合不受支持。
|
||||||
|
|
||||||
- 如需体验更佳的唱歌风格,必须在目标文本最开头添加 `(唱歌)` 标签,格式为:`(唱歌)歌词`。歌词 建议采用中文,可获得更优合成效果。标签内标识支持以下取值,效果等效:`唱歌`、`sing`、`singing`
|
- 如需体验更佳的唱歌风格,必须在目标文本最开头添加 `(唱歌)` 标签,格式为:`(唱歌)歌词`。歌词 建议采用中文,可获得更优合成效果。标签内标识支持以下取值,效果等效:`唱歌`、`sing`、`singing`
|
||||||
|
|
||||||
## 执行步骤
|
## 执行步骤
|
||||||
@ -190,10 +192,46 @@ python3 scripts/voice_message.py --content '这是一条语音消息' --emotion
|
|||||||
|
|
||||||
- Doubao:`content` 写入文本字段;支持的 `emotion` 写入音频情绪参数;`voice` 可覆盖 speaker;其他风格控制会合并到 `context_texts` 辅助信息。
|
- Doubao:`content` 写入文本字段;支持的 `emotion` 写入音频情绪参数;`voice` 可覆盖 speaker;其他风格控制会合并到 `context_texts` 辅助信息。
|
||||||
- MiMo V2.5:`content` 写入 `assistant` 消息;`style_prompt`、`voice_prompt`、`context_texts`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect` 会合并为 `user` 风格/音色控制;`audio_tags` 会作为整体标签加到要合成的文本前。
|
- MiMo V2.5:`content` 写入 `assistant` 消息;`style_prompt`、`voice_prompt`、`context_texts`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect` 会合并为 `user` 风格/音色控制;`audio_tags` 会作为整体标签加到要合成的文本前。
|
||||||
- MiMo 会默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 流式兼容模式并在脚本内封装为 `wav`。
|
- MiMo 默认使用非流式 `wav` 输出;配置中 `stream: true` 时使用 `pcm16` 并在脚本内封装为 24 kHz、单声道 `wav`。普通朗读支持低延迟流式;音色设计、复刻当前在推理完成后以流式格式返回结果。
|
||||||
- MiMo 在 `auto_model` 未关闭时,会根据 `voice_prompt` 自动选择 `mimo-v2.5-tts-voicedesign`;如果 `ROBOT_REF_MESSAGE_ID` 指向数据库中 `messages.type = 34` 的语音消息,则脚本会调用客户端接口下载该语音 wav,并自动选择 `mimo-v2.5-tts-voiceclone`。
|
- 引用语音消息时,按 `messages.id` 查询引用消息并检查 `type = 34`,下载 wav 后选择 `mimo-v2.5-tts-voiceclone`。例如引用一条语音并要求「用这个声音说:晚上好」。也可在后台配置固定的 `voice_clone_audio` 样本。
|
||||||
|
- 上下文明确指定预置 `voice` 时使用普通朗读模型;`auto_model` 开启时,上下文的 `voice_prompt` 选择音色设计模型。这些明确要求优先于配置中的默认复刻样本和音色描述。
|
||||||
- 引用消息下载接口为 `GET http://127.0.0.1:{ROBOT_WECHAT_CLIENT_PORT}/api/v1/robot/chat/voice/download?message_id={ROBOT_REF_MESSAGE_ID}`,返回 wav 后由脚本封装为 MiMo 需要的 `data:audio/wav;base64,...`。
|
- 引用消息下载接口为 `GET http://127.0.0.1:{ROBOT_WECHAT_CLIENT_PORT}/api/v1/robot/chat/voice/download?message_id={ROBOT_REF_MESSAGE_ID}`,返回 wav 后由脚本封装为 MiMo 需要的 `data:audio/wav;base64,...`。
|
||||||
|
|
||||||
|
## MiMo 配置
|
||||||
|
|
||||||
|
管理后台文本转语音配置中的 `mimo` 提供以下参数,不需要填写 `model`。模型由脚本根据上下文提取的音色描述、复刻音频等信息自动选择。
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"base_url": "https://api.xiaomimimo.com/v1",
|
||||||
|
"api_key": "",
|
||||||
|
"voice": "mimo_default",
|
||||||
|
"audio_format": "wav",
|
||||||
|
"stream": false,
|
||||||
|
"timeout": 300,
|
||||||
|
"auto_model": true,
|
||||||
|
"voice_prompt": "",
|
||||||
|
"style_prompt": [],
|
||||||
|
"context_texts": [],
|
||||||
|
"audio_tags": [],
|
||||||
|
"emotion": "",
|
||||||
|
"speaking_rate": "",
|
||||||
|
"pitch": "",
|
||||||
|
"volume": "",
|
||||||
|
"dialect": "",
|
||||||
|
"voice_clone_audio": "",
|
||||||
|
"voice_clone_mime_type": "audio/mpeg"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
- `base_url`、`api_key` 用于语音合成请求;留空时沿用聊天接口的对应配置。`timeout` 是请求超时秒数,必须大于 0。
|
||||||
|
- `voice`、`voice_prompt`、`emotion`、`speaking_rate`、`pitch`、`volume`、`dialect`、`audio_tags` 是默认语音控制值,上下文明确指定时优先使用上下文参数;`style_prompt` 和 `context_texts` 会与上下文提供的内容合并。
|
||||||
|
- `audio_format` 控制非流式输出格式;`stream: true` 时使用 `pcm16` 并封装为 `wav`。PCM 采样率按接口协议处理。
|
||||||
|
- `voice_clone_audio` 支持 MP3/WAV 的 Base64 或音频 data URL,Base64 内容不能超过 10 MB。仅填写 Base64 时,使用 `voice_clone_mime_type` 指定类型:`audio/mpeg`、`audio/mp3` 或 `audio/wav`。引用语音优先于配置样本,格式错误、数据为空或超限时会在调用 MiMo 前报错。
|
||||||
|
- `auto_model` 默认开启,优先使用上下文的音色要求,再按配置样本、配置音色描述、普通朗读的顺序选择;关闭时使用普通朗读模型。引用语音始终选择音色复刻;唱歌使用普通朗读模型,与引用语音克隆冲突时会明确报错。
|
||||||
|
|
||||||
|
协议依据:[小米 MiMo V2.5 语音合成官方文档](https://mimo.mi.com/docs/zh-CN/quick-start/usage-guide/audio/speech-synthesis-v2.5)。
|
||||||
|
|
||||||
## 依赖安装
|
## 依赖安装
|
||||||
|
|
||||||
- 脚本首次运行时会自动创建虚拟环境并安装依赖,无需手动执行。
|
- 脚本首次运行时会自动创建虚拟环境并安装依赖,无需手动执行。
|
||||||
|
|||||||
@ -6,7 +6,9 @@ import argparse
|
|||||||
import base64
|
import base64
|
||||||
import gzip
|
import gzip
|
||||||
import json
|
import json
|
||||||
|
import math
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
@ -54,6 +56,8 @@ MIMO_STREAM_AUDIO_FORMAT = "pcm16"
|
|||||||
MIMO_PCM_SAMPLE_RATE = 24000
|
MIMO_PCM_SAMPLE_RATE = 24000
|
||||||
MIMO_VOICE_DESIGN_MODEL = "mimo-v2.5-tts-voicedesign"
|
MIMO_VOICE_DESIGN_MODEL = "mimo-v2.5-tts-voicedesign"
|
||||||
MIMO_VOICE_CLONE_MODEL = "mimo-v2.5-tts-voiceclone"
|
MIMO_VOICE_CLONE_MODEL = "mimo-v2.5-tts-voiceclone"
|
||||||
|
MIMO_VOICE_CLONE_MAX_BASE64_SIZE = 10_000_000
|
||||||
|
MIMO_VOICE_CLONE_MIME_TYPES = {"audio/mpeg", "audio/mp3", "audio/wav"}
|
||||||
WECHAT_VOICE_MESSAGE_TYPE = 34
|
WECHAT_VOICE_MESSAGE_TYPE = 34
|
||||||
MAX_CONTENT_LENGTH = 260
|
MAX_CONTENT_LENGTH = 260
|
||||||
STREAM_END_CODE = 20000000
|
STREAM_END_CODE = 20000000
|
||||||
@ -263,7 +267,8 @@ def _download_referenced_voice_clone(message_id: str) -> str:
|
|||||||
)
|
)
|
||||||
try:
|
try:
|
||||||
with urllib.request.urlopen(req, timeout=60) as response:
|
with urllib.request.urlopen(req, timeout=60) as response:
|
||||||
wav_data = response.read()
|
max_audio_size = MIMO_VOICE_CLONE_MAX_BASE64_SIZE // 4 * 3
|
||||||
|
wav_data = response.read(max_audio_size + 1)
|
||||||
except urllib.error.HTTPError as exc:
|
except urllib.error.HTTPError as exc:
|
||||||
error_body = exc.read().decode("utf-8", errors="replace")
|
error_body = exc.read().decode("utf-8", errors="replace")
|
||||||
raise RuntimeError(f"下载引用语音失败,状态码 {exc.code}: {error_body}") from exc
|
raise RuntimeError(f"下载引用语音失败,状态码 {exc.code}: {error_body}") from exc
|
||||||
@ -272,6 +277,8 @@ def _download_referenced_voice_clone(message_id: str) -> str:
|
|||||||
|
|
||||||
if not wav_data:
|
if not wav_data:
|
||||||
raise RuntimeError("下载引用语音失败: 响应为空")
|
raise RuntimeError("下载引用语音失败: 响应为空")
|
||||||
|
if len(wav_data) > max_audio_size:
|
||||||
|
raise RuntimeError("引用语音过大,音色复刻样本的 Base64 不能超过 10 MB")
|
||||||
|
|
||||||
audio_b64 = base64.b64encode(wav_data).decode("utf-8")
|
audio_b64 = base64.b64encode(wav_data).decode("utf-8")
|
||||||
return f"data:audio/wav;base64,{audio_b64}"
|
return f"data:audio/wav;base64,{audio_b64}"
|
||||||
@ -282,7 +289,14 @@ def _load_referenced_voice_clone(conn) -> str:
|
|||||||
if not ref_message_id:
|
if not ref_message_id:
|
||||||
return ""
|
return ""
|
||||||
|
|
||||||
message = _query_one(conn, "SELECT * FROM messages WHERE msg_id = %s LIMIT 1", (ref_message_id,))
|
try:
|
||||||
|
message_id = int(ref_message_id)
|
||||||
|
except ValueError:
|
||||||
|
return ""
|
||||||
|
if message_id <= 0:
|
||||||
|
return ""
|
||||||
|
|
||||||
|
message = _query_one(conn, "SELECT id, type FROM messages WHERE id = %s LIMIT 1", (message_id,))
|
||||||
if not message:
|
if not message:
|
||||||
return ""
|
return ""
|
||||||
|
|
||||||
@ -294,7 +308,7 @@ def _load_referenced_voice_clone(conn) -> str:
|
|||||||
if message_type != WECHAT_VOICE_MESSAGE_TYPE:
|
if message_type != WECHAT_VOICE_MESSAGE_TYPE:
|
||||||
return ""
|
return ""
|
||||||
|
|
||||||
return _download_referenced_voice_clone(ref_message_id)
|
return _download_referenced_voice_clone(str(message["id"]))
|
||||||
|
|
||||||
|
|
||||||
def _parse_cli_params(argv: list[str]) -> dict:
|
def _parse_cli_params(argv: list[str]) -> dict:
|
||||||
@ -520,18 +534,34 @@ def _config_texts(config: dict, key: str) -> list[str]:
|
|||||||
return [text] if text else []
|
return [text] if text else []
|
||||||
|
|
||||||
|
|
||||||
|
def _mimo_singing_requested(config: dict, params: dict) -> bool:
|
||||||
|
tags = list(params.get("audio_tags") or _config_texts(config, "audio_tags"))
|
||||||
|
leading_tags = re.match(
|
||||||
|
r"^(?:(?:\([^)]*\)|([^)]*)|\[[^\]]*\])\s*)+",
|
||||||
|
_clean_text(params.get("content")),
|
||||||
|
)
|
||||||
|
if leading_tags:
|
||||||
|
tags.append(leading_tags.group())
|
||||||
|
return any(re.search(r"唱歌|\bsing(?:ing)?\b", tag, re.IGNORECASE) for tag in tags)
|
||||||
|
|
||||||
|
|
||||||
def _resolve_mimo_model(config: dict, params: dict) -> str:
|
def _resolve_mimo_model(config: dict, params: dict) -> str:
|
||||||
configured_model = _clean_text(config.get("model"))
|
if _mimo_singing_requested(config, params):
|
||||||
|
if _clean_text(params.get("voice_clone_audio")):
|
||||||
|
raise RuntimeError("MiMo 音色复刻不支持唱歌,请改为朗读,或取消引用语音后使用预置音色唱歌")
|
||||||
|
return DEFAULT_MIMO_MODEL
|
||||||
if _clean_text(params.get("voice_clone_audio")):
|
if _clean_text(params.get("voice_clone_audio")):
|
||||||
return MIMO_VOICE_CLONE_MODEL
|
return MIMO_VOICE_CLONE_MODEL
|
||||||
|
if _clean_text(params.get("voice")):
|
||||||
|
return DEFAULT_MIMO_MODEL
|
||||||
|
|
||||||
auto_model = _coerce_bool(config.get("auto_model"), True)
|
auto_model = _coerce_bool(config.get("auto_model"), True)
|
||||||
|
if auto_model and _clean_text(params.get("voice_prompt")):
|
||||||
|
return MIMO_VOICE_DESIGN_MODEL
|
||||||
if auto_model and _clean_text(config.get("voice_clone_audio")):
|
if auto_model and _clean_text(config.get("voice_clone_audio")):
|
||||||
return MIMO_VOICE_CLONE_MODEL
|
return MIMO_VOICE_CLONE_MODEL
|
||||||
if auto_model and (_clean_text(params.get("voice_prompt")) or _clean_text(config.get("voice_prompt"))):
|
if auto_model and _clean_text(config.get("voice_prompt")):
|
||||||
return MIMO_VOICE_DESIGN_MODEL
|
return MIMO_VOICE_DESIGN_MODEL
|
||||||
if configured_model:
|
|
||||||
return configured_model
|
|
||||||
return DEFAULT_MIMO_MODEL
|
return DEFAULT_MIMO_MODEL
|
||||||
|
|
||||||
|
|
||||||
@ -542,9 +572,9 @@ def _format_mimo_audio_tags(tags: list[str]) -> str:
|
|||||||
return f"({' '.join(cleaned_tags)})"
|
return f"({' '.join(cleaned_tags)})"
|
||||||
|
|
||||||
|
|
||||||
def _build_mimo_assistant_content(params: dict) -> str:
|
def _build_mimo_assistant_content(config: dict, params: dict) -> str:
|
||||||
content = _clean_text(params.get("content"))
|
content = _clean_text(params.get("content"))
|
||||||
tags = _format_mimo_audio_tags(params.get("audio_tags") or [])
|
tags = _format_mimo_audio_tags(params.get("audio_tags") or _config_texts(config, "audio_tags"))
|
||||||
return f"{tags}{content}" if tags else content
|
return f"{tags}{content}" if tags else content
|
||||||
|
|
||||||
|
|
||||||
@ -580,6 +610,30 @@ def _build_mimo_user_content(config: dict, params: dict, model: str) -> str:
|
|||||||
return "\n".join(parts)
|
return "\n".join(parts)
|
||||||
|
|
||||||
|
|
||||||
|
def _mimo_voice_clone_data_url(audio: str, mime_type: str) -> str:
|
||||||
|
encoded = audio
|
||||||
|
if audio.lower().startswith("data:"):
|
||||||
|
header, separator, encoded = audio.partition(",")
|
||||||
|
if not separator or not header.lower().endswith(";base64"):
|
||||||
|
raise RuntimeError("音色复刻样本必须使用 data:audio/...;base64,... 格式")
|
||||||
|
mime_type = header[5:-7]
|
||||||
|
|
||||||
|
mime_type = mime_type.strip().lower()
|
||||||
|
if mime_type not in MIMO_VOICE_CLONE_MIME_TYPES:
|
||||||
|
raise RuntimeError("MiMo 音色复刻仅支持 MP3/WAV 样本,MIME 类型须为 audio/mpeg、audio/mp3 或 audio/wav")
|
||||||
|
|
||||||
|
encoded = "".join(encoded.split())
|
||||||
|
if len(encoded) > MIMO_VOICE_CLONE_MAX_BASE64_SIZE:
|
||||||
|
raise RuntimeError("音色复刻样本的 Base64 不能超过 10 MB")
|
||||||
|
try:
|
||||||
|
audio_bytes = base64.b64decode(encoded, validate=True)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise RuntimeError("音色复刻样本不是有效的 Base64 音频数据") from exc
|
||||||
|
if not audio_bytes:
|
||||||
|
raise RuntimeError("音色复刻样本不能为空")
|
||||||
|
return f"data:{mime_type};base64,{encoded}"
|
||||||
|
|
||||||
|
|
||||||
def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str:
|
def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str:
|
||||||
if model == MIMO_VOICE_DESIGN_MODEL:
|
if model == MIMO_VOICE_DESIGN_MODEL:
|
||||||
return ""
|
return ""
|
||||||
@ -588,14 +642,12 @@ def _resolve_mimo_voice(config: dict, params: dict, model: str) -> str:
|
|||||||
voice_clone_audio = _clean_text(params.get("voice_clone_audio")) or _clean_text(config.get("voice_clone_audio"))
|
voice_clone_audio = _clean_text(params.get("voice_clone_audio")) or _clean_text(config.get("voice_clone_audio"))
|
||||||
if not voice_clone_audio:
|
if not voice_clone_audio:
|
||||||
raise RuntimeError("mimo 音色复刻模型需要引用一条语音消息或配置 voice_clone_audio")
|
raise RuntimeError("mimo 音色复刻模型需要引用一条语音消息或配置 voice_clone_audio")
|
||||||
if voice_clone_audio.startswith("data:"):
|
|
||||||
return voice_clone_audio
|
|
||||||
mime_type = (
|
mime_type = (
|
||||||
_clean_text(params.get("voice_clone_mime_type"))
|
_clean_text(params.get("voice_clone_mime_type"))
|
||||||
or _clean_text(config.get("voice_clone_mime_type"))
|
or _clean_text(config.get("voice_clone_mime_type"))
|
||||||
or "audio/mpeg"
|
or "audio/mpeg"
|
||||||
)
|
)
|
||||||
return f"data:{mime_type};base64,{voice_clone_audio}"
|
return _mimo_voice_clone_data_url(voice_clone_audio, mime_type)
|
||||||
|
|
||||||
return _clean_text(params.get("voice")) or _clean_text(config.get("voice")) or DEFAULT_MIMO_VOICE
|
return _clean_text(params.get("voice")) or _clean_text(config.get("voice")) or DEFAULT_MIMO_VOICE
|
||||||
|
|
||||||
@ -604,14 +656,14 @@ def _build_mimo_payload(config: dict, params: dict) -> tuple[dict, str, bool]:
|
|||||||
model = _resolve_mimo_model(config, params)
|
model = _resolve_mimo_model(config, params)
|
||||||
stream = _coerce_bool(config.get("stream"), False)
|
stream = _coerce_bool(config.get("stream"), False)
|
||||||
audio_format = MIMO_STREAM_AUDIO_FORMAT if stream else (
|
audio_format = MIMO_STREAM_AUDIO_FORMAT if stream else (
|
||||||
_clean_text(config.get("audio_format")) or _clean_text(config.get("format")) or DEFAULT_MIMO_AUDIO_FORMAT
|
_clean_text(config.get("audio_format")) or DEFAULT_MIMO_AUDIO_FORMAT
|
||||||
)
|
)
|
||||||
|
|
||||||
messages = []
|
messages = []
|
||||||
user_content = _build_mimo_user_content(config, params, model)
|
user_content = _build_mimo_user_content(config, params, model)
|
||||||
if user_content or model == MIMO_VOICE_CLONE_MODEL:
|
if user_content or model == MIMO_VOICE_CLONE_MODEL:
|
||||||
messages.append({"role": "user", "content": user_content})
|
messages.append({"role": "user", "content": user_content})
|
||||||
messages.append({"role": "assistant", "content": _build_mimo_assistant_content(params)})
|
messages.append({"role": "assistant", "content": _build_mimo_assistant_content(config, params)})
|
||||||
|
|
||||||
audio = {"format": audio_format}
|
audio = {"format": audio_format}
|
||||||
voice = _resolve_mimo_voice(config, params, model)
|
voice = _resolve_mimo_voice(config, params, model)
|
||||||
@ -739,6 +791,13 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]:
|
|||||||
if not api_key:
|
if not api_key:
|
||||||
raise RuntimeError("mimo api_key 不能为空")
|
raise RuntimeError("mimo api_key 不能为空")
|
||||||
|
|
||||||
|
try:
|
||||||
|
timeout = float(config.get("timeout", 300))
|
||||||
|
except (TypeError, ValueError) as exc:
|
||||||
|
raise RuntimeError("mimo timeout 必须是大于 0 的秒数") from exc
|
||||||
|
if not math.isfinite(timeout) or timeout <= 0:
|
||||||
|
raise RuntimeError("mimo timeout 必须是大于 0 的秒数")
|
||||||
|
|
||||||
# 兼容用户把 base_url 配成不带 /v1 的根地址(如 New API / OneAPI 等网关),
|
# 兼容用户把 base_url 配成不带 /v1 的根地址(如 New API / OneAPI 等网关),
|
||||||
# 避免请求被前端 SPA 兜底返回 index.html。
|
# 避免请求被前端 SPA 兜底返回 index.html。
|
||||||
parsed_base = urllib.parse.urlsplit(base_url)
|
parsed_base = urllib.parse.urlsplit(base_url)
|
||||||
@ -763,7 +822,7 @@ def synthesize_audio_mimo(config: dict, params: dict) -> tuple[bytes, str]:
|
|||||||
)
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
response = urllib.request.urlopen(req, timeout=300)
|
response = urllib.request.urlopen(req, timeout=timeout)
|
||||||
except urllib.error.HTTPError as exc:
|
except urllib.error.HTTPError as exc:
|
||||||
try:
|
try:
|
||||||
error_body = _read_response_text(exc)
|
error_body = _read_response_text(exc)
|
||||||
@ -903,7 +962,7 @@ def main() -> int:
|
|||||||
return 1
|
return 1
|
||||||
|
|
||||||
try:
|
try:
|
||||||
if tts_model == "mimo":
|
if enabled and tts_model == "mimo":
|
||||||
voice_clone_audio = _load_referenced_voice_clone(conn)
|
voice_clone_audio = _load_referenced_voice_clone(conn)
|
||||||
if voice_clone_audio:
|
if voice_clone_audio:
|
||||||
params = dict(params)
|
params = dict(params)
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user