package tts import ( "bytes" "context" "encoding/base64" "encoding/json" "fmt" "io" "net/http" "strings" "time" "github.com/hhs/camtalk/internal/trace" "github.com/hhs/camtalk/internal/util" "go.uber.org/zap" ) // MiMoService 基于 Xiaomi MiMo TTS API 的语音合成实现。 // 接口兼容 OpenAI chat/completions 格式,通过 messages 传递待合成文本与风格指令。 type MiMoService struct { apiKey string model string voice string endpoint string timeout time.Duration logger *zap.SugaredLogger client *http.Client } // NewMiMoService 创建 MiMo TTS 服务。 // model、voice、endpoint 由 config 层保证非空。 func NewMiMoService(apiKey, model, voice, endpoint string, timeoutSec, httpClientTimeoutSec int, logger *zap.SugaredLogger) *MiMoService { timeout := time.Duration(timeoutSec) * time.Second if timeout <= 0 { timeout = 5 * time.Second } httpClientTimeout := time.Duration(httpClientTimeoutSec) * time.Second if httpClientTimeout <= 0 { httpClientTimeout = 30 * time.Second } return &MiMoService{ apiKey: apiKey, model: model, voice: voice, endpoint: endpoint, timeout: timeout, logger: logger, client: &http.Client{Timeout: httpClientTimeout}, } } // mimoTTSRequest MiMo TTS API 请求体。 type mimoTTSRequest struct { Model string `json:"model"` Messages []mimoTTSMessage `json:"messages"` Audio mimoTTSAudio `json:"audio"` Stream bool `json:"stream"` } // mimoTTSMessage MiMo TTS 消息。 type mimoTTSMessage struct { Role string `json:"role"` // "user"(风格指令)| "assistant"(待合成文本) Content string `json:"content"` } // mimoTTSAudio MiMo TTS 音频配置。 type mimoTTSAudio struct { Format string `json:"format"` // "mp3" | "wav" | "pcm16" Voice string `json:"voice"` // 预置音色 ID } // mimoTTSResponse MiMo TTS 非流式响应。 type mimoTTSResponse struct { Choices []struct { Message struct { Audio struct { Data string `json:"data"` // base64 编码的音频数据 } `json:"audio"` } `json:"message"` } `json:"choices"` } // mimoTTSStreamResponse MiMo TTS 流式响应。 type mimoTTSStreamResponse struct { Choices []struct { Delta struct { Audio struct { Data string `json:"data"` // base64 编码的音频数据片段 } `json:"audio"` } `json:"delta"` } `json:"choices"` } // SynthesizeStream 实现 tts.Service。从 textStream 读取句子,逐句调用 MiMo TTS API。 func (m *MiMoService) SynthesizeStream(ctx context.Context, textStream <-chan string, opts Options) (<-chan Chunk, error) { voice := opts.Voice if voice == "" { voice = m.voice } ch := make(chan Chunk, 4) go func() { defer close(ch) for text := range textStream { if text == "" { continue } audio, err := m.synthesize(ctx, text, voice) if err != nil { log := trace.FromContext(ctx) log.Warnw("mimo tts: synthesize failed", "error", err, "text_len", len(text), "text_preview", util.Truncate(text, 100)) // 静默跳过,不中断整个流 continue } select { case ch <- Chunk{Audio: audio, IsLast: true, Final: false}: case <-ctx.Done(): return } } // textStream 关闭,发送 Final 标记 select { case ch <- Chunk{Audio: nil, IsLast: false, Final: true}: case <-ctx.Done(): } }() return ch, nil } // synthesize 调用 MiMo TTS API 合成单个句子。 // 使用非流式调用,返回完整音频数据(base64 解码后)。 func (m *MiMoService) synthesize(ctx context.Context, text, voice string) ([]byte, error) { // 单句超时 ctx, cancel := context.WithTimeout(ctx, m.timeout) defer cancel() // 构建 MiMo TTS 请求:文本放在 assistant 消息中 reqBody := mimoTTSRequest{ Model: m.model, Messages: []mimoTTSMessage{ { Role: "assistant", Content: text, }, }, Audio: mimoTTSAudio{ Format: "mp3", Voice: voice, }, Stream: false, } payload, err := json.Marshal(reqBody) if err != nil { return nil, fmt.Errorf("mimo tts: marshal request: %w", err) } url := strings.TrimRight(m.endpoint, "/") + "/chat/completions" req, err := http.NewRequestWithContext(ctx, http.MethodPost, url, bytes.NewReader(payload)) if err != nil { return nil, fmt.Errorf("mimo tts: create request: %w", err) } req.Header.Set("Content-Type", "application/json") req.Header.Set("api-key", m.apiKey) resp, err := m.client.Do(req) if err != nil { return nil, fmt.Errorf("mimo tts: send request: %w", err) } defer resp.Body.Close() if resp.StatusCode != http.StatusOK { errBody, _ := io.ReadAll(resp.Body) return nil, fmt.Errorf("mimo tts: api error (status %d): %s", resp.StatusCode, string(errBody)) } // 非流式响应:解析 JSON,提取 base64 音频数据 respBody, err := io.ReadAll(resp.Body) if err != nil { return nil, fmt.Errorf("mimo tts: read response: %w", err) } var ttsResp mimoTTSResponse if err := json.Unmarshal(respBody, &ttsResp); err != nil { return nil, fmt.Errorf("mimo tts: unmarshal response: %w", err) } if len(ttsResp.Choices) == 0 { return nil, fmt.Errorf("mimo tts: empty choices in response") } audioData := ttsResp.Choices[0].Message.Audio.Data if audioData == "" { return nil, fmt.Errorf("mimo tts: empty audio data in response") } // base64 解码音频数据 audio, err := base64.StdEncoding.DecodeString(audioData) if err != nil { return nil, fmt.Errorf("mimo tts: decode audio base64: %w", err) } return audio, nil }