diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..e7ef1d8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,14 @@ +# Go build cache +.gocache/ +*.exe +*.test +*.out + +# Editor / OS +.vscode/ +.idea/ +.DS_Store +Thumbs.db + +# Logs +*.log \ No newline at end of file diff --git a/README.md b/README.md index 3fad183..b562195 100644 --- a/README.md +++ b/README.md @@ -65,6 +65,7 @@ tts_server.exe | `BYTEDANCE_TTS_API_KEY` | 火山引擎新版控制台 API Key | `your_api_key_here` | | `BYTEDANCE_TTS_RESOURCE_ID` | 资源ID,决定模型版本 | `seed-tts-1.0` | | `BYTEDANCE_TTS_SPEAKER` | 发音人(音色)ID | `zh_female_qingxin` | +| `BYTEDANCE_TTS_MODEL` | 模型子版本(复刻音色必填,不设默认 `seed-tts-2.0-standard`) | `seed-tts-2.0-standard` | ### 可选参数 @@ -90,6 +91,18 @@ tts_server.exe **注意:** 1.0音色只能搭配 `seed-tts-1.0` Resource ID,2.0音色只能搭配 `seed-tts-2.0` Resource ID。 +### v3 API 调用说明 + +本项目按火山 v3 单向流式 TTS API 实现([官方文档](https://www.volcengine.com/docs/6561/2528925)),相比 v1/v2 有以下关键差异: + +- **不再使用业务集群**(`cluster` 字段在 v3 已废弃),改用 `X-Api-Resource-Id` HTTP header 路由模型 +- **鉴权 header 只有** `X-Api-Key` 一个,无 `Authorization`,无 app 对象 +- **`req_params.model` 字段**:v3 必须显式传子模型版本。可选值: + - `seed-tts-2.0-standard`(默认,标准版,常规音色/复刻音色通用) + - `seed-tts-2.0-expressive`(表现力增强版,部分复刻音色推荐) + - 留空时会用 `seed-tts-2.0-standard` 作为兜底 +- **复刻音色(`S_` 开头的 speaker)必须显式传 model**,否则可能因默认模型与复刻音色不匹配返回 `55000000` + ## CORS 跨域配置 跨域请求由 `ALLOWED_ORIGINS` 环境变量控制,按**完整 origin**(含协议 + 域名 + 端口)精确匹配: @@ -232,6 +245,7 @@ OPENAI_TTS_API_KEY=sk-key1,sk-key2,sk-key3 2. 用控制台的在线体验/调试试一下同一对 `BYTEDANCE_TTS_RESOURCE_ID` + 音色 3. 控制台能合成的组合才是正确的 4. 把控制台显示的**实际资源 ID 字符串**(通常是 `volc.megatts.*` 格式)填到 Zeabur 的 `BYTEDANCE_TTS_RESOURCE_ID` +5. 如果你用的是**声音复刻**音色(speaker 以 `S_` 开头),同时确认设置了 `BYTEDANCE_TTS_MODEL`(推荐 `seed-tts-2.0-standard` 或 `seed-tts-2.0-expressive`)。复刻音色不传 `model` 字段是 55000000 的常见原因之一 ### 5. PowerShell 下 `curl` 命令被解释错 @@ -334,3 +348,4 @@ sudo systemctl start tts-server 5. ALLOWED_ORIGINS 是否包含前端完整 origin(含 https://) 6. 客户端请求 URL 是否以 https:// 开头 7. 生产环境凭据是否定期轮换(API Key 明文出现在日志/对话中时立刻重置) +8. 复刻音色(speaker 以 `S_` 开头)是否设置了 `BYTEDANCE_TTS_MODEL`(默认 `seed-tts-2.0-standard`) diff --git a/adapter/volcano/volcano.go b/adapter/volcano/volcano.go index ce2544a..7681755 100644 --- a/adapter/volcano/volcano.go +++ b/adapter/volcano/volcano.go @@ -49,13 +49,16 @@ func (h *HTTPClient) PostStream(url string, headers map[string]string, body []by return h.client.Do(req) } +// convertSpeedToSpeechRate 把 OpenAI 风格的 speed(倍率)转成火山 v3 的 speech_rate(百分比)。 +// 文档规定 speech_rate 范围 [-50, 100],对应 0.5x ~ 2.0x 倍速。 +// 输入超出范围会被截断到边界值。 func convertSpeedToSpeechRate(speed float64) int { rate := int((speed - 1.0) * 100) - if rate < -200 { - rate = -200 + if rate < -50 { + rate = -50 } - if rate > 500 { - rate = 500 + if rate > 100 { + rate = 100 } return rate } @@ -69,14 +72,23 @@ func Synthesis(config *dto.ByteDanceTTSConfig, httpClient *HTTPClient, text stri speaker = voice } + model := config.Model + if model == "" { + model = "seed-tts-2.0-standard" // 文档默认值 复刻音色可设为 seed-tts-2.0-expressive + } + + // 请求体结构严格按火山 v3 单向流式 API 文档构造 + // https://www.volcengine.com/docs/6561/2528925 + // v3 鉴权只依赖 X-Api-Key 一个 header,不再需要业务集群参数 params := map[string]interface{}{ "user": map[string]interface{}{ - "uid": "uid", + "uid": reqID, // 文档要求随机字符串,这里复用请求级 UUID }, "namespace": "UnidirectionalTTS", "req_params": map[string]interface{}{ "text": text, "speaker": speaker, + "model": model, // 复刻音色必填 "audio_params": map[string]interface{}{ "format": "wav", "sample_rate": 24000, @@ -88,9 +100,9 @@ func Synthesis(config *dto.ByteDanceTTSConfig, httpClient *HTTPClient, text stri headers := map[string]string{ "Content-Type": "application/json", "Connection": "keep-alive", - "X-Api-Resource-Id": config.ResourceId, + "X-Api-Resource-Id": config.ResourceId, // 模型路由(seed-tts-2.0 / seed-icl-2.0) "X-Api-Request-Id": reqID, - "X-Api-Key": config.ApiKey, + "X-Api-Key": config.ApiKey, // v3 鉴权 key } bodyStr, err := json.Marshal(params) diff --git a/dto/tts.go b/dto/tts.go index 2e1a17f..d874b47 100644 --- a/dto/tts.go +++ b/dto/tts.go @@ -30,6 +30,7 @@ type ByteDanceTTSConfig struct { ApiKey string ResourceId string Speaker string + Model string // v3 声音复刻/语音大模型 子模型版本,复刻音色必填 URL string Timeout time.Duration } diff --git a/setting/config.go b/setting/config.go index 5389f7b..8a7174a 100644 --- a/setting/config.go +++ b/setting/config.go @@ -19,6 +19,10 @@ func InitTTSConfig() error { apiKey := os.Getenv("BYTEDANCE_TTS_API_KEY") resourceId := os.Getenv("BYTEDANCE_TTS_RESOURCE_ID") speaker := os.Getenv("BYTEDANCE_TTS_SPEAKER") + model := os.Getenv("BYTEDANCE_TTS_MODEL") + if model == "" { + model = "seed-tts-2.0-standard" // 文档默认值 复刻音色可设为 seed-tts-2.0-expressive + } missingVars := []string{} if apiKey == "" { @@ -50,6 +54,7 @@ func InitTTSConfig() error { ApiKey: apiKey, ResourceId: resourceId, Speaker: speaker, + Model: model, URL: url, Timeout: timeout, } @@ -72,6 +77,7 @@ func CheckEnvironmentVariables() map[string]interface{} { optionalVars := map[string]bool{ "BYTEDANCE_TTS_TIMEOUT": os.Getenv("BYTEDANCE_TTS_TIMEOUT") != "", + "BYTEDANCE_TTS_MODEL": os.Getenv("BYTEDANCE_TTS_MODEL") != "", "OPENAI_TTS_API_KEY": os.Getenv("OPENAI_TTS_API_KEY") != "", "PORT": os.Getenv("PORT") != "", }