feat: 添加并发限流保护并更新文档

- 新增全局并发请求限制,最多同时处理10个TTS请求
- 更新健康检查接口,新增配置检查和统计信息返回
- 调整响应格式文档,仅保留wav格式支持说明
- 补充限流机制说明和触发后的响应示例
- 优化日志输出,限流时打印中文警告日志
This commit is contained in:
sun
2026-05-17 23:42:10 +08:00
parent 08c529c375
commit 00460771c4
2 changed files with 68 additions and 17 deletions
+35 -6
View File
@@ -12,6 +12,7 @@
- ✅ 支持多种发音人和模型版本 - ✅ 支持多种发音人和模型版本
- ✅ 内置速率限制和统计功能 - ✅ 内置速率限制和统计功能
- ✅ 支持配置API密钥验证 - ✅ 支持配置API密钥验证
- ✅ 并发限制:最多同时处理10个请求(保护上游API)
- ✅ 跨平台支持(Windows/Linux/macOS) - ✅ 跨平台支持(Windows/Linux/macOS)
## 文件说明 ## 文件说明
@@ -111,7 +112,7 @@ tts_server.exe
- `model` - 模型名称(OpenAI兼容,实际不影响) - `model` - 模型名称(OpenAI兼容,实际不影响)
- `input` - 要合成的文本 - `input` - 要合成的文本
- `voice` - 发音人(OpenAI兼容,实际不影响) - `voice` - 发音人(OpenAI兼容,实际不影响)
- `response_format` - 输出格式:`wav`、`mp3`、`opus`、`aac`、`flac` - `response_format` - 输出格式:仅支持 `wav`
- `speed` - 语速:0.25 ~ 4.0 - `speed` - 语速:0.25 ~ 4.0
**示例调用:** **示例调用:**
@@ -123,18 +124,46 @@ curl -X POST "http://localhost:8080/v1/audio/speech" \
-o output.wav -o output.wav
``` ```
### 健康检查 ### 健康检查(含统计信息)
```bash ```bash
curl http://localhost:8080/health curl http://localhost:8080/health
``` ```
### 统计信息 返回包含:服务状态、请求统计、错误记录、配置检查结果
```bash ## 限流机制
curl http://localhost:8080/stats
为保护上游火山引擎API,服务实现了两层限流保护:
### 1. 全局并发限制
- **限制**:最多同时处理 **10个** TTS请求
- **触发**:超过10个并发请求时
- **错误码**:`503 Service Unavailable`
- **说明**:确保不超过上游API的并发限制
### 2. IP速率限制
- **限制**:每个IP每分钟 **100个** 请求
- **触发**:单个IP调用过于频繁
- **错误码**:`429 Too Many Requests`
- **说明**:防止单个客户端滥用服务
### 触发限流时的响应
```json
{
"error": {
"message": "Server is busy, maximum concurrent requests reached.",
"type": "concurrency_limit_error",
"code": "max_concurrent_requests"
}
}
``` ```
### 服务器日志
触发限流时服务器会输出中文警告日志:
- `警告: 已达到最大并发请求数限制,拒绝请求 - 客户端IP: x.x.x.x`
- `警告: 已超过IP速率限制,拒绝请求 - 客户端IP: x.x.x.x`
## 支持的发音人 ## 支持的发音人
具体发音人列表请参考火山引擎官方文档: 具体发音人列表请参考火山引擎官方文档:
@@ -173,7 +202,7 @@ OPENAI_TTS_API_KEY=sk-key1,sk-key2,sk-key3
服务启动后会输出详细日志,包括: 服务启动后会输出详细日志,包括:
- 服务启动信息 - 服务启动信息
- 鉴权模式和配置状态 - 配置状态
- 请求统计信息 - 请求统计信息
- 错误详情 - 错误详情
+33 -11
View File
@@ -24,17 +24,18 @@ import (
) )
const ( const (
DEFAULT_PORT = "8080" DEFAULT_PORT = "8080"
DEFAULT_TIMEOUT = 30 * time.Second DEFAULT_TIMEOUT = 30 * time.Second
MAX_TEXT_LENGTH = 5000 MAX_TEXT_LENGTH = 5000
MIN_SPEED = 0.25 MIN_SPEED = 0.25
MAX_SPEED = 4.0 MAX_SPEED = 4.0
DEFAULT_SPEED = 1.0 DEFAULT_SPEED = 1.0
MAX_REQUEST_BODY_SIZE = 1024 * 1024 MAX_REQUEST_BODY_SIZE = 1024 * 1024
RATE_LIMIT_REQUESTS = 100 RATE_LIMIT_REQUESTS = 100
RATE_LIMIT_WINDOW = time.Minute RATE_LIMIT_WINDOW = time.Minute
MAX_RESPONSE_TIMES = 100 MAX_RESPONSE_TIMES = 100
MAX_ERRORS = 10 MAX_ERRORS = 10
MAX_CONCURRENT_REQUESTS = 10
) )
type V3TTSResponse struct { type V3TTSResponse struct {
@@ -95,6 +96,7 @@ var (
globalHTTPClient *http.Client globalHTTPClient *http.Client
apiStats *Stats apiStats *Stats
rateLimiter *RateLimiter rateLimiter *RateLimiter
concurrencySem chan struct{}
) )
func init() { func init() {
@@ -118,6 +120,8 @@ func init() {
limit: RATE_LIMIT_REQUESTS, limit: RATE_LIMIT_REQUESTS,
window: RATE_LIMIT_WINDOW, window: RATE_LIMIT_WINDOW,
} }
concurrencySem = make(chan struct{}, MAX_CONCURRENT_REQUESTS)
} }
func (rl *RateLimiter) Allow(key string) bool { func (rl *RateLimiter) Allow(key string) bool {
@@ -442,6 +446,7 @@ func openaiTTSHandler(w http.ResponseWriter, r *http.Request) {
clientIP := getClientIP(r) clientIP := getClientIP(r)
if !rateLimiter.Allow(clientIP) { if !rateLimiter.Allow(clientIP) {
log.Printf("警告: 已超过IP速率限制,拒绝请求 - 客户端IP: %s", clientIP)
w.Header().Set("Content-Type", "application/json") w.Header().Set("Content-Type", "application/json")
w.WriteHeader(http.StatusTooManyRequests) w.WriteHeader(http.StatusTooManyRequests)
json.NewEncoder(w).Encode(map[string]interface{}{ json.NewEncoder(w).Encode(map[string]interface{}{
@@ -454,6 +459,23 @@ func openaiTTSHandler(w http.ResponseWriter, r *http.Request) {
return return
} }
select {
case concurrencySem <- struct{}{}:
defer func() { <-concurrencySem }()
default:
log.Printf("警告: 已达到最大并发请求数限制,拒绝请求 - 客户端IP: %s", getClientIP(r))
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(http.StatusServiceUnavailable)
json.NewEncoder(w).Encode(map[string]interface{}{
"error": map[string]interface{}{
"message": "Server is busy, maximum concurrent requests reached. Please try again later.",
"type": "concurrency_limit_error",
"code": "max_concurrent_requests",
},
})
return
}
body, err := io.ReadAll(http.MaxBytesReader(w, r.Body, MAX_REQUEST_BODY_SIZE)) body, err := io.ReadAll(http.MaxBytesReader(w, r.Body, MAX_REQUEST_BODY_SIZE))
if err != nil { if err != nil {
http.Error(w, "Request body too large", http.StatusRequestEntityTooLarge) http.Error(w, "Request body too large", http.StatusRequestEntityTooLarge)