Files
aiagent/backend/app/api/voice.py
renjianbo 3483c6b3be fix: update TTS voice mapping to only use valid Edge TTS voices
Previously the voice map referenced 10+ non-existent Edge TTS voice names
(xiaohan, xiaochen, xiaoshuang, etc.), causing NoAudioReceived errors and
fallback to browser default TTS. Now only maps to the 6 actually available
Chinese voices (xiaoxiao, xiaoyi, yunxi, yunyang, yunjian, yunxia).
Added voice speed adjustments and fixed --rate argument format for Windows.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-07-04 17:06:51 +08:00

326 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
语音 API — Android 应用语音交互接口
提供语音转文字 (ASR) 和文字转语音 (TTS) 的 HTTP API。
TTS 优先使用 OpenAI TTS需配置 OPENAI_API_KEY否则使用免费的 Edge TTS。
"""
from __future__ import annotations
import asyncio
import logging
import os
import subprocess
import sys
import tempfile
import time
from pathlib import Path
from fastapi import APIRouter, Depends, File, HTTPException, Query, UploadFile
from fastapi.responses import FileResponse
from pydantic import BaseModel, Field
from app.api.auth import get_current_user
from app.models.user import User
from app.core.config import settings
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/api/v1/voice", tags=["voice"])
# Edge TTS 中文语音映射 (voice name -> Edge TTS Chinese voice)
# 注意edge-tts 免费版仅支持以下中文语音:
# 女声xiaoxiao, xiaoyi
# 男声yunxi, yunyang, yunjian, yunxia
# 方言liaoning-Xiaobei, shaanxi-Xiaoni
_EDGE_VOICE_MAP = {
# ── 女声 ──
"xiaoxiao": "zh-CN-XiaoxiaoNeural", # 晓晓 · 温柔
"xiaoyi": "zh-CN-XiaoyiNeural", # 晓怡 · 活泼
# ── 男声 ──
"yunxi": "zh-CN-YunxiNeural", # 云希 · 阳光
"yunyang": "zh-CN-YunyangNeural", # 云扬 · 专业
"yunjian": "zh-CN-YunjianNeural", # 云健 · 激情
"yunxia": "zh-CN-YunxiaNeural", # 云夏 · 可爱
# ── 女声变体(使用语速微调区分)──
"xiaochen": "zh-CN-XiaoxiaoNeural", # 晓辰 · 温和xiaoxiao 慢速)
"xiaohan": "zh-CN-XiaoyiNeural", # 晓涵 · 知性xiaoyi 慢速)
# ── 教师女声(映射到可用女声)──
"teacher1": "zh-CN-XiaoxiaoNeural", # 晓晓老师 · 温柔教导
"teacher2": "zh-CN-XiaoyiNeural", # 晓怡老师 · 活泼教学
# ── 兼容旧 OpenAI 风格名称 ──
"alloy": "zh-CN-YunxiNeural",
"echo": "zh-CN-YunyangNeural",
"fable": "zh-CN-XiaoxiaoNeural",
"onyx": "zh-CN-YunjianNeural",
"nova": "zh-CN-XiaoyiNeural",
"shimmer": "zh-CN-XiaoxiaoNeural",
}
# 语音语速微调 (voice -> adjusted speed multiplier)
# 通过微调语速来区分映射到同一 Edge TTS 语音的不同选项
_VOICE_SPEED_ADJUST = {
"xiaochen": 0.90, # 晓辰 · 稍慢更温和
"xiaohan": 0.90, # 晓涵 · 稍慢更知性
"teacher1": 0.95, # 晓晓老师 · 稍慢有教导感
"teacher2": 0.95, # 晓怡老师 · 稍慢有教学感
}
# 新增语音 → OpenAI TTS 降级映射
_OPENAI_FALLBACK_VOICE = {
"xiaoxiao": "nova", "xiaoyi": "nova",
"xiaochen": "nova", "xiaohan": "nova",
"yunxia": "alloy",
"teacher1": "nova", "teacher2": "nova",
}
class AsrResponse(BaseModel):
"""语音识别响应"""
text: str = Field(..., description="识别出的文字")
language: str = Field(default="zh", description="语言代码")
class TtsRequest(BaseModel):
"""文字转语音请求"""
text: str = Field(..., min_length=1, max_length=4000, description="要合成的文字")
voice: str = Field(
default="xiaoxiao",
description="语音风格xiaoxiao/xiaoyi/xiaochen/xiaohan/xiaomeng/xiaomo/xiaorui/xiaoshuang/xiaoxuan/xiaoyan/xiaozhen(女) yunxi/yunyang/yunjian/yunfeng/yunze/yunhao(男)",
)
speed: float = Field(
default=1.0,
ge=0.5,
le=2.0,
description="语速0.5(慢)到 2.0(快),默认 1.0",
)
class TtsResponse(BaseModel):
"""文字转语音响应"""
audio_url: str = Field(..., description="音频文件下载 URL")
text_length: int = Field(..., description="文字长度")
voice: str = Field(..., description="使用的语音风格")
# TTS 输出缓存目录
_TTS_DIR = Path(settings.LOCAL_FILE_TOOLS_ROOT) / "tts_outputs"
@router.post("/asr", response_model=AsrResponse)
async def voice_to_text(
file: UploadFile = File(..., description="音频文件AAC/WAV/MP3/WebM/M4A"),
language: str = Query("zh", description="语言代码"),
current_user: User = Depends(get_current_user),
):
"""
语音转文字 (ASR)。
接收 Android 端录音上传的 AAC 音频文件,调用 Whisper API 返回识别文字。
采样率建议 16000 Hz。
"""
# 验证文件
if not file.filename:
raise HTTPException(status_code=400, detail="文件名为空")
ext = (file.filename.rsplit(".", 1)[-1].lower() if "." in file.filename else "aac")
allowed = {"aac", "wav", "mp3", "webm", "m4a", "ogg", "flac", "mpeg"}
if ext not in allowed:
raise HTTPException(status_code=400, detail=f"不支持的音频格式: {ext}")
# 检查文件大小
content = await file.read()
max_bytes = 25 * 1024 * 1024 # 25 MB
if len(content) > max_bytes:
raise HTTPException(status_code=400, detail=f"文件过大 ({len(content) / 1024 / 1024:.1f} MB)")
# 写入临时文件
tmp = tempfile.NamedTemporaryFile(suffix=f".{ext}", delete=False)
try:
tmp.write(content)
tmp.close()
api_key = (getattr(settings, "OPENAI_API_KEY", "") or "").strip()
base_url = (
getattr(settings, "OPENAI_BASE_URL", "https://api.openai.com/v1")
or "https://api.openai.com/v1"
).strip()
if not api_key:
raise HTTPException(status_code=503, detail="ASR 服务未配置 (OPENAI_API_KEY)")
import httpx
async with httpx.AsyncClient(timeout=120) as client:
with open(tmp.name, "rb") as f:
files_payload = {
"file": (Path(tmp.name).name, f, f"audio/{ext}"),
"model": (None, "whisper-1"),
"language": (None, language),
}
resp = await client.post(
f"{base_url.rstrip('/')}/audio/transcriptions",
headers={"Authorization": f"Bearer {api_key}"},
files=files_payload,
)
if resp.status_code != 200:
logger.error("Whisper API 错误 %d: %s", resp.status_code, resp.text[:500])
raise HTTPException(
status_code=502,
detail=f"语音识别服务返回错误 (HTTP {resp.status_code})",
)
data = resp.json()
text = data.get("text", "")
logger.info("ASR 完成: user=%s file=%s len=%d", current_user.id, file.filename, len(text))
return AsrResponse(text=text, language=language)
finally:
Path(tmp.name).unlink(missing_ok=True)
@router.post("/tts", response_model=TtsResponse)
async def text_to_voice(
req: TtsRequest,
current_user: User = Depends(get_current_user),
):
"""
文字转语音 (TTS)。
优先使用 OpenAI TTS需配置有效 OPENAI_API_KEY否则使用免费 Edge TTS。
Android 端使用 ExoPlayer 播放返回的 audio_url。
"""
valid_voices = set(_EDGE_VOICE_MAP.keys())
voice = req.voice if req.voice in valid_voices else "xiaoxiao"
text = req.text.strip()
if len(text) > 4000:
text = text[:4000] + "..."
_TTS_DIR.mkdir(parents=True, exist_ok=True)
filename = f"tts_{current_user.id}_{int(time.time())}.mp3"
filepath = _TTS_DIR / filename
# 语速微调:某些语音选项通过微调语速来区分
speed_adj = _VOICE_SPEED_ADJUST.get(voice, 1.0)
adjusted_speed = req.speed * speed_adj
# 尝试 OpenAI TTS
api_key = (getattr(settings, "OPENAI_API_KEY", "") or "").strip()
base_url = (
getattr(settings, "OPENAI_BASE_URL", "https://api.openai.com/v1")
or "https://api.openai.com/v1"
).strip()
use_edge = False
if api_key and api_key not in ("your-openai-api-key", "sk-your-"):
import httpx
# OpenAI TTS 仅支持 alloy/echo/fable/onyx/nova/shimmer其他语音用映射
openai_voices = {"alloy", "echo", "fable", "onyx", "nova", "shimmer"}
openai_voice = voice if voice in openai_voices else _OPENAI_FALLBACK_VOICE.get(voice, "nova")
try:
async with httpx.AsyncClient(timeout=60) as client:
resp = await client.post(
f"{base_url.rstrip('/')}/audio/speech",
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
},
json={
"model": "tts-1",
"voice": openai_voice,
"input": text,
"speed": adjusted_speed,
},
)
if resp.status_code == 200:
filepath.write_bytes(resp.content)
logger.info("OpenAI TTS 完成: user=%s text_len=%d voice=%s speed=%.1f", current_user.id, len(req.text), voice, adjusted_speed)
return TtsResponse(
audio_url=f"/api/v1/voice/audio/{filename}",
text_length=len(req.text),
voice=voice,
)
else:
logger.warning("OpenAI TTS 失败 (%d), 回退到 Edge TTS", resp.status_code)
use_edge = True
except Exception as exc:
logger.warning("OpenAI TTS 异常: %s, 回退到 Edge TTS", exc)
use_edge = True
else:
use_edge = True
# 回退Edge TTS免费无需 API KEY
if use_edge:
edge_voice = _EDGE_VOICE_MAP.get(voice, "zh-CN-XiaoxiaoNeural")
try:
await _edge_tts_synthesize(text, edge_voice, str(filepath), speed=adjusted_speed)
logger.info("Edge TTS 完成: user=%s text_len=%d edge_voice=%s speed=%.1f", current_user.id, len(req.text), edge_voice, adjusted_speed)
return TtsResponse(
audio_url=f"/api/v1/voice/audio/{filename}",
text_length=len(req.text),
voice=voice,
)
except Exception as exc:
logger.error("Edge TTS 失败: %s", exc)
raise HTTPException(status_code=502, detail=f"TTS 服务不可用: {exc}")
async def _edge_tts_synthesize(
text: str, voice: str, output_path: str, speed: float = 1.0,
) -> None:
"""使用 edge-tts 命令行工具合成语音。"""
import shutil
exe = shutil.which("edge-tts") or shutil.which("edge-tts", path=(
os.environ.get("PATH", "") + os.pathsep +
os.path.join(os.path.dirname(sys.executable), "Scripts") + os.pathsep +
os.path.join(os.path.dirname(sys.executable), "..", "Scripts")
))
if not exe:
exe = "edge-tts"
rate_str = f"{int((speed - 1.0) * 100):+d}%"
logger.debug("edge-tts exe: %s text_len=%d rate=%s", exe, len(text), rate_str)
proc = await asyncio.create_subprocess_exec(
exe,
"--text", text,
"--voice", voice,
f"--rate={rate_str}",
"--write-media", output_path,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
)
stdout, stderr = await proc.communicate()
out_str = stdout.decode(errors="replace") if stdout else ""
err_str = stderr.decode(errors="replace") if stderr else ""
if proc.returncode != 0:
raise RuntimeError(
f"edge-tts CLI 失败 (exit={proc.returncode}): {err_str or out_str}"
)
if not Path(output_path).is_file():
raise RuntimeError(f"edge-tts 未生成输出文件: {out_str} {err_str}")
logger.info("edge-tts CLI 成功: %d bytes", Path(output_path).stat().st_size)
@router.get("/audio/{filename}")
async def get_tts_audio(filename: str):
"""获取 TTS 生成的音频文件无需认证URL 包含随机名)。"""
if ".." in filename or "/" in filename or "\\" in filename:
raise HTTPException(status_code=400, detail="非法文件名")
filepath = _TTS_DIR / filename
if not filepath.is_file():
raise HTTPException(status_code=404, detail="音频文件不存在或已过期")
return FileResponse(filepath, media_type="audio/mpeg")