python-连接阿里语音合成-Cosyvoice-v2模型-方言
# coding=utf-8
"""
阿里云百炼 CosyVoice 东北方言语音合成测试脚本
功能:
1. 非流式合成(call) —— 一次性返回完整音频,保存为 MP3 并播放
2. 流式合成(streaming_call) —— 边合成边播放,PyAudio 实时输出
使用方式:
python test_tts_northeast.py --mode non-stream # 非流式测试
python test_tts_northeast.py --mode stream # 流式测试
python test_tts_northeast.py --mode both # 两种都测试(默认)
环境变量:
DASHSCOPE_API_KEY —— 阿里云百炼 API Key
获取地址: https://platform.qianwenai.com/home/api-keys
"""
import os
import sys
import time
import wave
import argparse
import threading
import dashscope
import pyaudio
from dashscope.audio.tts_v2 import SpeechSynthesizer, ResultCallback, AudioFormat
# ==================== 配置 ====================
os.environ["DASHSCOPE_API_KEY"] = ""
DASHSCOPE_API_KEY = os.environ.get("DASHSCOPE_API_KEY", "")
#MODEL = "cosyvoice-v3-flash"#v3模型
MODEL = "cosyvoice-v2" # v2 模型
# VOICE = "longanhuan_v3" # 东北话·女声
# VOICE = "longxiaoxia_v3" # 东北话·女声
#VOICE = "longxiang_v3" # v3 通用女声
# VOICE = "longwan_v3" # v3 通用男声
VOICE = "longlaotie_v2" # 东北话·口语男声(v2模型)
#VOICE = "longxiaochun_v2" # 通用普通话女声
# VOICE = "longshu_v2" # 通用普通话男声
TEST_TEXT = "哎呀妈呀,你可算来啦!长白山这嘎达可太美了,天池那水蓝得跟画似的,赶紧跟我走,咱上山顶看看去!"
OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "tts_output")
# ==================== PyAudio 播放器 ====================
class PyAudioPlayer:
"""PyAudio 实时播放器,支持流式写入 PCM 数据"""
def __init__(self, sample_rate=24000, channels=1, sample_width=2):
self.sample_rate = sample_rate
self.channels = channels
self.sample_width = sample_width
self._pyaudio = pyaudio.PyAudio()
self._stream = None
def open(self):
self._stream = self._pyaudio.open(
format=pyaudio.paInt16,
channels=self.channels,
rate=self.sample_rate,
output=True,
frames_per_buffer=4096,
)
def write(self, pcm_data: bytes):
if self._stream and self._stream.is_active():
self._stream.write(pcm_data)
def close(self):
if self._stream:
self._stream.stop_stream()
self._stream.close()
self._pyaudio.terminate()
def play_pcm_file(filepath: str, sample_rate=24000, channels=1, sample_width=2):
"""播放 PCM 文件"""
p = pyaudio.PyAudio()
stream = p.open(
format=pyaudio.paInt16,
channels=channels,
rate=sample_rate,
output=True,
)
with open(filepath, "rb") as f:
data = f.read(4096)
while data:
stream.write(data)
data = f.read(4096)
stream.stop_stream()
stream.close()
p.terminate()
def save_pcm_as_wav(pcm_data: bytes, filepath: str, sample_rate=24000, channels=1, sample_width=2):
"""将 PCM 数据保存为 WAV 文件"""
with wave.open(filepath, "wb") as wf:
wf.setnchannels(channels)
wf.setsampwidth(sample_width)
wf.setframerate(sample_rate)
wf.writeframes(pcm_data)
# ==================== 非流式合成 ====================
def test_non_stream():
"""
非流式合成测试
使用 SpeechSynthesizer.call() 一次性获取完整音频
"""
print("=" * 60)
print("【非流式合成测试】cosyvoice-v3-flash + 东北方言")
print("=" * 60)
if not DASHSCOPE_API_KEY:
print("[错误] 请设置环境变量 DASHSCOPE_API_KEY")
sys.exit(1)
dashscope.api_key = DASHSCOPE_API_KEY
dashscope.base_websocket_api_url = "wss://dashscope.aliyuncs.com/api-ws/v1/inference"
os.makedirs(OUTPUT_DIR, exist_ok=True)
synthesizer = SpeechSynthesizer(
model=MODEL,
voice=VOICE,
format=AudioFormat.PCM_24000HZ_MONO_16BIT,
)
print(f"[信息] 模型: {MODEL}")
print(f"[信息] 音色: {VOICE}")
print(f"[信息] 合成文本: {TEST_TEXT}")
print("[信息] 正在合成(非流式),请等待...")
start_time = time.time()
audio_data = synthesizer.call(TEST_TEXT)
elapsed = (time.time() - start_time) * 1000
if not audio_data:
print("[错误] 合成失败,未获取到音频数据")
return
wav_path = os.path.join(OUTPUT_DIR, "test_non_stream_northeast.wav")
save_pcm_as_wav(audio_data, wav_path)
print(f"[指标] requestId: {synthesizer.get_last_request_id()}")
print(f"[指标] 首包延迟: {synthesizer.get_first_package_delay()} 毫秒")
print(f"[指标] 总耗时: {elapsed:.0f} 毫秒")
print(f"[指标] 音频大小: {len(audio_data)} 字节")
print(f"[输出] 已保存: {wav_path}")
print("[信息] 正在播放...")
player = PyAudioPlayer(sample_rate=24000)
player.open()
player.write(audio_data)
player.close()
print("[完成] 非流式合成测试结束\n")
# ==================== 流式合成 ====================
class StreamingCallback(ResultCallback):
"""流式合成回调,实时播放 PCM 音频"""
def __init__(self):
super().__init__()
self.player = PyAudioPlayer(sample_rate=24000)
self.player.open()
self._audio_chunks = []
self._first_audio_received = False
self._start_time = None
self._complete_event = threading.Event()
self._error = None
def on_open(self):
print("[流式] WebSocket 连接已建立")
def on_data(self, data: bytes) -> None:
if not self._first_audio_received:
self._first_audio_received = True
if self._start_time:
delay = (time.time() - self._start_time) * 1000
print(f"[指标] 首包音频到达延迟: {delay:.0f} 毫秒")
self._audio_chunks.append(data)
self.player.write(data)
def on_complete(self):
print("[流式] 合成完成")
self._complete_event.set()
def on_error(self, message):
print(f"[流式] 错误: {message}")
self._error = message
self._complete_event.set()
def on_close(self):
self.player.close()
def on_event(self, message):
pass
def get_all_audio(self) -> bytes:
return b"".join(self._audio_chunks)
def wait_complete(self, timeout=30):
self._complete_event.wait(timeout=timeout)
return not self._error
def test_stream():
"""
流式合成测试
使用 SpeechSynthesizer.streaming_call() 边合成边播放
"""
print("=" * 60)
print("【流式合成测试】cosyvoice-v3-flash + 东北方言")
print("=" * 60)
if not DASHSCOPE_API_KEY:
print("[错误] 请设置环境变量 DASHSCOPE_API_KEY")
sys.exit(1)
dashscope.api_key = DASHSCOPE_API_KEY
dashscope.base_websocket_api_url = "wss://dashscope.aliyuncs.com/api-ws/v1/inference"
os.makedirs(OUTPUT_DIR, exist_ok=True)
callback = StreamingCallback()
synthesizer = SpeechSynthesizer(
model=MODEL,
voice=VOICE,
format=AudioFormat.PCM_24000HZ_MONO_16BIT,
callback=callback,
)
print(f"[信息] 模型: {MODEL}")
print(f"[信息] 音色: {VOICE}")
# 模拟 LLM 分段输出:将文本分成多段,逐段发送
text_chunks = [
"哎呀妈呀,你可算来啦!",
"长白山这嘎达可太美了,",
"天池那水蓝得跟画似的,",
"赶紧跟我走,咱上山顶看看去!",
]
print(f"[信息] 将分 {len(text_chunks)} 段发送文本:")
for i, chunk in enumerate(text_chunks):
print(f" 段{i+1}: {chunk}")
callback._start_time = time.time()
print("[信息] 正在流式合成,边合成边播放...")
for chunk in text_chunks:
synthesizer.streaming_call(chunk)
print(f" [发送] {chunk}")
time.sleep(0.1)
print("[信息] 所有文本已发送,等待合成完成...")
synthesizer.streaming_complete()
success = callback.wait_complete(timeout=30)
if success:
all_audio = callback.get_all_audio()
wav_path = os.path.join(OUTPUT_DIR, "test_stream_northeast.wav")
save_pcm_as_wav(all_audio, wav_path)
print(f"[指标] requestId: {synthesizer.get_last_request_id()}")
print(f"[指标] 首包延迟: {synthesizer.get_first_package_delay()} 毫秒")
print(f"[指标] 音频总大小: {len(all_audio)} 字节")
print(f"[输出] 已保存: {wav_path}")
else:
print(f"[错误] 流式合成失败: {callback._error}")
print("[完成] 流式合成测试结束\n")
# ==================== 主入口 ====================
def main():
parser = argparse.ArgumentParser(description="阿里云 CosyVoice 东北方言 TTS 测试")
parser.add_argument(
"--mode",
choices=["non-stream", "stream", "both"],
default="both",
help="测试模式: non-stream(非流式), stream(流式), both(两种都测)",
)
args = parser.parse_args()
if not DASHSCOPE_API_KEY:
print("[错误] 未检测到 DASHSCOPE_API_KEY 环境变量")
print(" 请执行: set DASHSCOPE_API_KEY=sk-xxx")
print(" 获取地址: https://platform.qianwenai.com/home/api-keys")
sys.exit(1)
print(f"\n阿里云 CosyVoice 东北方言 TTS 测试")
print(f"API Key: {DASHSCOPE_API_KEY[:8]}...")
print(f"输出目录: {OUTPUT_DIR}\n")
if args.mode in ("non-stream", "both"):
test_non_stream()
if args.mode in ("stream", "both"):
test_stream()
print("全部测试完成!")
if __name__ == "__main__":
main()
DASHSCOPE_API_KEY-设置为自己的APIkey
DASHSCOPE_API_KEY-设置为自己的APIkey
浙公网安备 33010602011771号