python-连接阿里语音合成-Cosyvoice-v2模型-方言

# coding=utf-8
"""
阿里云百炼 CosyVoice 东北方言语音合成测试脚本

功能:
  1. 非流式合成(call)  —— 一次性返回完整音频,保存为 MP3 并播放
  2. 流式合成(streaming_call) —— 边合成边播放,PyAudio 实时输出

使用方式:
  python test_tts_northeast.py --mode non-stream    # 非流式测试
  python test_tts_northeast.py --mode stream        # 流式测试
  python test_tts_northeast.py --mode both          # 两种都测试(默认)

环境变量:
  DASHSCOPE_API_KEY —— 阿里云百炼 API Key
  获取地址: https://platform.qianwenai.com/home/api-keys
"""

import os
import sys
import time
import wave
import argparse
import threading

import dashscope
import pyaudio
from dashscope.audio.tts_v2 import SpeechSynthesizer, ResultCallback, AudioFormat

# ==================== 配置 ====================
os.environ["DASHSCOPE_API_KEY"] = ""

DASHSCOPE_API_KEY = os.environ.get("DASHSCOPE_API_KEY", "")

#MODEL = "cosyvoice-v3-flash"#v3模型
MODEL = "cosyvoice-v2"      # v2 模型


# VOICE = "longanhuan_v3"    # 东北话·女声
# VOICE = "longxiaoxia_v3"  # 东北话·女声
#VOICE = "longxiang_v3"      # v3 通用女声
# VOICE = "longwan_v3"      # v3 通用男声

VOICE = "longlaotie_v2"   # 东北话·口语男声(v2模型)
#VOICE = "longxiaochun_v2"   # 通用普通话女声
# VOICE = "longshu_v2"      # 通用普通话男声
TEST_TEXT = "哎呀妈呀,你可算来啦!长白山这嘎达可太美了,天池那水蓝得跟画似的,赶紧跟我走,咱上山顶看看去!"

OUTPUT_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "tts_output")


# ==================== PyAudio 播放器 ====================

class PyAudioPlayer:
    """PyAudio 实时播放器,支持流式写入 PCM 数据"""

    def __init__(self, sample_rate=24000, channels=1, sample_width=2):
        self.sample_rate = sample_rate
        self.channels = channels
        self.sample_width = sample_width
        self._pyaudio = pyaudio.PyAudio()
        self._stream = None

    def open(self):
        self._stream = self._pyaudio.open(
            format=pyaudio.paInt16,
            channels=self.channels,
            rate=self.sample_rate,
            output=True,
            frames_per_buffer=4096,
        )

    def write(self, pcm_data: bytes):
        if self._stream and self._stream.is_active():
            self._stream.write(pcm_data)

    def close(self):
        if self._stream:
            self._stream.stop_stream()
            self._stream.close()
        self._pyaudio.terminate()


def play_pcm_file(filepath: str, sample_rate=24000, channels=1, sample_width=2):
    """播放 PCM 文件"""
    p = pyaudio.PyAudio()
    stream = p.open(
        format=pyaudio.paInt16,
        channels=channels,
        rate=sample_rate,
        output=True,
    )
    with open(filepath, "rb") as f:
        data = f.read(4096)
        while data:
            stream.write(data)
            data = f.read(4096)
    stream.stop_stream()
    stream.close()
    p.terminate()


def save_pcm_as_wav(pcm_data: bytes, filepath: str, sample_rate=24000, channels=1, sample_width=2):
    """将 PCM 数据保存为 WAV 文件"""
    with wave.open(filepath, "wb") as wf:
        wf.setnchannels(channels)
        wf.setsampwidth(sample_width)
        wf.setframerate(sample_rate)
        wf.writeframes(pcm_data)


# ==================== 非流式合成 ====================

def test_non_stream():
    """
    非流式合成测试
    使用 SpeechSynthesizer.call() 一次性获取完整音频
    """
    print("=" * 60)
    print("【非流式合成测试】cosyvoice-v3-flash + 东北方言")
    print("=" * 60)

    if not DASHSCOPE_API_KEY:
        print("[错误] 请设置环境变量 DASHSCOPE_API_KEY")
        sys.exit(1)

    dashscope.api_key = DASHSCOPE_API_KEY
    dashscope.base_websocket_api_url = "wss://dashscope.aliyuncs.com/api-ws/v1/inference"

    os.makedirs(OUTPUT_DIR, exist_ok=True)

    synthesizer = SpeechSynthesizer(
        model=MODEL,
        voice=VOICE,
        format=AudioFormat.PCM_24000HZ_MONO_16BIT,
    )

    print(f"[信息] 模型: {MODEL}")
    print(f"[信息] 音色: {VOICE}")
    print(f"[信息] 合成文本: {TEST_TEXT}")
    print("[信息] 正在合成(非流式),请等待...")

    start_time = time.time()
    audio_data = synthesizer.call(TEST_TEXT)
    elapsed = (time.time() - start_time) * 1000

    if not audio_data:
        print("[错误] 合成失败,未获取到音频数据")
        return

    wav_path = os.path.join(OUTPUT_DIR, "test_non_stream_northeast.wav")
    save_pcm_as_wav(audio_data, wav_path)

    print(f"[指标] requestId: {synthesizer.get_last_request_id()}")
    print(f"[指标] 首包延迟: {synthesizer.get_first_package_delay()} 毫秒")
    print(f"[指标] 总耗时: {elapsed:.0f} 毫秒")
    print(f"[指标] 音频大小: {len(audio_data)} 字节")
    print(f"[输出] 已保存: {wav_path}")
    print("[信息] 正在播放...")

    player = PyAudioPlayer(sample_rate=24000)
    player.open()
    player.write(audio_data)
    player.close()

    print("[完成] 非流式合成测试结束\n")


# ==================== 流式合成 ====================

class StreamingCallback(ResultCallback):
    """流式合成回调,实时播放 PCM 音频"""

    def __init__(self):
        super().__init__()
        self.player = PyAudioPlayer(sample_rate=24000)
        self.player.open()
        self._audio_chunks = []
        self._first_audio_received = False
        self._start_time = None
        self._complete_event = threading.Event()
        self._error = None

    def on_open(self):
        print("[流式] WebSocket 连接已建立")

    def on_data(self, data: bytes) -> None:
        if not self._first_audio_received:
            self._first_audio_received = True
            if self._start_time:
                delay = (time.time() - self._start_time) * 1000
                print(f"[指标] 首包音频到达延迟: {delay:.0f} 毫秒")

        self._audio_chunks.append(data)
        self.player.write(data)

    def on_complete(self):
        print("[流式] 合成完成")
        self._complete_event.set()

    def on_error(self, message):
        print(f"[流式] 错误: {message}")
        self._error = message
        self._complete_event.set()

    def on_close(self):
        self.player.close()

    def on_event(self, message):
        pass

    def get_all_audio(self) -> bytes:
        return b"".join(self._audio_chunks)

    def wait_complete(self, timeout=30):
        self._complete_event.wait(timeout=timeout)
        return not self._error


def test_stream():
    """
    流式合成测试
    使用 SpeechSynthesizer.streaming_call() 边合成边播放
    """
    print("=" * 60)
    print("【流式合成测试】cosyvoice-v3-flash + 东北方言")
    print("=" * 60)

    if not DASHSCOPE_API_KEY:
        print("[错误] 请设置环境变量 DASHSCOPE_API_KEY")
        sys.exit(1)

    dashscope.api_key = DASHSCOPE_API_KEY
    dashscope.base_websocket_api_url = "wss://dashscope.aliyuncs.com/api-ws/v1/inference"

    os.makedirs(OUTPUT_DIR, exist_ok=True)

    callback = StreamingCallback()

    synthesizer = SpeechSynthesizer(
        model=MODEL,
        voice=VOICE,
        format=AudioFormat.PCM_24000HZ_MONO_16BIT,
        callback=callback,
    )

    print(f"[信息] 模型: {MODEL}")
    print(f"[信息] 音色: {VOICE}")

    # 模拟 LLM 分段输出:将文本分成多段,逐段发送
    text_chunks = [
        "哎呀妈呀,你可算来啦!",
        "长白山这嘎达可太美了,",
        "天池那水蓝得跟画似的,",
        "赶紧跟我走,咱上山顶看看去!",
    ]

    print(f"[信息] 将分 {len(text_chunks)} 段发送文本:")
    for i, chunk in enumerate(text_chunks):
        print(f"  段{i+1}: {chunk}")

    callback._start_time = time.time()
    print("[信息] 正在流式合成,边合成边播放...")

    for chunk in text_chunks:
        synthesizer.streaming_call(chunk)
        print(f"  [发送] {chunk}")
        time.sleep(0.1)

    print("[信息] 所有文本已发送,等待合成完成...")
    synthesizer.streaming_complete()

    success = callback.wait_complete(timeout=30)

    if success:
        all_audio = callback.get_all_audio()
        wav_path = os.path.join(OUTPUT_DIR, "test_stream_northeast.wav")
        save_pcm_as_wav(all_audio, wav_path)

        print(f"[指标] requestId: {synthesizer.get_last_request_id()}")
        print(f"[指标] 首包延迟: {synthesizer.get_first_package_delay()} 毫秒")
        print(f"[指标] 音频总大小: {len(all_audio)} 字节")
        print(f"[输出] 已保存: {wav_path}")
    else:
        print(f"[错误] 流式合成失败: {callback._error}")

    print("[完成] 流式合成测试结束\n")


# ==================== 主入口 ====================

def main():
    parser = argparse.ArgumentParser(description="阿里云 CosyVoice 东北方言 TTS 测试")
    parser.add_argument(
        "--mode",
        choices=["non-stream", "stream", "both"],
        default="both",
        help="测试模式: non-stream(非流式), stream(流式), both(两种都测)",
    )
    args = parser.parse_args()

    if not DASHSCOPE_API_KEY:
        print("[错误] 未检测到 DASHSCOPE_API_KEY 环境变量")
        print("       请执行: set DASHSCOPE_API_KEY=sk-xxx")
        print("       获取地址: https://platform.qianwenai.com/home/api-keys")
        sys.exit(1)

    print(f"\n阿里云 CosyVoice 东北方言 TTS 测试")
    print(f"API Key: {DASHSCOPE_API_KEY[:8]}...")
    print(f"输出目录: {OUTPUT_DIR}\n")

    if args.mode in ("non-stream", "both"):
        test_non_stream()

    if args.mode in ("stream", "both"):
        test_stream()

    print("全部测试完成!")


if __name__ == "__main__":
    main()


DASHSCOPE_API_KEY-设置为自己的APIkey
posted @ 2026-08-26 11:55  一克嗽  阅读(29)  评论(0)    收藏  举报