first commit

This commit is contained in:
2026-06-17 08:33:47 +02:00
commit 94cbf29972
6 changed files with 436 additions and 0 deletions

93
test_ws.py Normal file
View File

@@ -0,0 +1,93 @@
#!/usr/bin/env python3
"""WebSocket streaming test for Qwen3-TTS /v1/audio/speech/stream endpoint."""
import sys
import json
import time
import wave
import struct
import asyncio
try:
import websockets
except ImportError:
print("Installing websockets...")
import subprocess
subprocess.check_call([sys.executable, "-m", "pip", "install", "websockets"])
import websockets
WS_URL = "ws://localhost:8091/v1/audio/speech/stream"
OUTPUT_FILE = "test_stream_output.wav"
SAMPLE_RATE = 24000
async def stream_tts(text: str, voice: str = "Ryan", language: str = "German"):
print(f"[→] WebSocket connect: {WS_URL}")
print(f" text: {text!r}")
print(f" voice: {voice}, language: {language}")
pcm_chunks = []
total_bytes = 0
t0 = time.perf_counter()
first_chunk_time = None
async with websockets.connect(WS_URL) as ws:
# Send session config
await ws.send(json.dumps({
"type": "session.config",
"model": "Qwen/Qwen3-TTS-12Hz-1.7B-CustomVoice",
"task_type": "CustomVoice",
"voice": voice,
"language": language,
"response_format": "pcm",
"stream_audio": True,
}))
# Send text
await ws.send(json.dumps({
"type": "input.text",
"text": text,
}))
await ws.send(json.dumps({"type": "input.done"}))
# Receive
async for message in ws:
if isinstance(message, bytes):
if first_chunk_time is None:
first_chunk_time = time.perf_counter() - t0
print(f"[OK] First audio chunk in {first_chunk_time:.3f}s")
pcm_chunks.append(message)
total_bytes += len(message)
elif isinstance(message, str):
msg = json.loads(message)
msg_type = msg.get("type", "")
if msg_type == "session.done":
break
elif msg_type == "error":
print(f"[FAIL] Server error: {msg.get('message')}")
sys.exit(1)
elif msg_type in ("audio.start", "audio.done"):
print(f" [{msg_type}] sentence {msg.get('sentence_index', '?')}")
elapsed = time.perf_counter() - t0
# Write PCM chunks as WAV
raw_pcm = b"".join(pcm_chunks)
with wave.open(OUTPUT_FILE, "wb") as wf:
wf.setnchannels(1)
wf.setsampwidth(2) # int16
wf.setframerate(SAMPLE_RATE)
wf.writeframes(raw_pcm)
audio_duration = (total_bytes / 2) / SAMPLE_RATE
print(f"[OK] Done: {elapsed:.2f}s total, {total_bytes/1024:.1f} KB PCM")
print(f" Audio duration: {audio_duration:.2f}s → saved to {OUTPUT_FILE}")
if first_chunk_time:
print(f" First-chunk latency: {first_chunk_time:.3f}s")
if __name__ == "__main__":
asyncio.run(stream_tts(
text="Herzlich willkommen! Dieses System nutzt Qwen3-TTS für deutsche Sprachausgabe.",
voice="Ryan",
language="German",
))