mirror of
https://github.com/Nighthawk42/Qwen3-TTS-streaming.git
synced 2026-08-30 08:52:27 +00:00
Update language and text for voice cloning example
This commit is contained in:
@@ -23,11 +23,11 @@ clone_model = Qwen3TTSModel.from_pretrained(
|
||||
start = log_time(start, "Load Base model")
|
||||
|
||||
# for real speedup, use vLLM for LM inference (or SGlang probably)
|
||||
# torch.compile doesn't help much for autoregressive generation due to dynamic shapes ( I think but idk )
|
||||
# torch.compile doesn't help much for autoregressive generation due to dynamic shapes
|
||||
|
||||
ref_audio_path = "neurona-10sec (3).wav"
|
||||
ref_audio_path = "ref-audio.wav"
|
||||
ref_text = (
|
||||
"реф текст"
|
||||
"ref text"
|
||||
)
|
||||
|
||||
voice_clone_prompt = clone_model.create_voice_clone_prompt(
|
||||
@@ -37,14 +37,14 @@ voice_clone_prompt = clone_model.create_voice_clone_prompt(
|
||||
start = log_time(start, "Create voice clone prompt")
|
||||
|
||||
# Test sentence
|
||||
test_text = "Привет всем! Я того всё ебала, что за новый голос тут на обзоре у вилсакома? А? Так он мне понравился. Ганс оф буллщит."
|
||||
test_text = "Hello! This is the test text"
|
||||
|
||||
# ============== Standard generation ==============
|
||||
print("\n--- Standard generation ---")
|
||||
start = time.time()
|
||||
wavs, sr = clone_model.generate_voice_clone(
|
||||
text=test_text,
|
||||
language="Russian",
|
||||
language="English",
|
||||
voice_clone_prompt=voice_clone_prompt,
|
||||
)
|
||||
standard_time = time.time() - start
|
||||
@@ -60,7 +60,7 @@ chunk_count = 0
|
||||
|
||||
for chunk, chunk_sr in clone_model.stream_generate_voice_clone(
|
||||
text=test_text,
|
||||
language="Russian",
|
||||
language="English",
|
||||
voice_clone_prompt=voice_clone_prompt,
|
||||
emit_every_frames=8,
|
||||
decode_window_frames=80,
|
||||
@@ -85,4 +85,4 @@ print(f"Streaming first chunk: {first_chunk_time:.2f}s")
|
||||
print(f"Streaming total: {streaming_time:.2f}s")
|
||||
print(f"Latency improvement: {standard_time - first_chunk_time:.2f}s faster to first audio")
|
||||
|
||||
print(f"\n[{time.time() - total_start:.2f}s] TOTAL")
|
||||
print(f"\n[{time.time() - total_start:.2f}s] TOTAL")
|
||||
|
||||
Reference in New Issue
Block a user