lct-hack/backend/scripts/smoke_voice.py
2026-09-24 01:10:49 +03:00

40 lines
1.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Проверка локального TTS → Whisper small без микрофона и внешней сети.
Это не замер сквозной задержки звонка: TTS создаёт синтетический образец, а
загрузка моделей оплачивается отдельно. На Windows выполнить ту же команду.
"""
import time
import numpy as np
from scipy.signal import resample_poly
from app.voice.models import Synthesizer, WhisperRecognizer
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
PHRASE = "Пожар на улице Ленина, дом четырнадцать. Нужна помощь."
def main() -> None:
started = time.monotonic()
tts = Synthesizer(ROOT / "models" / "silero-tts" / "v5_ru.pt")
print(f"TTS загружен за {time.monotonic() - started:.2f} с", flush=True)
started = time.monotonic()
pcm = tts.synthesize(PHRASE)
audio = np.frombuffer(pcm, dtype=np.int16).astype(np.float32) / 32768
audio = resample_poly(audio, 2, 3).astype(np.float32)
print(f"TTS: {time.monotonic() - started:.2f} с, звук {len(audio) / 16000:.2f} с", flush=True)
started = time.monotonic()
stt = WhisperRecognizer("http://127.0.0.1:18082")
stt.warmup()
print(f"Whisper server готов за {time.monotonic() - started:.2f} с", flush=True)
started = time.monotonic()
answer = stt.transcribe(audio)
print(f"Whisper: {time.monotonic() - started:.2f} с", flush=True)
print(f"ожидалось: {PHRASE}\nполучено: {answer}", flush=True)
if __name__ == "__main__":
main()