From becd782f2f81ceb60f316f4809d4b44f7ba61a1b Mon Sep 17 00:00:00 2001 From: Liam Pettigrew Date: Fri, 6 Feb 2026 12:44:13 +1100 Subject: [PATCH] voice clone config and improved installation/launch --- .gitignore | 2 +- CLAUDE.md | 6 +- CONTRIBUTING.md | 3 + Dockerfile | 17 ++++-- Dockerfile_gpu | 42 +++++--------- SECURITY.md | 6 +- app.py | 8 ++- core/assistant.py | 15 +++-- core/tts.py | 71 ++++++++++++++--------- core/tts_tiny.py | 3 +- data/config.example.yml | 2 + launch.sh | 125 ++++++++++++++++++++++++++++------------ pyproject.toml | 5 +- readme.md | 7 +++ requirements.txt | 22 +++---- 15 files changed, 215 insertions(+), 119 deletions(-) diff --git a/.gitignore b/.gitignore index efee886..eb10556 100755 --- a/.gitignore +++ b/.gitignore @@ -96,5 +96,5 @@ data/voices/private # Keep Example Configs (negation patterns) # ============================================================================= !data/config.example.yml -!.env.example +!.env.example\ publish.sh diff --git a/CLAUDE.md b/CLAUDE.md index d481dbb..0ac2af6 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -11,6 +11,9 @@ Fulloch (the **Full**y **Loc**al **H**ome voice assistant) is a fully local, pri ### Development ```bash pip install -r requirements.txt +# Install special packages (see requirements.txt for details) +pip install --no-deps git+https://github.com/rekuenkdr/Qwen3-TTS-streaming.git@97da215 +# GPU only: pip install --no-build-isolation --no-deps git+https://github.com/Dao-AILab/flash-attention.git@ef9e6a6 pip install -e ".[dev]" # Install with dev dependencies python app.py ``` @@ -86,9 +89,10 @@ SILENCE_THRESHOLD = 0.001 # RMS threshold (lower = more sensitive) ``` ### Config Files (not in git) -- `data/config.yml`: Service endpoints, wakeword, integration settings +- `data/config.yml`: Service endpoints, wakeword, voice_clone, integration settings - `.env`: Credentials (Spotify, Google, etc.) - `data/models/`: Local model cache (~4-5GB) +- `data/voices/`: Voice clone reference files (wav/txt pairs) ### Example Config Files (in git) - `data/config.example.yml`: Template with all settings documented diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 09e786c..8dab930 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -11,6 +11,9 @@ Thank you for your interest in contributing to Fulloch! This document provides g python -m venv .venv source .venv/bin/activate # On Windows: .venv\Scripts\activate pip install -r requirements.txt + # Install special packages (see requirements.txt for details) + pip install --no-deps git+https://github.com/rekuenkdr/Qwen3-TTS-streaming.git@97da215 + # GPU only: pip install --no-build-isolation --no-deps git+https://github.com/Dao-AILab/flash-attention.git@ef9e6a6 pip install -e ".[dev]" # Install dev dependencies ``` 4. Copy configuration files: diff --git a/Dockerfile b/Dockerfile index 819ded0..6cf543d 100755 --- a/Dockerfile +++ b/Dockerfile @@ -2,16 +2,25 @@ FROM python:3.12-slim WORKDIR /app -# Install dependencies first to leverage Docker cache +# Install system dependencies +RUN apt-get update && apt-get install -y \ + sox \ + libsox-dev \ + libsox-fmt-all \ + ffmpeg \ + git \ + && rm -rf /var/lib/apt/lists/* + +# Install Python dependencies COPY requirements.txt . -RUN apt install sox libsox-dev libsox-fmt-all ffmpeg -RUN pip install --no-cache-dir -r requirements.txt +RUN pip install --no-cache-dir -r requirements.txt && \ + pip install --no-deps git+https://github.com/rekuenkdr/Qwen3-TTS-streaming.git@97da215 # Copy application code COPY app.py . +COPY core/ core/ COPY tools/ tools/ COPY utils/ utils/ -COPY wav/ wav/ COPY audio/ audio/ # Run the app diff --git a/Dockerfile_gpu b/Dockerfile_gpu index 8c2a04c..11a382e 100644 --- a/Dockerfile_gpu +++ b/Dockerfile_gpu @@ -1,45 +1,35 @@ -# Use NVIDIA CUDA base image (includes nvcc compiler for building) -FROM nvidia/cuda:12.4.1-devel-ubuntu22.04 +# Use PyTorch image with CUDA support (includes Python 3.11, PyTorch, CUDA) +FROM pytorch/pytorch:2.5.1-cuda12.4-cudnn9-devel -# Avoid interactive prompts during package installation +# Avoid interactive prompts ENV DEBIAN_FRONTEND=noninteractive -# Install Python 3.12 and build tools +# Install system dependencies RUN apt-get update && apt-get install -y \ - software-properties-common \ - && add-apt-repository ppa:deadsnakes/ppa \ - && apt-get update && apt-get install -y \ - python3.12 \ - python3.12-venv \ - python3.12-dev \ - python3-pip \ git \ - build-essential \ - cmake \ + sox \ + libsox-dev \ + libsox-fmt-all \ + ffmpeg \ && rm -rf /var/lib/apt/lists/* -# Set Python 3.12 as default -RUN ln -s /usr/bin/python3.12 /usr/bin/python - WORKDIR /app -# 2. Set Environment Variables to force CUDA build -# -DGGML_CUDA=on is the flag for recent llama-cpp-python versions (0.3.x) +# Set Environment Variables for llama-cpp-python CUDA build ENV CMAKE_ARGS="-DGGML_CUDA=on" ENV FORCE_CMAKE=1 +# Install Python dependencies COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt && \ + pip install --no-deps git+https://github.com/rekuenkdr/Qwen3-TTS-streaming.git@97da215 && \ + pip install flash-attn --no-build-isolation -# 3. Install dependencies -# This will now compile llama-cpp-python with CUDA support -RUN python3.12 -m pip install --upgrade pip && \ - python3.12 -m pip install --no-cache-dir -r requirements.txt - -# Copy application files (ignoring __pycache__ via .dockerignore) +# Copy application code COPY app.py . +COPY core/ core/ COPY tools/ tools/ COPY utils/ utils/ -COPY wav/ wav/ COPY audio/ audio/ -CMD ["python3.12", "app.py"] +CMD ["python", "app.py"] diff --git a/SECURITY.md b/SECURITY.md index 5897c2b..1363d6b 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -4,9 +4,9 @@ Fulloch is designed with privacy as a core principle. All processing happens locally on your device: -- **Speech Recognition**: Moonshine ASR runs entirely on-device -- **Text-to-Speech**: Kokoro TTS runs entirely on-device -- **Language Model**: Qwen runs entirely on-device via llama.cpp +- **Speech Recognition**: Qwen3 ASR runs entirely on-device (or Moonshine Tiny for edge devices) +- **Text-to-Speech**: Qwen3 TTS with voice cloning runs entirely on-device (or Kokoro for edge devices) +- **Language Model**: Qwen 3 4B runs entirely on-device via llama.cpp - **No Cloud Dependencies**: No data is sent to external servers for AI processing ## Reporting a Vulnerability diff --git a/app.py b/app.py index 720ec08..ae84660 100644 --- a/app.py +++ b/app.py @@ -31,7 +31,10 @@ os.environ["VLLM_NO_USAGE_STATS"] = "1" import logging -logging.basicConfig(level=logging.INFO) +logging.basicConfig( + level=logging.INFO, + format="%(asctime)s - %(name)s - %(levelname)s - %(message)s" +) logger = logging.getLogger(__name__) from core.assistant import Assistant @@ -41,10 +44,11 @@ WAKEWORD = config['general']['wakeword'] USE_AI = config['general']['use_ai'] USE_TINY_ASR = config['general'].get('use_tiny_asr', False) USE_TINY_TTS = config['general'].get('use_tiny_tts', False) +VOICE_CLONE = config['general'].get('voice_clone', None) def main(): """Main entry point for the voice assistant.""" - assistant = Assistant(wakeword=WAKEWORD, use_ai=USE_AI, use_tiny_asr=USE_TINY_ASR, use_tiny_tts=USE_TINY_TTS) + assistant = Assistant(wakeword=WAKEWORD, use_ai=USE_AI, use_tiny_asr=USE_TINY_ASR, use_tiny_tts=USE_TINY_TTS, voice_clone=VOICE_CLONE) assistant.run() diff --git a/core/assistant.py b/core/assistant.py index a6c0ae5..c0e7083 100644 --- a/core/assistant.py +++ b/core/assistant.py @@ -31,7 +31,7 @@ class Assistant: grammar: JSON grammar for structured output """ - def __init__(self, wakeword: str, use_ai: bool, use_tiny_asr: bool = False, use_tiny_tts: bool = False): + def __init__(self, wakeword: str, use_ai: bool, use_tiny_asr: bool = False, use_tiny_tts: bool = False, voice_clone: str = "cori"): """ Initialize the assistant. @@ -40,11 +40,13 @@ class Assistant: use_ai: Whether to use the SLM for intent detection use_tiny_asr: Whether to use Moonshine Tiny ASR instead of Qwen ASR use_tiny_tts: Whether to use Kokoro TTS instead of Qwen TTS + voice_clone: Name of the voice clone wav/txt file to use for TTS """ self.wakeword = wakeword.lower() self.use_ai = use_ai self.use_tiny_asr = use_tiny_asr self.use_tiny_tts = use_tiny_tts + self.voice_clone = voice_clone self.audio_capture = AudioCapture() # Models loaded lazily in transcriber thread @@ -54,6 +56,7 @@ class Assistant: self.grammar = None self.intent_prompt = None self.chat_prompt = None + self.voice_prompt = None def _load_models(self): """Load ASR, TTS and SLM models.""" @@ -71,8 +74,10 @@ class Assistant: from .tts_tiny import speak_stream, remove_emoji logger.info("Using Kokoro TTS") else: - from .tts import speak_stream, remove_emoji - logger.info("Using Qwen TTS") + from .tts import speak_stream, remove_emoji, set_voice, warmup_model + self.voice_prompt = set_voice(self.voice_clone) + logger.info(f"Using Qwen TTS with voice clone: {self.voice_clone}") + warmup_model(self.voice_prompt) self.remove_emoji = remove_emoji self.speak_stream = speak_stream @@ -142,7 +147,7 @@ class Assistant: "Just a second.", "Got it, let me think.", "Let's see." - ])) + ]), self.voice_prompt) logger.info(f"AI chat query: {user_prompt}") answer = generate_slm( @@ -198,7 +203,7 @@ class Assistant: # Process and respond answer = self._handle_wakeword(user_prompt) cleaned = self.remove_emoji(answer.replace('"', '').replace('*', '')) - self.speak_stream(cleaned) + self.speak_stream(cleaned, self.voice_prompt) # Resume transcription self.audio_capture.transcribing = True diff --git a/core/tts.py b/core/tts.py index e2a865a..4518c19 100644 --- a/core/tts.py +++ b/core/tts.py @@ -8,13 +8,16 @@ import queue import re import threading import torch + +# Set precision before any CUDA operations +torch.set_float32_matmul_precision('high') + import sounddevice as sd from qwen_tts import Qwen3TTSModel logger = logging.getLogger(__name__) -REF_AUDIO = "./data/voices/cori.wav" -REF_TEXT = "./data/voices/cori.txt" +VOICES_DIR = "./data/voices" # Device configuration DEVICE = "cuda" if torch.cuda.is_available() else "cpu" @@ -43,10 +46,7 @@ def remove_emoji(text: str, rem_think: bool = True) -> str: return EMOJI_PATTERN.sub("", text) -# Get reference audio text -with open(REF_TEXT) as f: - ref_text = f.read() - + # Load model model = Qwen3TTSModel.from_pretrained( "Qwen/Qwen3-TTS-12Hz-1.7B-Base", @@ -62,35 +62,52 @@ model.enable_streaming_optimizations( compile_mode="reduce-overhead", # Includes CUDA graphs automatically ) -# Create voice clone prompt from reference audio -prompt = model.create_voice_clone_prompt( - ref_audio=REF_AUDIO, - ref_text=ref_text, -) + +def set_voice(voice_name: str): + """ + Load voice clone prompt from audio/text files to use for TTS. + + Args: + voice_name: Name of the voice (matches wav/txt files in data/voices) + """ + ref_audio = f"{VOICES_DIR}/{voice_name}.wav" + ref_text_path = f"{VOICES_DIR}/{voice_name}.txt" + + with open(ref_text_path) as f: + ref_text = f.read() + + logger.info(f"Setting voice clone to: {voice_name}") + return model.create_voice_clone_prompt( + ref_audio=ref_audio, + ref_text=ref_text, + ) + # Warmup: run dummy generation to initialize torch.compile and CUDA graphs -logger.info("Warming up TTS model...") -for _ in model.stream_generate_voice_clone( - # Reference text must be longer to properly initialise model - text="A rainbow is a meteorological phenomenon that is caused by reflection, refraction and dispersion of light in water droplets resulting in a spectrum of light appearing in the sky.", - language="english", - voice_clone_prompt=prompt, - overlap_samples=512, - emit_every_frames=12, - decode_window_frames=80, - first_chunk_emit_every=5, - first_chunk_decode_window=48, - first_chunk_frames=48, -): - pass # Discard output -logger.info("TTS model ready") +def warmup_model(prompt): + logger.info("Warming up TTS model...") + for _ in model.stream_generate_voice_clone( + # Reference text must be longer to properly initialise model + text="A rainbow is a meteorological phenomenon that is caused by reflection, refraction and dispersion of light in water droplets.", + language="english", + voice_clone_prompt=prompt, + overlap_samples=512, + emit_every_frames=12, + decode_window_frames=80, + first_chunk_emit_every=5, + first_chunk_decode_window=48, + first_chunk_frames=48, + ): + pass # Discard output + logger.info("TTS model ready") -def speak_stream(text: str, voice: str = "cori", speed: float = 1.0): +def speak_stream(text: str, prompt, voice: str = "cori", speed: float = 1.0): """ Generate speech from text using stream optimised Qwen3 TTS from rekuenkdr Args: text: Text to synthesize + prompt: Prepared voice cloning prompt voice: Not used speed: Not used """ diff --git a/core/tts_tiny.py b/core/tts_tiny.py index deb2d42..e2a6f62 100644 --- a/core/tts_tiny.py +++ b/core/tts_tiny.py @@ -53,12 +53,13 @@ def remove_emoji(text: str, rem_think: bool = True) -> str: return EMOJI_PATTERN.sub("", text) -def speak_stream(text: str, voice: str = "af_bella", speed: float = 1.2): +def speak_stream(text: str, prompt=None, voice: str = "af_bella", speed: float = 1.2): """ Generate speech from text using Kokoro and stream to speakers. Args: text: Text to synthesize + prompt: Not used voice: Voice model to use (default: af_bella) speed: Speech speed multiplier (default: 1.2) """ diff --git a/data/config.example.yml b/data/config.example.yml index 13340e6..fddfb6b 100644 --- a/data/config.example.yml +++ b/data/config.example.yml @@ -15,6 +15,8 @@ general: use_tiny_asr: False # Use the Kokoro TTS for faster voice streaming (voice cloning not available) use_tiny_tts: False + # Set name of wav/txt file if using voice cloning + voice_clone: "cori" # ============================================================================= # Spotify Integration diff --git a/launch.sh b/launch.sh index 13dff01..20e84f9 100755 --- a/launch.sh +++ b/launch.sh @@ -1,6 +1,14 @@ #!/bin/bash set -e +# Helper function to ask user for confirmation +ask_download() { + local model_name="$1" + read -p "Download $model_name? (y/n): " response + response=${response,,} + [[ "$response" == "y" || "$response" == "yes" ]] +} + # Directory definitions BASE_DIR="$(pwd)/data/models" GRAMMAR_DIR="$BASE_DIR/grammars" @@ -26,74 +34,115 @@ fi echo "📂 Checking directory structure..." mkdir -p "$HUB_DIR" +# 2a. Check for config.yml +CONFIG_FILE="$(pwd)/data/config.yml" +CONFIG_EXAMPLE="$(pwd)/data/config.example.yml" + +if [ ! -f "$CONFIG_FILE" ]; then + echo "📝 config.yml not found. Creating from template..." + cp "$CONFIG_EXAMPLE" "$CONFIG_FILE" + echo "" + echo "⚠️ Please edit data/config.yml with your settings before continuing." + echo " See data/config.example.yml for documentation on each option." + echo "" + echo " Run ./launch.sh again when ready." + exit 0 +else + echo "✅ config.yml exists." +fi + # 3. Check and Download json.gbnf if [ ! -f "$GRAMMAR_DIR/json.gbnf" ]; then -echo "⬇️ Downloading json.gbnf..." -wget -q --show-progress -O "$GRAMMAR_DIR/json.gbnf" \ -"https://raw.githubusercontent.com/ggml-org/llama.cpp/master/grammars/json.gbnf" + if ask_download "json.gbnf (grammar file)"; then + echo "⬇️ Downloading json.gbnf..." + wget -q --show-progress -O "$GRAMMAR_DIR/json.gbnf" \ + "https://raw.githubusercontent.com/ggml-org/llama.cpp/master/grammars/json.gbnf" + else + echo "⏭️ Skipping json.gbnf" + fi else -echo "✅ json.gbnf exists." + echo "✅ json.gbnf exists." fi # 4. Check and Download Qwen3 GGUF if [ ! -f "$BASE_DIR/$QWEN_FILE" ]; then - echo "⬇️ Downloading $QWEN_FILE..." - huggingface-cli download "$QWEN_REPO" "$QWEN_FILE" \ - --local-dir "$BASE_DIR" \ - --local-dir-use-symlinks False + if ask_download "Qwen3 4B SLM (2.5GB)"; then + echo "⬇️ Downloading $QWEN_FILE..." + huggingface-cli download "$QWEN_REPO" "$QWEN_FILE" \ + --local-dir "$BASE_DIR" \ + --local-dir-use-symlinks False + else + echo "⏭️ Skipping Qwen3 SLM" + fi else echo "✅ $QWEN_FILE exists." fi # 5. Check and Download Qwen3 TTS model if [ ! -d "$TTS_DIR" ]; then -echo "⬇️ Downloading Qwen3 TTS..." -huggingface-cli download Qwen/Qwen3-TTS-12Hz-1.7B-Base \ ---cache-dir "$HUB_DIR" + if ask_download "Qwen3 TTS (3.4GB)"; then + echo "⬇️ Downloading Qwen3 TTS..." + huggingface-cli download Qwen/Qwen3-TTS-12Hz-1.7B-Base \ + --cache-dir "$HUB_DIR" + else + echo "⏭️ Skipping Qwen3 TTS" + fi else -echo "✅ Qwen3 TTS model exists." + echo "✅ Qwen3 TTS model exists." fi # 5a. Check and Download Kokoro-82M (TTS) if [ ! -d "$TTS_TINY_DIR" ]; then -echo "⬇️ Downloading Kokoro-82M..." -huggingface-cli download hexgrad/Kokoro-82M \ ---cache-dir "$HUB_DIR" + if ask_download "Kokoro-82M TTS Tiny (200MB)"; then + echo "⬇️ Downloading Kokoro-82M..." + huggingface-cli download hexgrad/Kokoro-82M \ + --cache-dir "$HUB_DIR" + else + echo "⏭️ Skipping Kokoro-82M" + fi else -echo "✅ Kokoro-82M exists." + echo "✅ Kokoro-82M exists." fi # 6. Check and Download Qwen3 ASR model if [ ! -d "$ASR_DIR" ]; then - echo "⬇️ Downloading Qwen3 ASR..." - huggingface-cli download Qwen/Qwen3-ASR-1.7B \ - --cache-dir "$HUB_DIR" + if ask_download "Qwen3 ASR (3.4GB)"; then + echo "⬇️ Downloading Qwen3 ASR..." + huggingface-cli download Qwen/Qwen3-ASR-1.7B \ + --cache-dir "$HUB_DIR" + else + echo "⏭️ Skipping Qwen3 ASR" + fi else echo "✅ Qwen3 ASR model exists." fi # 6a. Check and Download Moonshine Tiny ASR model if [ ! -d "$ASR_TINY_DIR" ]; then -echo "⬇️ Downloading Moonshine Tiny..." -huggingface-cli download UsefulSensors/moonshine-tiny \ ---cache-dir "$HUB_DIR" + if ask_download "Moonshine Tiny ASR (60MB)"; then + echo "⬇️ Downloading Moonshine Tiny..." + huggingface-cli download UsefulSensors/moonshine-tiny \ + --cache-dir "$HUB_DIR" + else + echo "⏭️ Skipping Moonshine Tiny" + fi else -echo "✅ Moonshine-tiny exists." + echo "✅ Moonshine-tiny exists." fi -# # 7. Prompt the user -# read -p "Are you using a GPU? (y/n): " response -# response=${response,,} -# if [[ "$response" == "y" || "$response" == "yes" ]]; then -# mv Dockerfile Dockerfile_cpu -# mv Dockerfile_gpu Dockerfile -# mv compose.yml compose_cpu.yml -# mv compose_gpu.yml compose.yml -# echo "✅ Using GPU enabled containers" -# else -# echo "✅ Using default containers" -# fi +# 7. Prompt the user +read -p "Are you using a GPU? (y/n): " response +response=${response,,} +if [[ "$response" == "y" || "$response" == "yes" ]]; then + mv Dockerfile Dockerfile_cpu + mv Dockerfile_gpu Dockerfile + mv compose.yml compose_cpu.yml + mv compose_gpu.yml compose.yml + echo "✅ Using GPU enabled containers" +else + echo "✅ Using default containers" +fi -# # 8. Launch Docker Compose -# echo "🚀 All files checked. Starting services..." -# docker compose up -d +# 8. Launch Docker Compose +echo "🚀 All files checked. Starting services..." +docker compose up -d diff --git a/pyproject.toml b/pyproject.toml index 7f5987d..27d75f4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -37,20 +37,23 @@ classifiers = [ ] # Core dependencies required for basic operation +# Note: qwen-tts and flash-attn require special install flags, see requirements.txt dependencies = [ - "numpy>=1.22.0", + "numpy>=1.24.0", "setuptools>=80.0.0", "pyyaml>=6.0", "python-dotenv>=1.0.0", # Audio "sounddevice>=0.5.0", "soundfile>=0.13.0", + "torchaudio>=2.0.0", # AI/ML Core "torch>=2.0.0", "transformers>=4.40.0", "kokoro>=0.9.0", "accelerate>=1.0.0", "llama-cpp-python>=0.3.0", + "qwen-asr>=0.0.6", ] [project.optional-dependencies] diff --git a/readme.md b/readme.md index 48abb81..ff3329d 100644 --- a/readme.md +++ b/readme.md @@ -71,6 +71,9 @@ A privacy-focused voice assistant that runs speech recognition, text-to-speech, git clone https://github.com/yourusername/fulloch.git cd fulloch pip install -r requirements.txt +# Install special packages (see requirements.txt for details) +pip install --no-deps git+https://github.com/rekuenkdr/Qwen3-TTS-streaming.git@97da215 +# GPU only: pip install --no-build-isolation --no-deps git+https://github.com/Dao-AILab/flash-attention.git@ef9e6a6 ``` ### 2. Configure @@ -123,6 +126,7 @@ general: use_ai: true # Enable SLM for intent detection use_tiny_asr: false # Use Moonshine Tiny ASR for edge devices use_tiny_tts: false # Use Kokoro TTS for edge devices + voice_clone: "cori" # Voice clone name for Qwen3 TTS ``` **ASR Options:** @@ -133,6 +137,9 @@ general: - `use_tiny_tts: false` (default) — Uses Qwen3-TTS with voice cloning for natural speech - `use_tiny_tts: true` — Uses Kokoro TTS for faster synthesis on low-resource edge devices +**Voice Cloning:** +- `voice_clone: "name"` — Specifies which voice to clone for Qwen3 TTS. Place your reference audio (`name.wav`) and transcript (`name.txt`) in `data/voices/`. Only used when `use_tiny_tts: false`. + ### Spotify 1. Create an app at [Spotify Developer Dashboard](https://developer.spotify.com/dashboard) diff --git a/requirements.txt b/requirements.txt index 1e7c7dd..89808d5 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,24 +1,26 @@ # Core dependencies -numpy==1.22.0 +numpy>=1.24.0 setuptools==80.9.0 xmltodict==1.0.2 word2number==1.1 beautifulsoup4==4.14.2 # Audio processing -sounddevice==0.5.3 -soundfile==0.13.1 -torchaudio==2.9.1 +sounddevice>=0.5.0 +soundfile>=0.13.0 # AI and ML -torch==2.9.1 -transformers==4.57.6 -kokoro==0.9.4 +torch>=2.4.0 +torchaudio>=2.4.0 +transformers==4.57.3 +kokoro>=0.9.4 accelerate==1.12.0 -llama_cpp_python==0.3.16 +llama_cpp_python>=0.3.16 qwen-asr==0.0.6 -git+https://github.com/rekuenkdr/Qwen3-TTS-streaming.git@97da215#egg=qwen-tts --no-deps -git+https://github.com/Dao-AILab/flash-attention.git@ef9e6a6#egg=flash-attn --no-build-isolation --no-deps + +# These packages require special install flags - see Dockerfile +# qwen-tts: pip install --no-deps git+https://github.com/rekuenkdr/Qwen3-TTS-streaming.git@97da215 +# flash-attn: pip install --no-build-isolation --no-deps git+https://github.com/Dao-AILab/flash-attention.git@ef9e6a6 # Smart home integration phue==1.1.0