Qwen3 ASR as default with tiny ASR edge as option

This commit is contained in:
Liam Pettigrew
2026-02-04 15:59:01 +11:00
parent ff27dc5c7e
commit e7223bd32e
4 changed files with 26 additions and 8 deletions
+2 -1
View File
@@ -39,10 +39,11 @@ from core.assistant import Assistant
# Configuration
WAKEWORD = config['general']['wakeword']
USE_AI = config['general']['use_ai']
USE_TINY_ASR = config['general'].get('use_tiny_asr', False)
def main():
"""Main entry point for the voice assistant."""
assistant = Assistant(wakeword=WAKEWORD, use_ai=USE_AI)
assistant = Assistant(wakeword=WAKEWORD, use_ai=USE_AI, use_tiny_asr=USE_TINY_ASR)
assistant.run()
+14 -4
View File
@@ -11,7 +11,6 @@ import threading
from typing import Optional
from .audio import AudioCapture
from .asr import load_asr_model, stream_generator
from .tts import speak_stream, remove_emoji
from .slm import load_slm, generate_slm
@@ -34,20 +33,23 @@ class Assistant:
grammar: JSON grammar for structured output
"""
def __init__(self, wakeword: str, use_ai: bool):
def __init__(self, wakeword: str, use_ai: bool, use_tiny_asr: bool = False):
"""
Initialize the assistant.
Args:
wakeword: Activation phrase to listen for
slm_model_path: Path to SLM model (empty to disable AI)
use_ai: Whether to use the SLM for intent detection
use_tiny_asr: Whether to use Moonshine Tiny ASR instead of Qwen ASR
"""
self.wakeword = wakeword.lower()
self.use_ai = use_ai
self.use_tiny_asr = use_tiny_asr
self.audio_capture = AudioCapture()
# Models loaded lazily in transcriber thread
self.asr_pipe = None
self.asr_stream_generator = None
self.slm_model = None
self.grammar = None
self.intent_prompt = None
@@ -55,7 +57,15 @@ class Assistant:
def _load_models(self):
"""Load ASR and optionally SLM models."""
if self.use_tiny_asr:
from .asr_tiny import load_asr_model, stream_generator
logger.info("Using Moonshine Tiny ASR")
else:
from .asr import load_asr_model, stream_generator
logger.info("Using Qwen ASR")
self.asr_pipe = load_asr_model()
self.asr_stream_generator = stream_generator
if self.use_ai:
self.grammar, self.slm_model = load_slm()
@@ -151,7 +161,7 @@ class Assistant:
logger.info("Transcriber started")
for result in self.asr_pipe(
stream_generator(self.audio_capture.audio_queue),
self.asr_stream_generator(self.audio_capture.audio_queue),
batch_size=1,
generate_kwargs={"max_new_tokens": 256}
):
+10 -3
View File
@@ -28,7 +28,7 @@ A privacy-focused voice assistant that runs speech recognition, text-to-speech,
+--------+---------+
|
+--------v---------+
| Moonshine ASR |
| Qwen3 ASR |
| (Speech→Text) |
+--------+---------+
|
@@ -92,7 +92,7 @@ The launch script handles model downloads automatically:
Or manually download:
- [Qwen3-4B-Instruct GGUF](https://huggingface.co/Qwen) → `data/models/`
- Moonshine and Kokoro download automatically on first run
- Qwen3-ASR, Moonshine Tiny, and Kokoro download automatically on first run
### 4. Run
@@ -120,8 +120,14 @@ The launch script:
```yaml
general:
wakeword: "computer" # Activation phrase
use_ai: true # Enable SLM for intent detection
use_tiny_asr: false # Use Moonshine Tiny ASR for edge devices
```
**ASR Options:**
- `use_tiny_asr: false` (default) — Uses Qwen3-ASR-0.6B for higher accuracy
- `use_tiny_asr: true` — Uses Moonshine Tiny for low-resource edge devices
### Spotify
1. Create an app at [Spotify Developer Dashboard](https://developer.spotify.com/dashboard)
@@ -226,7 +232,8 @@ fulloch/
├── app.py # Entry point
├── core/ # Core modules
│ ├── audio.py # Audio capture and silence detection
│ ├── asr.py # Moonshine speech recognition
│ ├── asr.py # Qwen3 ASR (default)
│ ├── asr_tiny.py # Moonshine Tiny ASR (edge devices)
│ ├── tts.py # Kokoro text-to-speech
│ ├── slm.py # Qwen language model
│ └── assistant.py # Main orchestration