|
|
vor 1 Jahr | |
|---|---|---|
| .. | ||
| README.md | vor 1 Jahr | |
| __init__.py | vor 1 Jahr | |
| audio_buffer.py | vor 1 Jahr | |
| audio_features.py | vor 1 Jahr | |
| config.py | vor 1 Jahr | |
| detection_engine.py | vor 1 Jahr | |
| detector.py | vor 1 Jahr | |
| examples.py | vor 1 Jahr | |
| model_loader.py | vor 1 Jahr | |
| optimization.py | vor 1 Jahr | |
A comprehensive, production-ready wakeword detection system for the Trixy voice assistant. This system provides real-time detection of custom wakewords using PyTorch models with advanced features for low-latency, high-accuracy performance.
from trixy_core.ml.wakeword import WakewordDetector, create_default_config
# Create configuration
config = create_default_config()
config.model_config.model_path = "path/to/your/model.pth"
config.model_config.model_password = "your_password"
# Initialize detector with event handler
detector = WakewordDetector(config, event_handler)
# Start detection
detector.start()
# Process audio data (numpy array, 16kHz, mono)
results = detector.process_audio(audio_data)
# Stop when done
detector.stop()
from trixy_core.ml.wakeword import create_wakeword_detector
# Simplified creation
detector = create_wakeword_detector(
model_path="path/to/model.pth",
model_password="password",
event_handler=event_handler,
custom_threshold=0.8,
system_command_threshold=0.9
)
with WakewordDetector(config, event_handler) as detector:
# Process audio
results = detector.process_audio(audio_data)
# Automatically stopped when exiting context
from trixy_core.ml.wakeword import (
create_lightweight_config,
create_high_performance_config,
auto_tune_config
)
# For resource-constrained devices
lightweight_config = create_lightweight_config()
# For high-performance systems
high_perf_config = create_high_performance_config()
# Auto-tuned for specific requirements
tuned_config = auto_tune_config(
base_config,
target_latency_ms=50.0,
available_memory_mb=256.0
)
from trixy_core.ml.wakeword import WakewordConfig, ModelConfig, AudioConfig
config = WakewordConfig(
model_config=ModelConfig(
model_path="models/wakeword/custom.pth",
model_password="secure_password",
device="cuda",
use_half_precision=True
),
audio_config=AudioConfig(
sample_rate=16000,
chunk_duration=1.0,
overlap_ratio=0.5
),
detection_config=DetectionConfig(
custom_threshold=0.75,
system_command_threshold=0.85,
use_temporal_smoothing=True
)
)
The system integrates with Trixy's event system to trigger wakeword_received events:
from trixy_core.events.decorators import TrixyEvent
class MyPlugin(TrixyPlugin):
@TrixyEvent(["wakeword_received"])
def on_wakeword_detected(self, event_name, event_data):
wakeword_type = event_data.wakeword_type # "custom" or "system_command"
confidence = event_data.confidence
satellite_id = event_data.satellite_info.satellite_id
if wakeword_type == "system_command":
# Handle admin commands
self.handle_system_command()
else:
# Handle regular wakeword
self.start_conversation()
from trixy_core.ml.wakeword import SecureModelLoader
loader = SecureModelLoader(device=torch.device("cuda"))
model, metadata = loader.load_model(
model_path="model.pth",
password="password",
verify_hash=True
)
print(f"Loaded model: {metadata.model_name}")
print(f"Classes: {metadata.class_labels}")
print(f"Input shape: {metadata.input_shape}")
Models include comprehensive metadata:
metadata = ModelMetadata(
model_name="Custom Wakeword v2.1",
model_version="2.1.0",
model_type="wakeword",
architecture="ImprovedRepCNN",
num_classes=3,
class_labels=["custom", "system_command", "negative"],
final_accuracy=0.956,
sample_rate=16000,
n_mels=40,
time_frames=151
)
from trixy_core.ml.wakeword import PerformanceMonitor
monitor = PerformanceMonitor(
monitoring_interval=1.0,
memory_warning_threshold=512.0,
latency_warning_threshold=100.0
)
monitor.start_monitoring()
# ... process audio ...
monitor.stop_monitoring()
stats = monitor.get_performance_summary()
print(f"Average latency: {stats['avg_inference_latency_ms']:.1f}ms")
from trixy_core.ml.wakeword import ModelOptimizer, optimize_for_deployment
# Benchmark model performance
optimizer = ModelOptimizer()
benchmark_results = optimizer.benchmark_model(
model, input_shape=(1, 40, 151), device=device
)
# Optimize for deployment
optimized_model, report = optimize_for_deployment(
model,
config={'enable_torchscript': True, 'enable_compilation': True},
example_input
)
from trixy_core.ml.wakeword import WakewordTester, AudioSimulator
# Create test environment
audio_sim = AudioSimulator(sample_rate=16000)
tester = WakewordTester(detector, audio_sim)
# Run various tests
basic_result = tester.run_basic_test()
perf_result = tester.run_performance_test(duration_seconds=30.0)
accuracy_result = tester.run_accuracy_test(test_cases)
latency_result = tester.run_latency_test(num_iterations=100)
# Generate report
report = tester.get_test_report()
tester.save_test_report("test_results.json")
# Generate test audio
silence = audio_sim.generate_silence(1.0)
noise = audio_sim.generate_white_noise(1.0, amplitude=0.1)
synthetic_wakeword = audio_sim.generate_synthetic_wakeword(1.0)
test_sequence = audio_sim.create_test_sequence()
┌─────────────────┐ ┌──────────────────┐ ┌─────────────────┐
│ Audio Input │───▶│ Audio Buffer │───▶│ Feature Extract │
└─────────────────┘ └──────────────────┘ └─────────────────┘
│
┌─────────────────┐ ┌──────────────────┐ ┌─────────────────┐
│ Event System │◀───│ Detection Engine │◀───│ Model Loader │
└─────────────────┘ └──────────────────┘ └─────────────────┘
The system uses RepCNN (Reparameterizable CNN) models optimized for wakeword detection:
{
"model_config": {
"model_path": "models/wakeword/default/model.pth",
"model_password": "",
"device": "auto",
"use_model_optimization": true
},
"audio_config": {
"sample_rate": 16000,
"chunk_duration": 1.0,
"overlap_ratio": 0.5,
"spectrogram_config": {
"n_mels": 40,
"time_frames": 151,
"hop_length": 160
}
},
"detection_config": {
"custom_threshold": 0.7,
"system_command_threshold": 0.8,
"use_temporal_smoothing": true
}
}
export TRIXY_WAKEWORD_MODEL_PATH="/path/to/model.pth"
export TRIXY_WAKEWORD_MODEL_PASSWORD="secure_password"
export TRIXY_WAKEWORD_DEVICE="cuda"
export TRIXY_WAKEWORD_CUSTOM_THRESHOLD="0.75"
export TRIXY_WAKEWORD_SYSTEM_THRESHOLD="0.85"
export TRIXY_WAKEWORD_DEBUG_MODE="true"
class TrixyClient:
def __init__(self):
config = create_default_config()
config.satellite_id = self.get_satellite_id()
self.wakeword_detector = WakewordDetector(config, self.event_handler)
def start_audio_processing(self):
self.wakeword_detector.start()
# Start audio capture thread
self.audio_thread = threading.Thread(target=self.audio_capture_loop)
self.audio_thread.start()
def audio_capture_loop(self):
while self.running:
audio_data = self.capture_audio_chunk()
self.wakeword_detector.process_audio(audio_data)
class WakewordPlugin(TrixyPlugin):
def __init__(self):
super().__init__()
self.detector = create_wakeword_detector(
config_path=self.config['wakeword_config'],
event_handler=self.application.event_handler
)
def on_plugin_start(self):
self.detector.start()
def on_plugin_stop(self):
self.detector.stop()
@TrixyEvent(["audio_chunk_received"])
def process_audio_chunk(self, event_name, event_data):
self.detector.process_audio(event_data.audio_data)
Model Loading Errors
High Latency
High Memory Usage
Low Accuracy
Enable debug mode for detailed logging:
config.debug_mode = True
config.save_debug_audio = True
config.debug_output_path = "/tmp/wakeword_debug"
When contributing to the wakeword detection system:
This wakeword detection system is part of the Trixy voice assistant project.