context.py 5.4 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198
  1. # -*- coding: utf-8 -*-
  2. """
  3. Audio Processing Context.
  4. Stellt Kontextinformationen für Audio-Prozessoren bereit.
  5. """
  6. from dataclasses import dataclass, field
  7. from enum import Enum, auto
  8. from typing import TYPE_CHECKING, Any
  9. if TYPE_CHECKING:
  10. from trixy_core.music.track import Track
  11. class AudioType(Enum):
  12. """
  13. Typ der Audio-Quelle.
  14. Ermöglicht Plugins, nur bestimmte Audio-Typen zu verarbeiten.
  15. """
  16. # Text-to-Speech Ausgabe
  17. TTS = auto()
  18. # Sound-Effekte aus dem Assets-Ordner
  19. ASSET = auto()
  20. # Musik-Streaming
  21. MUSIC = auto()
  22. # Unbekannt/Generic
  23. UNKNOWN = auto()
  24. @dataclass
  25. class AudioFormatInfo:
  26. """
  27. Audio-Format-Informationen für einen bestimmten Audio-Typ.
  28. Wird von Plugins genutzt um das korrekte Format zu kennen.
  29. """
  30. sample_rate: int = 44100
  31. channels: int = 2
  32. bit_depth: int = 16 # Bits pro Sample
  33. @property
  34. def sample_width(self) -> int:
  35. """Bytes pro Sample."""
  36. return self.bit_depth // 8
  37. @property
  38. def bytes_per_second(self) -> int:
  39. """Bytes pro Sekunde."""
  40. return self.sample_rate * self.channels * self.sample_width
  41. @property
  42. def bytes_per_ms(self) -> int:
  43. """Bytes pro Millisekunde."""
  44. return self.bytes_per_second // 1000
  45. @dataclass
  46. class AudioProcessingContext:
  47. """
  48. Kontext für Audio-Verarbeitung.
  49. Enthält alle Informationen die ein Prozessor benötigt,
  50. um Entscheidungen zu treffen (z.B. Ducking bei Wakeword,
  51. Crossfade am Track-Ende).
  52. Attributes:
  53. position_ms: Aktuelle Position im Track (Millisekunden)
  54. duration_ms: Gesamtdauer des Tracks (Millisekunden)
  55. chunk_duration_ms: Dauer des aktuellen Chunks (Millisekunden)
  56. track: Aktueller Track
  57. next_track: Nächster Track (falls vorhanden)
  58. sample_rate: Sample-Rate in Hz
  59. channels: Anzahl Kanäle (1=mono, 2=stereo)
  60. sample_width: Bytes pro Sample (2=16-bit)
  61. is_wakeword_active: Wakeword wurde erkannt
  62. is_conversation_active: Conversation läuft
  63. is_tts_playing: TTS wird gerade abgespielt
  64. volume: Aktuelle Master-Lautstärke (0.0 - 1.0)
  65. is_muted: Ist stummgeschaltet
  66. satellite_ids: Ziel-Satellites für diesen Chunk
  67. metadata: Zusätzliche Metadaten
  68. """
  69. # Audio-Typ (TTS, Asset, Musik)
  70. audio_type: AudioType = AudioType.UNKNOWN
  71. # Track-Position
  72. position_ms: int = 0
  73. duration_ms: int = 0
  74. chunk_duration_ms: int = 0
  75. # Track-Informationen
  76. track: "Track | None" = None
  77. next_track: "Track | None" = None
  78. # Audio-Format
  79. sample_rate: int = 44100
  80. channels: int = 2
  81. sample_width: int = 2 # 16-bit
  82. bit_depth: int = 16 # Bits pro Sample (für Kompatibilität)
  83. # Systemzustand
  84. is_wakeword_active: bool = False
  85. is_conversation_active: bool = False
  86. is_tts_playing: bool = False
  87. # Lautstärke
  88. volume: float = 1.0
  89. is_muted: bool = False
  90. # Ziel
  91. satellite_ids: list[str] = field(default_factory=list)
  92. # Erweiterbar
  93. metadata: dict[str, Any] = field(default_factory=dict)
  94. @property
  95. def remaining_ms(self) -> int:
  96. """Verbleibende Zeit bis Track-Ende in Millisekunden."""
  97. if self.duration_ms > 0:
  98. return max(0, self.duration_ms - self.position_ms)
  99. return 0
  100. @property
  101. def progress(self) -> float:
  102. """Fortschritt im Track (0.0 - 1.0)."""
  103. if self.duration_ms > 0:
  104. return min(1.0, self.position_ms / self.duration_ms)
  105. return 0.0
  106. @property
  107. def is_near_end(self) -> bool:
  108. """Prüft ob Track fast zu Ende ist (< 5 Sekunden)."""
  109. return self.remaining_ms < 5000
  110. @property
  111. def has_next_track(self) -> bool:
  112. """Prüft ob ein nächster Track vorhanden ist."""
  113. return self.next_track is not None
  114. @property
  115. def should_duck(self) -> bool:
  116. """Prüft ob Audio gedämpft werden sollte (Wakeword/Conversation/TTS)."""
  117. return self.is_wakeword_active or self.is_conversation_active or self.is_tts_playing
  118. @property
  119. def is_tts(self) -> bool:
  120. """Prüft ob Audio TTS ist."""
  121. return self.audio_type == AudioType.TTS
  122. @property
  123. def is_asset(self) -> bool:
  124. """Prüft ob Audio ein Asset-Sound ist."""
  125. return self.audio_type == AudioType.ASSET
  126. @property
  127. def is_music(self) -> bool:
  128. """Prüft ob Audio Musik ist."""
  129. return self.audio_type == AudioType.MUSIC
  130. @property
  131. def format_info(self) -> AudioFormatInfo:
  132. """Gibt Audio-Format-Informationen zurück."""
  133. return AudioFormatInfo(
  134. sample_rate=self.sample_rate,
  135. channels=self.channels,
  136. bit_depth=self.bit_depth,
  137. )
  138. @property
  139. def bytes_per_sample(self) -> int:
  140. """Bytes pro Sample (channels * sample_width)."""
  141. return self.channels * self.sample_width
  142. @property
  143. def samples_per_ms(self) -> float:
  144. """Samples pro Millisekunde."""
  145. return self.sample_rate / 1000.0
  146. def copy(self, **updates) -> "AudioProcessingContext":
  147. """
  148. Erstellt eine Kopie mit optionalen Updates.
  149. Args:
  150. **updates: Felder die überschrieben werden sollen
  151. Returns:
  152. Neue Kontext-Instanz
  153. """
  154. from dataclasses import asdict
  155. data = asdict(self)
  156. data.update(updates)
  157. return AudioProcessingContext(**data)