tone.py 8.4 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272
  1. # -*- coding: utf-8 -*-
  2. """
  3. Tone/Emotion-Analyse fuer Spracheingaben.
  4. Erkennt den emotionalen Ton eines Satzes anhand von
  5. Schlagwoertern und Satzstruktur.
  6. Schlagwoerter sind in einer externen JSON-Datei definiert
  7. (config/tone_keywords.json) und koennen pro Sprache angepasst werden.
  8. Events:
  9. - tone_detected: Wird fuer jeden erkannten Tone-Tag ausgeloest
  10. Daten: tag, category, score, text
  11. Plugins koennen auf die Tags reagieren:
  12. ```python
  13. @TrixyEvent(["tone_detected"])
  14. async def on_tone(self, event_name, data):
  15. if data["tag"] == "rude":
  16. # Freche Antwort geben
  17. ```
  18. """
  19. from __future__ import annotations
  20. import json
  21. import re
  22. from dataclasses import dataclass, field
  23. from pathlib import Path
  24. from typing import Any
  25. from trixy_core.utils.debug import pdebug, pinfo, pwarn
  26. # Tone-Kategorien
  27. TONE_CATEGORIES: dict[str, list[str]] = {
  28. "friendliness": ["friendly", "polite", "nice", "grateful"],
  29. "negativity": ["rude", "aggressive", "frustrated", "impatient", "annoyed", "sarcastic"],
  30. "emotion": ["happy", "sad", "surprised", "scared", "excited", "bored"],
  31. "personal": ["romantic", "cute", "shy", "flirty", "sexist"],
  32. "style": ["commanding", "questioning", "casual", "formal", "playful", "urgent", "whispering"],
  33. "special": ["confused", "apologetic", "motivating"],
  34. }
  35. # Alle gueltige Tags (flat)
  36. ALL_TONE_TAGS: set[str] = {tag for tags in TONE_CATEGORIES.values() for tag in tags}
  37. # Tag → Kategorie Lookup
  38. TAG_TO_CATEGORY: dict[str, str] = {
  39. tag: cat for cat, tags in TONE_CATEGORIES.items() for tag in tags
  40. }
  41. @dataclass
  42. class ToneResult:
  43. """Ein erkannter Tone-Tag mit Score und Kategorie."""
  44. tag: str
  45. category: str
  46. score: float
  47. def to_dict(self) -> dict[str, Any]:
  48. return {"tag": self.tag, "category": self.category, "score": self.score}
  49. class ToneAnalyzer:
  50. """
  51. Analysiert den emotionalen Ton von Texten.
  52. Laedt Schlagwoerter aus einer JSON-Datei und erkennt
  53. Tone-Tags anhand von Keywords und Satzstruktur.
  54. Beispiel:
  55. ```python
  56. analyzer = ToneAnalyzer()
  57. analyzer.load_keywords("config/tone_keywords.json")
  58. results = analyzer.analyze("Koenntest du bitte das Licht anmachen?")
  59. # [ToneResult(tag="friendly", category="friendliness", score=0.8),
  60. # ToneResult(tag="polite", category="friendliness", score=0.7),
  61. # ToneResult(tag="questioning", category="style", score=0.7)]
  62. ```
  63. """
  64. def __init__(self) -> None:
  65. # Keyword-Regeln: {keyword: [(tag, score), ...]}
  66. self._keyword_rules: dict[str, list[tuple[str, float]]] = {}
  67. # Imperative Verben (fuer commanding-Erkennung)
  68. self._imperative_verbs: list[str] = []
  69. # Score-Schwelle
  70. self._threshold: float = 0.5
  71. self._loaded = False
  72. @property
  73. def is_loaded(self) -> bool:
  74. """True wenn Schlagwoerter geladen."""
  75. return self._loaded
  76. @property
  77. def keyword_count(self) -> int:
  78. """Anzahl geladener Schlagwoerter."""
  79. return len(self._keyword_rules)
  80. def load_keywords(self, path: str | Path) -> bool:
  81. """
  82. Laedt Schlagwoerter aus einer JSON-Datei.
  83. Format:
  84. ```json
  85. {
  86. "threshold": 0.5,
  87. "imperative_verbs": ["mach", "stell", "sag", ...],
  88. "keywords": {
  89. "bitte": {"friendly": 0.7, "polite": 0.5},
  90. "verdammt": {"aggressive": 0.7, "frustrated": 0.6},
  91. ...
  92. }
  93. }
  94. ```
  95. Returns:
  96. True bei Erfolg
  97. """
  98. path = Path(path)
  99. if not path.is_file():
  100. pdebug(f"ToneAnalyzer: Keywords nicht gefunden: {path}")
  101. return False
  102. try:
  103. with open(path, encoding="utf-8") as f:
  104. data = json.load(f)
  105. self._threshold = data.get("threshold", 0.5)
  106. self._imperative_verbs = data.get("imperative_verbs", [])
  107. self._keyword_rules.clear()
  108. keywords = data.get("keywords", {})
  109. for keyword, tag_scores in keywords.items():
  110. rules = []
  111. for tag, score in tag_scores.items():
  112. if tag in ALL_TONE_TAGS:
  113. rules.append((tag, float(score)))
  114. else:
  115. pwarn(f"ToneAnalyzer: Unbekannter Tag '{tag}' fuer Keyword '{keyword}'")
  116. if rules:
  117. self._keyword_rules[keyword.lower()] = rules
  118. self._loaded = True
  119. pinfo(f"ToneAnalyzer: {len(self._keyword_rules)} Keywords geladen aus {path}")
  120. return True
  121. except (json.JSONDecodeError, OSError) as e:
  122. pwarn(f"ToneAnalyzer: Fehler beim Laden: {e}")
  123. return False
  124. def analyze(self, text: str) -> list[ToneResult]:
  125. """
  126. Analysiert den emotionalen Ton eines Textes.
  127. Args:
  128. text: Eingabetext
  129. Returns:
  130. Liste der erkannten ToneResults (sortiert nach Score, hoechster zuerst)
  131. """
  132. text_lower = text.lower().strip()
  133. scores: dict[str, float] = {}
  134. # 1. Keyword-Matching (laengste Keywords zuerst)
  135. if self._keyword_rules:
  136. for keyword in sorted(self._keyword_rules, key=len, reverse=True):
  137. if keyword in text_lower:
  138. for tag, score in self._keyword_rules[keyword]:
  139. scores[tag] = max(scores.get(tag, 0.0), score)
  140. # 2. Satzstruktur-Analyse
  141. # Fragezeichen → questioning
  142. if "?" in text:
  143. scores["questioning"] = max(scores.get("questioning", 0.0), 0.7)
  144. # Ausrufezeichen → commanding oder excited
  145. if "!" in text:
  146. if scores.get("happy", 0) > 0.5 or scores.get("excited", 0) > 0.3:
  147. scores["excited"] = max(scores.get("excited", 0.0), 0.6)
  148. else:
  149. scores["commanding"] = max(scores.get("commanding", 0.0), 0.4)
  150. # Kurzer Imperativ ohne Hoeflichkeit → commanding
  151. words = text_lower.split()
  152. if (len(words) <= 5
  153. and self._imperative_verbs
  154. and words
  155. and words[0] in self._imperative_verbs
  156. and "friendly" not in scores
  157. and "polite" not in scores):
  158. scores["commanding"] = max(scores.get("commanding", 0.0), 0.6)
  159. scores["rude"] = max(scores.get("rude", 0.0), 0.3)
  160. # GROSSBUCHSTABEN → aggressive/urgent
  161. upper_words = sum(1 for w in text.split() if w.isupper() and len(w) > 1)
  162. if upper_words >= 2:
  163. scores["aggressive"] = max(scores.get("aggressive", 0.0), 0.5)
  164. scores["urgent"] = max(scores.get("urgent", 0.0), 0.4)
  165. # 3. Schwelle anwenden und Ergebnisse erstellen
  166. results = []
  167. for tag, score in scores.items():
  168. if score >= self._threshold and tag in ALL_TONE_TAGS:
  169. category = TAG_TO_CATEGORY.get(tag, "unknown")
  170. results.append(ToneResult(tag=tag, category=category, score=score))
  171. # Sortieren nach Score (hoechster zuerst)
  172. results.sort(key=lambda r: r.score, reverse=True)
  173. return results
  174. # === Globale Instanz ===
  175. _analyzer: ToneAnalyzer | None = None
  176. def get_analyzer() -> ToneAnalyzer:
  177. """Gibt die globale ToneAnalyzer-Instanz zurueck."""
  178. global _analyzer
  179. if _analyzer is None:
  180. _analyzer = ToneAnalyzer()
  181. return _analyzer
  182. def analyze_tone(text: str) -> list[ToneResult]:
  183. """Shortcut: Analysiert Text mit der globalen Instanz."""
  184. return get_analyzer().analyze(text)
  185. def get_all_tone_tags() -> list[str]:
  186. """Gibt alle verfuegbaren Tone-Tags als Strings zurueck."""
  187. return sorted(ALL_TONE_TAGS)
  188. def get_tone_categories() -> dict[str, list[str]]:
  189. """Gibt alle Kategorien mit ihren Tags zurueck."""
  190. return dict(TONE_CATEGORIES)
  191. async def emit_tone_events(
  192. results: list[ToneResult],
  193. text: str,
  194. events: object,
  195. ) -> None:
  196. """
  197. Emittiert tone_detected Events fuer jeden erkannten Tag.
  198. Args:
  199. results: Ergebnis von analyze_tone()
  200. text: Eingabetext
  201. events: EventManager-Instanz
  202. """
  203. if not results or not events:
  204. return
  205. emit = getattr(events, "emit", None)
  206. if not emit:
  207. return
  208. for result in results:
  209. await emit("tone_detected", {
  210. "tag": result.tag,
  211. "category": result.category,
  212. "score": result.score,
  213. "text": text,
  214. })