| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272 |
- # -*- coding: utf-8 -*-
- """
- Tone/Emotion-Analyse fuer Spracheingaben.
- Erkennt den emotionalen Ton eines Satzes anhand von
- Schlagwoertern und Satzstruktur.
- Schlagwoerter sind in einer externen JSON-Datei definiert
- (config/tone_keywords.json) und koennen pro Sprache angepasst werden.
- Events:
- - tone_detected: Wird fuer jeden erkannten Tone-Tag ausgeloest
- Daten: tag, category, score, text
- Plugins koennen auf die Tags reagieren:
- ```python
- @TrixyEvent(["tone_detected"])
- async def on_tone(self, event_name, data):
- if data["tag"] == "rude":
- # Freche Antwort geben
- ```
- """
- from __future__ import annotations
- import json
- import re
- from dataclasses import dataclass, field
- from pathlib import Path
- from typing import Any
- from trixy_core.utils.debug import pdebug, pinfo, pwarn
- # Tone-Kategorien
- TONE_CATEGORIES: dict[str, list[str]] = {
- "friendliness": ["friendly", "polite", "nice", "grateful"],
- "negativity": ["rude", "aggressive", "frustrated", "impatient", "annoyed", "sarcastic"],
- "emotion": ["happy", "sad", "surprised", "scared", "excited", "bored"],
- "personal": ["romantic", "cute", "shy", "flirty", "sexist"],
- "style": ["commanding", "questioning", "casual", "formal", "playful", "urgent", "whispering"],
- "special": ["confused", "apologetic", "motivating"],
- }
- # Alle gueltige Tags (flat)
- ALL_TONE_TAGS: set[str] = {tag for tags in TONE_CATEGORIES.values() for tag in tags}
- # Tag → Kategorie Lookup
- TAG_TO_CATEGORY: dict[str, str] = {
- tag: cat for cat, tags in TONE_CATEGORIES.items() for tag in tags
- }
- @dataclass
- class ToneResult:
- """Ein erkannter Tone-Tag mit Score und Kategorie."""
- tag: str
- category: str
- score: float
- def to_dict(self) -> dict[str, Any]:
- return {"tag": self.tag, "category": self.category, "score": self.score}
- class ToneAnalyzer:
- """
- Analysiert den emotionalen Ton von Texten.
- Laedt Schlagwoerter aus einer JSON-Datei und erkennt
- Tone-Tags anhand von Keywords und Satzstruktur.
- Beispiel:
- ```python
- analyzer = ToneAnalyzer()
- analyzer.load_keywords("config/tone_keywords.json")
- results = analyzer.analyze("Koenntest du bitte das Licht anmachen?")
- # [ToneResult(tag="friendly", category="friendliness", score=0.8),
- # ToneResult(tag="polite", category="friendliness", score=0.7),
- # ToneResult(tag="questioning", category="style", score=0.7)]
- ```
- """
- def __init__(self) -> None:
- # Keyword-Regeln: {keyword: [(tag, score), ...]}
- self._keyword_rules: dict[str, list[tuple[str, float]]] = {}
- # Imperative Verben (fuer commanding-Erkennung)
- self._imperative_verbs: list[str] = []
- # Score-Schwelle
- self._threshold: float = 0.5
- self._loaded = False
- @property
- def is_loaded(self) -> bool:
- """True wenn Schlagwoerter geladen."""
- return self._loaded
- @property
- def keyword_count(self) -> int:
- """Anzahl geladener Schlagwoerter."""
- return len(self._keyword_rules)
- def load_keywords(self, path: str | Path) -> bool:
- """
- Laedt Schlagwoerter aus einer JSON-Datei.
- Format:
- ```json
- {
- "threshold": 0.5,
- "imperative_verbs": ["mach", "stell", "sag", ...],
- "keywords": {
- "bitte": {"friendly": 0.7, "polite": 0.5},
- "verdammt": {"aggressive": 0.7, "frustrated": 0.6},
- ...
- }
- }
- ```
- Returns:
- True bei Erfolg
- """
- path = Path(path)
- if not path.is_file():
- pdebug(f"ToneAnalyzer: Keywords nicht gefunden: {path}")
- return False
- try:
- with open(path, encoding="utf-8") as f:
- data = json.load(f)
- self._threshold = data.get("threshold", 0.5)
- self._imperative_verbs = data.get("imperative_verbs", [])
- self._keyword_rules.clear()
- keywords = data.get("keywords", {})
- for keyword, tag_scores in keywords.items():
- rules = []
- for tag, score in tag_scores.items():
- if tag in ALL_TONE_TAGS:
- rules.append((tag, float(score)))
- else:
- pwarn(f"ToneAnalyzer: Unbekannter Tag '{tag}' fuer Keyword '{keyword}'")
- if rules:
- self._keyword_rules[keyword.lower()] = rules
- self._loaded = True
- pinfo(f"ToneAnalyzer: {len(self._keyword_rules)} Keywords geladen aus {path}")
- return True
- except (json.JSONDecodeError, OSError) as e:
- pwarn(f"ToneAnalyzer: Fehler beim Laden: {e}")
- return False
- def analyze(self, text: str) -> list[ToneResult]:
- """
- Analysiert den emotionalen Ton eines Textes.
- Args:
- text: Eingabetext
- Returns:
- Liste der erkannten ToneResults (sortiert nach Score, hoechster zuerst)
- """
- text_lower = text.lower().strip()
- scores: dict[str, float] = {}
- # 1. Keyword-Matching (laengste Keywords zuerst)
- if self._keyword_rules:
- for keyword in sorted(self._keyword_rules, key=len, reverse=True):
- if keyword in text_lower:
- for tag, score in self._keyword_rules[keyword]:
- scores[tag] = max(scores.get(tag, 0.0), score)
- # 2. Satzstruktur-Analyse
- # Fragezeichen → questioning
- if "?" in text:
- scores["questioning"] = max(scores.get("questioning", 0.0), 0.7)
- # Ausrufezeichen → commanding oder excited
- if "!" in text:
- if scores.get("happy", 0) > 0.5 or scores.get("excited", 0) > 0.3:
- scores["excited"] = max(scores.get("excited", 0.0), 0.6)
- else:
- scores["commanding"] = max(scores.get("commanding", 0.0), 0.4)
- # Kurzer Imperativ ohne Hoeflichkeit → commanding
- words = text_lower.split()
- if (len(words) <= 5
- and self._imperative_verbs
- and words
- and words[0] in self._imperative_verbs
- and "friendly" not in scores
- and "polite" not in scores):
- scores["commanding"] = max(scores.get("commanding", 0.0), 0.6)
- scores["rude"] = max(scores.get("rude", 0.0), 0.3)
- # GROSSBUCHSTABEN → aggressive/urgent
- upper_words = sum(1 for w in text.split() if w.isupper() and len(w) > 1)
- if upper_words >= 2:
- scores["aggressive"] = max(scores.get("aggressive", 0.0), 0.5)
- scores["urgent"] = max(scores.get("urgent", 0.0), 0.4)
- # 3. Schwelle anwenden und Ergebnisse erstellen
- results = []
- for tag, score in scores.items():
- if score >= self._threshold and tag in ALL_TONE_TAGS:
- category = TAG_TO_CATEGORY.get(tag, "unknown")
- results.append(ToneResult(tag=tag, category=category, score=score))
- # Sortieren nach Score (hoechster zuerst)
- results.sort(key=lambda r: r.score, reverse=True)
- return results
- # === Globale Instanz ===
- _analyzer: ToneAnalyzer | None = None
- def get_analyzer() -> ToneAnalyzer:
- """Gibt die globale ToneAnalyzer-Instanz zurueck."""
- global _analyzer
- if _analyzer is None:
- _analyzer = ToneAnalyzer()
- return _analyzer
- def analyze_tone(text: str) -> list[ToneResult]:
- """Shortcut: Analysiert Text mit der globalen Instanz."""
- return get_analyzer().analyze(text)
- def get_all_tone_tags() -> list[str]:
- """Gibt alle verfuegbaren Tone-Tags als Strings zurueck."""
- return sorted(ALL_TONE_TAGS)
- def get_tone_categories() -> dict[str, list[str]]:
- """Gibt alle Kategorien mit ihren Tags zurueck."""
- return dict(TONE_CATEGORIES)
- async def emit_tone_events(
- results: list[ToneResult],
- text: str,
- events: object,
- ) -> None:
- """
- Emittiert tone_detected Events fuer jeden erkannten Tag.
- Args:
- results: Ergebnis von analyze_tone()
- text: Eingabetext
- events: EventManager-Instanz
- """
- if not results or not events:
- return
- emit = getattr(events, "emit", None)
- if not emit:
- return
- for result in results:
- await emit("tone_detected", {
- "tag": result.tag,
- "category": result.category,
- "score": result.score,
- "text": text,
- })
|