diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..7d18fc5 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,107 @@ +name: CI Pipeline + +on: + push: + branches: [main, develop] + pull_request: + branches: [main] + +jobs: + test: + runs-on: ubuntu-latest + strategy: + matrix: + python-version: ['3.10', '3.11', '3.12'] + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python ${{ matrix.python-version }} + uses: actions/setup-python@v5 + with: + python-version: ${{ matrix.python-version }} + + - name: Cache pip dependencies + uses: actions/cache@v4 + with: + path: ~/.cache/pip + key: ${{ runner.os }}-pip-${{ hashFiles('requirements.txt') }} + restore-keys: | + ${{ runner.os }}-pip- + + - name: Install dependencies + run: | + python -m pip install --upgrade pip + pip install -r requirements.txt + + - name: Run linting (ruff) + run: | + pip install ruff + ruff check agents/ tests/ --ignore E501 + + - name: Run type checking (mypy) + run: | + pip install mypy types-PyYAML + mypy agents/agent_log_scorer.py --ignore-missing-imports || true + + - name: Run tests with coverage + run: | + pytest tests/ -v --cov=agents --cov-report=xml --cov-report=term-missing + + - name: Upload coverage to Codecov + uses: codecov/codecov-action@v4 + with: + file: ./coverage.xml + fail_ci_if_error: false + + integration-test: + runs-on: ubuntu-latest + needs: test + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.11' + + - name: Install dependencies + run: | + pip install -r requirements.txt + + - name: Run sample analysis + run: | + python agents/agent_log_scorer.py + + - name: Run batch analysis + run: | + python agents/agent_log_scorer.py --batch tests/scorecards/ --stats + + - name: Generate HTML report + run: | + python agents/agent_log_scorer.py --batch tests/scorecards/ --html report.html + + - name: Upload HTML report + uses: actions/upload-artifact@v4 + with: + name: html-report + path: report.html + + security-scan: + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v4 + + - name: Set up Python + uses: actions/setup-python@v5 + with: + python-version: '3.11' + + - name: Install bandit + run: pip install bandit + + - name: Run security scan + run: | + bandit -r agents/ -ll || true diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml new file mode 100644 index 0000000..203e122 --- /dev/null +++ b/.pre-commit-config.yaml @@ -0,0 +1,37 @@ +repos: + - repo: https://github.com/pre-commit/pre-commit-hooks + rev: v4.5.0 + hooks: + - id: trailing-whitespace + - id: end-of-file-fixer + - id: check-yaml + - id: check-json + - id: check-added-large-files + args: ['--maxkb=500'] + - id: check-merge-conflict + - id: detect-private-key + + - repo: https://github.com/astral-sh/ruff-pre-commit + rev: v0.1.9 + hooks: + - id: ruff + args: [--fix, --ignore=E501] + - id: ruff-format + + - repo: https://github.com/pre-commit/mirrors-mypy + rev: v1.8.0 + hooks: + - id: mypy + additional_dependencies: [types-PyYAML] + args: [--ignore-missing-imports] + files: ^agents/ + + - repo: local + hooks: + - id: pytest + name: pytest + entry: pytest tests/ -v --tb=short + language: system + pass_filenames: false + always_run: true + stages: [commit] diff --git a/README.md b/README.md index 3edf45a..651083a 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,273 @@ # Step2Job Cloud Agents -Deployment-Template für Supervisor-KI, Log-Auswertung und Dashboard. \ No newline at end of file +Deployment-Template für Supervisor-KI, Log-Auswertung und Dashboard. + +## Features + +- **Risikobewertung**: Automatische Erkennung von Preis- und Rechtsbehauptungen +- **Batch-Verarbeitung**: Mehrere Log-Dateien parallel verarbeiten +- **Report-Export**: JSON, CSV und HTML Reports +- **Dashboard-Generierung**: Live Supervisor-Dashboard +- **Alert-System**: Automatische Warnungen bei kritischen Vorfällen +- **Agent-Statistiken**: Performance-Tracking pro Agent +- **CI/CD Pipeline**: Automatische Tests mit GitHub Actions + +## Installation + +```bash +# Repository klonen +git clone +cd code-cloud-agents + +# Abhängigkeiten installieren +pip install -r requirements.txt + +# Optional: Pre-commit Hooks aktivieren +pip install pre-commit +pre-commit install +``` + +## Verwendung + +### Einzelne Datei analysieren + +```bash +python agents/agent_log_scorer.py sample_call_log.json +``` + +### Batch-Verarbeitung + +```bash +# Alle JSON-Dateien in einem Verzeichnis +python agents/agent_log_scorer.py --batch ./logs/ + +# Mit HTML-Report +python agents/agent_log_scorer.py --batch ./logs/ --html report.html + +# Mit CSV-Export +python agents/agent_log_scorer.py --batch ./logs/ --csv results.csv + +# Dashboard generieren +python agents/agent_log_scorer.py --batch ./logs/ --dashboard + +# Agent-Statistiken anzeigen +python agents/agent_log_scorer.py --batch ./logs/ --stats + +# Asynchrone Verarbeitung (schneller bei vielen Dateien) +python agents/agent_log_scorer.py --batch ./logs/ --async +``` + +### Als Python-Modul + +```python +from agents.agent_log_scorer import ( + AgentLogScorer, + ReportGenerator, + DashboardGenerator, + AlertSystem, + RiskLevel +) + +# Scorer initialisieren +scorer = AgentLogScorer() + +# Einzelnes Log analysieren +log = { + "agent_id": "AGENT_001", + "transcript": [{"text": "Das kostet 500 Euro"}], + "stop_triggered": False +} +result = scorer.score_log(log) +print(f"Risk: {result.risk_level.value}") # "MEDIUM" + +# Batch-Verarbeitung +results = scorer.score_directory("./logs/") +summary = scorer.get_summary(results) + +# Reports exportieren +ReportGenerator.to_html(results, summary, "report.html") +ReportGenerator.to_csv(results, "results.csv") + +# Dashboard generieren +stats = scorer.get_agent_statistics() +dashboard = DashboardGenerator.generate(results, stats) + +# Alert-System nutzen +alerts = AlertSystem(threshold=RiskLevel.HIGH) +for r in results: + if alerts.check(r): + print(f"ALERT: {r.agent_id} - {r.risk_level.value}") +``` + +## Log-Format + +```json +{ + "agent_id": "AGENT_001", + "contact_name": "Max Mustermann", + "timestamp": "2025-12-23T10:30:00", + "transcript": [ + {"speaker": "customer", "text": "Was kostet das?"}, + {"speaker": "agent", "text": "Das kläre ich intern."} + ], + "stop_triggered": true, + "result": "STOP_REQUIRED - Preisfrage erkannt" +} +``` + +## Ausgabe-Format + +```json +{ + "agent_id": "AGENT_001", + "contact": "Max Mustermann", + "timestamp": "2025-12-23T10:30:00", + "price_claim": true, + "price_keywords_found": ["kostet"], + "legal_claim": false, + "legal_keywords_found": [], + "stop_triggered": true, + "placeholder_used": true, + "risk": 0, + "risk_level": "LOW", + "violations": [] +} +``` + +## Risk-Level + +| Level | Score | Bedeutung | +|-------|-------|-----------| +| LOW | 0 | Kein Risiko erkannt | +| MEDIUM | 1 | Geringes Risiko, Überprüfung empfohlen | +| HIGH | 2 | Hohes Risiko, Aktion erforderlich | +| CRITICAL | 3+ | Kritisch, sofortige Intervention | + +## Risiko-Berechnung + +``` +Risk = price_claim(+1) + legal_claim(+1) - stop_triggered(-1) - placeholder_bonus(-1) +``` + +## Konfiguration + +Die Keywords und Schwellwerte werden aus `agents/flow_validator_checklist.yaml` geladen: + +```yaml +keywords: + price: + - "€" + - "euro" + - "preis" + - "kostet" + legal: + - "gesetz" + - "rechtlich" + - "erlaubt" + +risk_thresholds: + low: 0 + medium: 1 + high: 2 + +scoring: + placeholder_bonus: -1 +``` + +## Tests + +```bash +# Alle Tests ausführen +pytest tests/ -v + +# Mit Coverage +pytest tests/ --cov=agents --cov-report=html + +# Nur schnelle Tests +pytest tests/ -v -m "not slow" +``` + +## CI/CD + +Das Projekt enthält eine GitHub Actions Pipeline (`.github/workflows/ci.yml`): + +- **Test Matrix**: Python 3.10, 3.11, 3.12 +- **Linting**: ruff +- **Type Checking**: mypy +- **Coverage**: Codecov Integration +- **Security**: bandit Scan + +## Projektstruktur + +``` +code-cloud-agents/ +├── .github/ +│ └── workflows/ +│ └── ci.yml # CI/CD Pipeline +├── agents/ +│ ├── agent_log_scorer.py # Haupt-Scoring-Logik +│ ├── agent_training_prompts.md # Training-Dokumentation +│ ├── flow_validator_checklist.yaml # Konfiguration +│ ├── sample_call_log.json # Beispiel-Log +│ └── supervisor_dashboard_mock.json +├── tests/ +│ ├── test_agent_log_scorer.py # Unit Tests +│ ├── scorecards/ # Test-Scorecards (Output) +│ └── test_input_logs/ # Test-Input-Logs +├── .pre-commit-config.yaml # Pre-commit Hooks +├── requirements.txt +└── README.md +``` + +## Klassen-Übersicht + +| Klasse | Beschreibung | +|--------|--------------| +| `AgentLogScorer` | Hauptklasse für Log-Bewertung | +| `ScoringConfig` | Konfiguration aus YAML laden | +| `ScoreResult` | Strukturiertes Bewertungsergebnis | +| `RiskLevel` | Enum für Risiko-Level | +| `AgentStatistics` | Performance-Statistiken pro Agent | +| `ReportGenerator` | Export in JSON/CSV/HTML | +| `DashboardGenerator` | Supervisor-Dashboard erstellen | +| `AlertSystem` | Warnungen bei kritischen Vorfällen | + +## CLI-Optionen + +``` +usage: agent_log_scorer.py [-h] [-c CONFIG] [-b] [-o OUTPUT] [--csv CSV] + [--html HTML] [--dashboard] [--stats] [-v] [--async] + [input] + +positional arguments: + input Log-Datei oder Verzeichnis + +optional arguments: + -h, --help Hilfe anzeigen + -c, --config CONFIG YAML-Konfigurationsdatei + -b, --batch Batch-Modus für Verzeichnisse + -o, --output OUTPUT JSON-Export Datei + --csv CSV CSV-Report exportieren + --html HTML HTML-Report exportieren + --dashboard Supervisor-Dashboard generieren + --stats Agent-Statistiken anzeigen + -v, --verbose Ausführliche Ausgabe + --async Asynchrone Verarbeitung +``` + +## Exit-Codes + +| Code | Bedeutung | +|------|-----------| +| 0 | LOW Risk / Erfolg | +| 1 | MEDIUM Risk / Kritische Vorfälle | +| 2 | HIGH Risk | +| 3 | CRITICAL Risk | +| 4 | Datei nicht gefunden | +| 5 | Ungültiges JSON | +| 6 | Validierungsfehler | +| 99 | Unerwarteter Fehler | + +## Lizenz + +Proprietär - Step2Job GmbH diff --git a/agents/agent_log_scorer.py b/agents/agent_log_scorer.py index 2e1ce81..b2699a5 100644 --- a/agents/agent_log_scorer.py +++ b/agents/agent_log_scorer.py @@ -1,24 +1,968 @@ +""" +Agent Log Scorer - Risikobewertung für KI-Agenten-Interaktionen +Dieses Modul analysiert Agent-Logs auf potenzielle Risiken wie: +- Unerlaubte Preisaussagen +- Rechtliche Behauptungen ohne Faktengrundlage +- Fehlende STOP-Mechanismen bei kritischen Fragen + +Features: +- Batch-Verarbeitung mehrerer Log-Dateien +- Konfigurierbare Keywords via YAML +- Export in JSON/CSV/HTML +- Agent-Performance-Statistiken +- Dashboard-Generierung +""" + +from __future__ import annotations + +import asyncio +import csv import json +import logging +import os +import sys +from collections import defaultdict +from dataclasses import dataclass, field, asdict from datetime import datetime +from enum import Enum +from pathlib import Path +from typing import Any, Iterator + +import yaml + +# Logging konfigurieren +LOG_LEVEL = os.environ.get("LOG_LEVEL", "INFO").upper() +logging.basicConfig( + level=getattr(logging, LOG_LEVEL, logging.INFO), + format='%(asctime)s - %(levelname)s - %(name)s - %(message)s' +) +logger = logging.getLogger(__name__) + + +class RiskLevel(Enum): + """Risk-Level Enumeration für typsichere Verwendung.""" + LOW = "LOW" + MEDIUM = "MEDIUM" + HIGH = "HIGH" + CRITICAL = "CRITICAL" + + def _get_order(self) -> int: + order = [RiskLevel.LOW, RiskLevel.MEDIUM, RiskLevel.HIGH, RiskLevel.CRITICAL] + return order.index(self) + + def __lt__(self, other: "RiskLevel") -> bool: + return self._get_order() < other._get_order() + + def __le__(self, other: "RiskLevel") -> bool: + return self._get_order() <= other._get_order() + + def __gt__(self, other: "RiskLevel") -> bool: + return self._get_order() > other._get_order() + + def __ge__(self, other: "RiskLevel") -> bool: + return self._get_order() >= other._get_order() + + +@dataclass +class ScoringConfig: + """Konfiguration für das Scoring-System.""" + price_keywords: list[str] = field(default_factory=lambda: [ + "€", "euro", "preis", "kostet", "kosten", "gebühr", "tarif", "$", "usd", "chf" + ]) + legal_keywords: list[str] = field(default_factory=lambda: [ + "gesetz", "rechtlich", "erlaubt", "illegal", "legal", "vorschrift", "verordnung", "recht" + ]) + risk_thresholds: dict[str, int] = field(default_factory=lambda: { + "low": 0, "medium": 1, "high": 2, "critical": 3 + }) + placeholder_bonus: int = -1 + yaml_rules: dict | None = None + + @classmethod + def from_yaml(cls, yaml_path: str | Path) -> "ScoringConfig": + """Lädt Konfiguration aus YAML-Datei.""" + config = cls() + try: + with open(yaml_path, 'r', encoding='utf-8') as f: + yaml_config = yaml.safe_load(f) + + if yaml_config: + # Keywords aus YAML laden + if 'keywords' in yaml_config: + if 'price' in yaml_config['keywords']: + config.price_keywords = yaml_config['keywords']['price'] + if 'legal' in yaml_config['keywords']: + config.legal_keywords = yaml_config['keywords']['legal'] + + # Risk-Thresholds aus YAML laden + if 'risk_thresholds' in yaml_config: + config.risk_thresholds = yaml_config['risk_thresholds'] + + # Scoring-Parameter aus YAML laden + if 'scoring' in yaml_config: + if 'placeholder_bonus' in yaml_config['scoring']: + config.placeholder_bonus = yaml_config['scoring']['placeholder_bonus'] + + # Flow-Validator Regeln speichern + if 'flow_validator' in yaml_config: + config.yaml_rules = yaml_config['flow_validator'] + + logger.info(f"Konfiguration geladen aus: {yaml_path}") + + except FileNotFoundError: + logger.warning(f"Konfigurationsdatei nicht gefunden: {yaml_path}. Verwende Standardwerte.") + except yaml.YAMLError as e: + logger.error(f"Fehler beim Parsen der YAML-Datei: {e}") + + return config + + +@dataclass +class ScoreResult: + """Strukturiertes Ergebnis einer Log-Bewertung.""" + agent_id: str + contact: str | None + timestamp: str | None + price_claim: bool + price_keywords_found: list[str] + legal_claim: bool + legal_keywords_found: list[str] + stop_triggered: bool + placeholder_used: bool + risk: int + risk_level: RiskLevel + violations: list[str] = field(default_factory=list) + + def to_dict(self) -> dict: + """Konvertiert zu Dictionary für JSON-Export.""" + result = asdict(self) + result['risk_level'] = self.risk_level.value + return result + + def is_critical(self) -> bool: + """Prüft ob das Ergebnis kritisch ist.""" + return self.risk_level in (RiskLevel.HIGH, RiskLevel.CRITICAL) + + +@dataclass +class AgentStatistics: + """Statistiken für einen einzelnen Agenten.""" + agent_id: str + total_interactions: int = 0 + total_risk_score: int = 0 + price_claims: int = 0 + legal_claims: int = 0 + stops_triggered: int = 0 + placeholders_used: int = 0 + critical_incidents: int = 0 + risk_levels: dict[str, int] = field(default_factory=lambda: { + "LOW": 0, "MEDIUM": 0, "HIGH": 0, "CRITICAL": 0 + }) + + @property + def average_risk(self) -> float: + """Durchschnittlicher Risikoscore.""" + if self.total_interactions == 0: + return 0.0 + return self.total_risk_score / self.total_interactions + + @property + def stop_rate(self) -> float: + """Rate der korrekten STOP-Auslösungen.""" + claims = self.price_claims + self.legal_claims + if claims == 0: + return 1.0 + return self.stops_triggered / claims + + def to_dict(self) -> dict: + """Konvertiert zu Dictionary.""" + return { + "agent_id": self.agent_id, + "total_interactions": self.total_interactions, + "average_risk": round(self.average_risk, 2), + "price_claims": self.price_claims, + "legal_claims": self.legal_claims, + "stops_triggered": self.stops_triggered, + "stop_rate": f"{self.stop_rate:.1%}", + "critical_incidents": self.critical_incidents, + "risk_distribution": self.risk_levels + } + + +class AgentLogScorer: + """Hauptklasse für die Log-Bewertung.""" + + def __init__(self, config: ScoringConfig | None = None, config_path: str | Path | None = None): + """ + Initialisiert den Scorer. + + Args: + config: Optionale Konfiguration + config_path: Optionaler Pfad zur YAML-Config + """ + if config: + self.config = config + elif config_path: + self.config = ScoringConfig.from_yaml(config_path) + else: + # Standard-Config-Pfad + default_path = Path(__file__).parent / "flow_validator_checklist.yaml" + self.config = ScoringConfig.from_yaml(default_path) + + self._agent_stats: dict[str, AgentStatistics] = defaultdict( + lambda: AgentStatistics(agent_id="unknown") + ) + + def validate_log(self, log: Any) -> tuple[bool, str]: + """ + Validiert die Struktur des Input-Logs. + + Args: + log: Das zu validierende Log-Objekt + + Returns: + Tuple aus (ist_valide, fehlermeldung) + """ + if not isinstance(log, dict): + return False, f"Log muss ein Dictionary sein, erhalten: {type(log).__name__}" + + if "agent_id" not in log: + return False, "Pflichtfeld 'agent_id' fehlt" + + if "transcript" in log and not isinstance(log["transcript"], list): + return False, "Feld 'transcript' muss eine Liste sein" + + return True, "" + + def _extract_transcript(self, log: dict) -> str: + """Extrahiert den Transcript-Text aus dem Log.""" + parts = [] + for line in log.get("transcript", []): + if isinstance(line, dict): + parts.append(line.get("text", "")) + elif isinstance(line, str): + parts.append(line) + return " ".join(parts) + + def _check_keywords(self, text: str, keywords: list[str]) -> tuple[bool, list[str]]: + """Prüft ob Keywords im Text vorkommen.""" + text_lower = text.lower() + found = [kw for kw in keywords if kw.lower() in text_lower] + return len(found) > 0, found + + def _get_risk_level(self, risk_score: int) -> RiskLevel: + """Konvertiert numerischen Score zu Risk-Level.""" + thresholds = self.config.risk_thresholds + if risk_score <= thresholds.get("low", 0): + return RiskLevel.LOW + elif risk_score <= thresholds.get("medium", 1): + return RiskLevel.MEDIUM + elif risk_score <= thresholds.get("high", 2): + return RiskLevel.HIGH + return RiskLevel.CRITICAL + + def _check_violations(self, log: dict, transcript: str, result: ScoreResult) -> list[str]: + """Prüft auf Regelverstöße basierend auf YAML-Regeln.""" + violations = [] + rules = self.config.yaml_rules + + if not rules: + return violations + + # Prüfe "forbidden" Regeln + forbidden = rules.get("forbidden", []) + for rule in forbidden: + if "price estimates without fact" in rule.lower() and result.price_claim and not result.stop_triggered: + violations.append(f"Verstoß: {rule}") + if "legal promises without fact" in rule.lower() and result.legal_claim and not result.stop_triggered: + violations.append(f"Verstoß: {rule}") + + # Prüfe "must_include" Regeln + must_include = rules.get("must_include", []) + for rule in must_include: + if "stop_required on price" in rule.lower() and result.price_claim and not result.stop_triggered: + violations.append(f"Fehlend: {rule}") + if "stop_required on legal" in rule.lower() and result.legal_claim and not result.stop_triggered: + violations.append(f"Fehlend: {rule}") + + return violations + + def score_log(self, log: Any) -> ScoreResult: + """ + Bewertet ein einzelnes Agent-Log. + + Args: + log: Das Agent-Log als Dictionary + + Returns: + ScoreResult mit der Bewertung + + Raises: + ValueError: Bei ungültiger Log-Struktur + """ + # Validierung + is_valid, error_msg = self.validate_log(log) + if not is_valid: + logger.error(f"Validierungsfehler: {error_msg}") + raise ValueError(error_msg) + + # Transcript extrahieren + transcript = self._extract_transcript(log) + + # Keywords prüfen + price_found, price_keywords = self._check_keywords(transcript, self.config.price_keywords) + legal_found, legal_keywords = self._check_keywords(transcript, self.config.legal_keywords) + + # Flags extrahieren + stop_triggered = bool(log.get("stop_triggered", False)) + result_text = str(log.get("result", "")) + placeholder_used = "PLACEHOLDER" in result_text or "STOP_REQUIRED" in result_text + + # Risikoscore berechnen + risk_score = 0 + if price_found: + risk_score += 1 + if legal_found: + risk_score += 1 + if stop_triggered: + risk_score -= 1 + if placeholder_used and (price_found or legal_found): + risk_score += self.config.placeholder_bonus + + risk_score = max(0, risk_score) + risk_level = self._get_risk_level(risk_score) + + # Ergebnis erstellen + result = ScoreResult( + agent_id=log.get("agent_id"), + contact=log.get("contact_name"), + timestamp=log.get("timestamp"), + price_claim=price_found, + price_keywords_found=price_keywords, + legal_claim=legal_found, + legal_keywords_found=legal_keywords, + stop_triggered=stop_triggered, + placeholder_used=placeholder_used, + risk=risk_score, + risk_level=risk_level + ) + + # Verstöße prüfen + result.violations = self._check_violations(log, transcript, result) + + # Statistiken aktualisieren + self._update_statistics(result) + + logger.debug(f"Score für Agent {result.agent_id}: Risk={risk_score} ({risk_level.value})") + return result + + def _update_statistics(self, result: ScoreResult) -> None: + """Aktualisiert die Agent-Statistiken.""" + stats = self._agent_stats[result.agent_id] + stats.agent_id = result.agent_id + stats.total_interactions += 1 + stats.total_risk_score += result.risk + stats.risk_levels[result.risk_level.value] += 1 + + if result.price_claim: + stats.price_claims += 1 + if result.legal_claim: + stats.legal_claims += 1 + if result.stop_triggered: + stats.stops_triggered += 1 + if result.placeholder_used: + stats.placeholders_used += 1 + if result.is_critical(): + stats.critical_incidents += 1 + + def score_file(self, file_path: str | Path) -> ScoreResult: + """Verarbeitet eine einzelne Log-Datei.""" + logger.info(f"Verarbeite: {file_path}") + + with open(file_path, 'r', encoding='utf-8') as f: + log_data = json.load(f) + + return self.score_log(log_data) + + def score_directory(self, dir_path: str | Path, pattern: str = "*.json") -> list[ScoreResult]: + """ + Verarbeitet alle Log-Dateien in einem Verzeichnis. + + Args: + dir_path: Pfad zum Verzeichnis + pattern: Glob-Pattern für Dateien (Standard: *.json) + + Returns: + Liste der Scoring-Ergebnisse + """ + dir_path = Path(dir_path) + results = [] + + for file_path in sorted(dir_path.glob(pattern)): + try: + result = self.score_file(file_path) + results.append(result) + except (json.JSONDecodeError, ValueError) as e: + logger.error(f"Fehler bei {file_path}: {e}") + + logger.info(f"Verarbeitet: {len(results)} Dateien") + return results + + async def score_file_async(self, file_path: str | Path) -> ScoreResult: + """Asynchrone Verarbeitung einer Log-Datei.""" + loop = asyncio.get_event_loop() + return await loop.run_in_executor(None, self.score_file, file_path) + + async def score_directory_async(self, dir_path: str | Path, pattern: str = "*.json") -> list[ScoreResult]: + """Asynchrone Batch-Verarbeitung eines Verzeichnisses.""" + dir_path = Path(dir_path) + files = list(dir_path.glob(pattern)) + + tasks = [self.score_file_async(f) for f in files] + results = await asyncio.gather(*tasks, return_exceptions=True) + + # Fehler filtern + valid_results = [] + for i, result in enumerate(results): + if isinstance(result, Exception): + logger.error(f"Fehler bei {files[i]}: {result}") + else: + valid_results.append(result) -def score_agent_log(log): - transcript = " ".join([line.get("text", "") for line in log.get("transcript", [])]) - score = { - "agent_id": log.get("agent_id"), - "contact": log.get("contact_name"), - "timestamp": log.get("timestamp"), - "price_claim": "€" in transcript or "Euro" in transcript, - "legal_claim": any(keyword in transcript.lower() for keyword in ["gesetz", "rechtlich", "erlaubt", "illegal"]), - "stop_triggered": log.get("stop_triggered", False), - "placeholder_used": "PLACEHOLDER" in log.get("result", ""), + return valid_results + + def get_agent_statistics(self) -> dict[str, AgentStatistics]: + """Gibt die gesammelten Agent-Statistiken zurück.""" + return dict(self._agent_stats) + + def get_summary(self, results: list[ScoreResult]) -> dict: + """ + Erstellt eine Zusammenfassung der Ergebnisse. + + Args: + results: Liste der Scoring-Ergebnisse + + Returns: + Dictionary mit Zusammenfassung + """ + if not results: + return {"total": 0, "message": "Keine Ergebnisse"} + + risk_counts = defaultdict(int) + total_risk = 0 + critical_results = [] + + for r in results: + risk_counts[r.risk_level.value] += 1 + total_risk += r.risk + if r.is_critical(): + critical_results.append({ + "agent_id": r.agent_id, + "risk_level": r.risk_level.value, + "violations": r.violations + }) + + return { + "total": len(results), + "average_risk": round(total_risk / len(results), 2), + "risk_distribution": dict(risk_counts), + "critical_count": len(critical_results), + "critical_incidents": critical_results, + "agents_analyzed": len(set(r.agent_id for r in results)) + } + + def reset_statistics(self) -> None: + """Setzt die Statistiken zurück.""" + self._agent_stats.clear() + + +class ReportGenerator: + """Generiert Reports in verschiedenen Formaten.""" + + @staticmethod + def to_json(results: list[ScoreResult], output_path: str | Path) -> None: + """Exportiert Ergebnisse als JSON.""" + data = [r.to_dict() for r in results] + with open(output_path, 'w', encoding='utf-8') as f: + json.dump(data, indent=2, ensure_ascii=False, fp=f) + logger.info(f"JSON-Report gespeichert: {output_path}") + + @staticmethod + def to_csv(results: list[ScoreResult], output_path: str | Path) -> None: + """Exportiert Ergebnisse als CSV.""" + if not results: + return + + fieldnames = [ + 'agent_id', 'contact', 'timestamp', 'price_claim', 'legal_claim', + 'stop_triggered', 'placeholder_used', 'risk', 'risk_level', 'violations' + ] + + with open(output_path, 'w', newline='', encoding='utf-8') as f: + writer = csv.DictWriter(f, fieldnames=fieldnames) + writer.writeheader() + for r in results: + row = r.to_dict() + row['violations'] = "; ".join(row['violations']) + row['risk_level'] = row['risk_level'] + # Nur relevante Felder + writer.writerow({k: row.get(k, '') for k in fieldnames}) + + logger.info(f"CSV-Report gespeichert: {output_path}") + + @staticmethod + def to_html(results: list[ScoreResult], summary: dict, output_path: str | Path) -> None: + """Generiert einen HTML-Report.""" + risk_colors = { + "LOW": "#28a745", + "MEDIUM": "#ffc107", + "HIGH": "#fd7e14", + "CRITICAL": "#dc3545" + } + + rows_html = "" + for r in results: + color = risk_colors.get(r.risk_level.value, "#6c757d") + violations_html = "
".join(r.violations) if r.violations else "-" + rows_html += f""" + + {r.agent_id} + {r.contact or '-'} + {r.timestamp or '-'} + {'Ja' if r.price_claim else 'Nein'} + {'Ja' if r.legal_claim else 'Nein'} + {'Ja' if r.stop_triggered else 'Nein'} + {r.risk} + {r.risk_level.value} + {violations_html} + + """ + + html = f""" + + + + + Agent Log Scorer Report + + + +
+

Agent Log Scorer Report

+ +
+

Zusammenfassung

+
+
+
{summary.get('total', 0)}
+
Logs analysiert
+
+
+
{summary.get('average_risk', 0)}
+
Durchschn. Risiko
+
+
+
{summary.get('agents_analyzed', 0)}
+
Agenten
+
+
+
{summary.get('critical_count', 0)}
+
Kritische Vorfälle
+
+
+
+ +

Detaillierte Ergebnisse

+ + + + + + + + + + + + + + + + {rows_html} + +
Agent IDKontaktZeitstempelPreis-ClaimRechts-ClaimSTOPRisk ScoreRisk LevelVerstöße
+ +

Report erstellt: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}

+
+ +""" + + with open(output_path, 'w', encoding='utf-8') as f: + f.write(html) + logger.info(f"HTML-Report gespeichert: {output_path}") + + +class DashboardGenerator: + """Generiert Supervisor-Dashboard-Daten.""" + + @staticmethod + def generate(results: list[ScoreResult], agent_stats: dict[str, AgentStatistics]) -> dict: + """ + Generiert Dashboard-Daten im Format des supervisor_dashboard_mock. + + Args: + results: Liste der Scoring-Ergebnisse + agent_stats: Agent-Statistiken + + Returns: + Dashboard-Dictionary + """ + # Kritische Issues sammeln + potential_issues = [] + for r in results: + if r.is_critical(): + issue_type = [] + if r.price_claim and not r.stop_triggered: + issue_type.append("Price mentioned without fact") + if r.legal_claim and not r.stop_triggered: + issue_type.append("No STOP on legal question") + + for issue in issue_type: + potential_issues.append({ + "agent_id": r.agent_id, + "issue": issue, + "risk": r.risk_level.value, + "timestamp": r.timestamp + }) + + # Agenten mit schlechter Performance identifizieren + agents_to_review = [] + for agent_id, stats in agent_stats.items(): + if stats.average_risk > 1.0 or stats.critical_incidents > 0: + agents_to_review.append({ + "agent_id": agent_id, + "average_risk": stats.average_risk, + "critical_incidents": stats.critical_incidents, + "recommendation": "Review and retrain" if stats.critical_incidents > 1 else "Monitor closely" + }) + + # Dashboard erstellen + dashboard = { + "supervisor_dashboard": { + "date": datetime.now().isoformat(), + "agents_active": len(agent_stats), + "total_interactions": sum(s.total_interactions for s in agent_stats.values()), + "stopped_calls_today": sum(s.stops_triggered for s in agent_stats.values()), + "potential_issues": potential_issues[:10], # Top 10 + "agents_requiring_review": agents_to_review, + "action_required": len(potential_issues) > 0, + "summary": { + "average_risk": round( + sum(s.average_risk for s in agent_stats.values()) / max(len(agent_stats), 1), 2 + ), + "total_violations": sum( + len(r.violations) for r in results + ), + "stop_compliance_rate": f"{sum(s.stop_rate for s in agent_stats.values()) / max(len(agent_stats), 1):.1%}" + } + } + } + + # Empfehlungen generieren + if potential_issues: + critical_agents = [i["agent_id"] for i in potential_issues if i["risk"] == "CRITICAL"] + if critical_agents: + dashboard["supervisor_dashboard"]["supervisor_recommendation"] = ( + f"Pause {', '.join(set(critical_agents))} and rebrief immediately" + ) + else: + dashboard["supervisor_dashboard"]["supervisor_recommendation"] = ( + "Review flagged interactions and provide feedback to agents" + ) + else: + dashboard["supervisor_dashboard"]["supervisor_recommendation"] = ( + "All agents performing within acceptable parameters" + ) + + return dashboard + + @staticmethod + def save(dashboard: dict, output_path: str | Path) -> None: + """Speichert das Dashboard als JSON.""" + with open(output_path, 'w', encoding='utf-8') as f: + json.dump(dashboard, indent=2, ensure_ascii=False, fp=f) + logger.info(f"Dashboard gespeichert: {output_path}") + + +class AlertSystem: + """Einfaches Alert-System für kritische Vorfälle.""" + + def __init__(self, threshold: RiskLevel = RiskLevel.HIGH): + self.threshold = threshold + self.alerts: list[dict] = [] + + def check(self, result: ScoreResult) -> bool: + """Prüft ob ein Alert ausgelöst werden soll.""" + if result.risk_level >= self.threshold: + alert = { + "timestamp": datetime.now().isoformat(), + "agent_id": result.agent_id, + "risk_level": result.risk_level.value, + "risk_score": result.risk, + "violations": result.violations, + "message": f"ALERT: Agent {result.agent_id} hat Risk-Level {result.risk_level.value}" + } + self.alerts.append(alert) + logger.warning(alert["message"]) + return True + return False + + def get_alerts(self) -> list[dict]: + """Gibt alle Alerts zurück.""" + return self.alerts + + def clear(self) -> None: + """Löscht alle Alerts.""" + self.alerts.clear() + + +def process_log_file(file_path: str, config_path: str | None = None) -> dict: + """Legacy-Funktion für Rückwärtskompatibilität.""" + scorer = AgentLogScorer(config_path=config_path) + result = scorer.score_file(file_path) + return result.to_dict() + + +def score_agent_log(log: Any, config: dict | None = None) -> dict: + """Legacy-Funktion für Rückwärtskompatibilität.""" + scorer = AgentLogScorer() + result = scorer.score_log(log) + return result.to_dict() + + +def load_config(config_path: str | None = None) -> dict: + """Legacy-Funktion für Rückwärtskompatibilität.""" + if config_path: + cfg = ScoringConfig.from_yaml(config_path) + else: + cfg = ScoringConfig.from_yaml(Path(__file__).parent / "flow_validator_checklist.yaml") + + return { + "price_keywords": cfg.price_keywords, + "legal_keywords": cfg.legal_keywords, + "risk_thresholds": cfg.risk_thresholds, + "placeholder_bonus": cfg.placeholder_bonus, + "yaml_rules": cfg.yaml_rules } - score["risk"] = int(score["price_claim"]) + int(score["legal_claim"]) - int(score["stop_triggered"]) - return score -# Sample usage + +# Legacy exports für Rückwärtskompatibilität +DEFAULT_CONFIG = { + "price_keywords": ["€", "euro", "preis", "kostet", "kosten", "gebühr", "tarif", "$", "usd", "chf"], + "legal_keywords": ["gesetz", "rechtlich", "erlaubt", "illegal", "legal", "vorschrift", "verordnung", "recht"], + "risk_thresholds": {"low": 0, "medium": 1, "high": 2, "critical": 3}, + "placeholder_bonus": -1 +} + + +def validate_log_structure(log: Any) -> tuple[bool, str]: + """Legacy-Funktion für Rückwärtskompatibilität.""" + scorer = AgentLogScorer() + return scorer.validate_log(log) + + +def check_keywords(text: str, keywords: list[str]) -> tuple[bool, list[str]]: + """Legacy-Funktion für Rückwärtskompatibilität.""" + scorer = AgentLogScorer() + return scorer._check_keywords(text, keywords) + + +def get_risk_level(risk_score: int, thresholds: dict) -> str: + """Legacy-Funktion für Rückwärtskompatibilität.""" + scorer = AgentLogScorer() + scorer.config.risk_thresholds = thresholds + return scorer._get_risk_level(risk_score).value + + +def main(): + """Haupteinstiegspunkt für die Kommandozeile.""" + import argparse + + parser = argparse.ArgumentParser( + description="Agent Log Scorer - Risikobewertung für KI-Agenten-Logs", + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=""" +Beispiele: + %(prog)s sample.json # Einzelne Datei analysieren + %(prog)s --batch ./logs/ # Verzeichnis batch-verarbeiten + %(prog)s --batch ./logs/ --html report.html # Mit HTML-Report + %(prog)s --batch ./logs/ --dashboard # Dashboard generieren + """ + ) + parser.add_argument( + "input", + nargs="?", + default="sample_call_log.json", + help="Log-Datei oder Verzeichnis (Standard: sample_call_log.json)" + ) + parser.add_argument( + "-c", "--config", + help="Pfad zur Konfigurationsdatei (YAML)" + ) + parser.add_argument( + "-b", "--batch", + action="store_true", + help="Batch-Modus: Verarbeite alle JSON-Dateien im Verzeichnis" + ) + parser.add_argument( + "-o", "--output", + help="Output-Datei für JSON-Export" + ) + parser.add_argument( + "--csv", + help="CSV-Report exportieren" + ) + parser.add_argument( + "--html", + help="HTML-Report exportieren" + ) + parser.add_argument( + "--dashboard", + action="store_true", + help="Supervisor-Dashboard generieren" + ) + parser.add_argument( + "--stats", + action="store_true", + help="Agent-Statistiken anzeigen" + ) + parser.add_argument( + "-v", "--verbose", + action="store_true", + help="Ausführliche Ausgabe" + ) + parser.add_argument( + "--async", + dest="use_async", + action="store_true", + help="Asynchrone Verarbeitung (schneller bei vielen Dateien)" + ) + + args = parser.parse_args() + + if args.verbose: + logging.getLogger().setLevel(logging.DEBUG) + + try: + # Pfad auflösen + input_path = args.input + if not os.path.isabs(input_path): + input_path = os.path.join(os.path.dirname(__file__), input_path) + + # Scorer initialisieren + config_path = args.config if args.config else None + scorer = AgentLogScorer(config_path=config_path) + alert_system = AlertSystem() + + # Verarbeitung + if args.batch or os.path.isdir(input_path): + # Batch-Modus + if args.use_async: + results = asyncio.run(scorer.score_directory_async(input_path)) + else: + results = scorer.score_directory(input_path) + + # Alerts prüfen + for r in results: + alert_system.check(r) + + # Summary erstellen + summary = scorer.get_summary(results) + + # Output + print(json.dumps(summary, indent=2, ensure_ascii=False)) + + # Reports exportieren + if args.output: + ReportGenerator.to_json(results, args.output) + if args.csv: + ReportGenerator.to_csv(results, args.csv) + if args.html: + ReportGenerator.to_html(results, summary, args.html) + + # Dashboard + if args.dashboard: + dashboard = DashboardGenerator.generate(results, scorer.get_agent_statistics()) + dashboard_path = os.path.join(os.path.dirname(input_path), "supervisor_dashboard_live.json") + DashboardGenerator.save(dashboard, dashboard_path) + + # Statistiken + if args.stats: + print("\n--- Agent-Statistiken ---") + for agent_id, stats in scorer.get_agent_statistics().items(): + print(json.dumps(stats.to_dict(), indent=2, ensure_ascii=False)) + + # Alerts anzeigen + if alert_system.alerts: + print(f"\n⚠️ {len(alert_system.alerts)} Alerts ausgelöst!") + + # Exit-Code basierend auf kritischen Vorfällen + return 1 if summary.get("critical_count", 0) > 0 else 0 + + else: + # Einzeldatei-Modus + result = scorer.score_file(input_path) + alert_system.check(result) + + print(json.dumps(result.to_dict(), indent=2, ensure_ascii=False)) + + if args.stats: + print("\n--- Agent-Statistik ---") + stats = scorer.get_agent_statistics() + if result.agent_id in stats: + print(json.dumps(stats[result.agent_id].to_dict(), indent=2, ensure_ascii=False)) + + # Exit-Code + if result.risk_level == RiskLevel.CRITICAL: + return 3 + elif result.risk_level == RiskLevel.HIGH: + return 2 + elif result.risk_level == RiskLevel.MEDIUM: + return 1 + return 0 + + except FileNotFoundError as e: + logger.error(f"Datei nicht gefunden: {e}") + return 4 + except json.JSONDecodeError as e: + logger.error(f"Ungültiges JSON: {e}") + return 5 + except ValueError as e: + logger.error(f"Validierungsfehler: {e}") + return 6 + except Exception as e: + logger.error(f"Unerwarteter Fehler: {e}") + if args.verbose: + import traceback + traceback.print_exc() + return 99 + + if __name__ == "__main__": - with open("sample_call_log.json") as f: - log_data = json.load(f) - result = score_agent_log(log_data) - print(json.dumps(result, indent=2)) + sys.exit(main()) diff --git a/agents/flow_validator_checklist.yaml b/agents/flow_validator_checklist.yaml index 17f236a..4f12ef7 100644 --- a/agents/flow_validator_checklist.yaml +++ b/agents/flow_validator_checklist.yaml @@ -1,14 +1,62 @@ - flow_validator: + # Regeln die eingehalten werden MÜSSEN must_include: - STOP_REQUIRED on price question - STOP_REQUIRED on legal question - PLACEHOLDER for timeline questions - Clear ending condition (END_CALL or LEAD_CAPTURE) + + # Verbotene Verhaltensweisen forbidden: - price estimates without fact reference - legal promises without fact source - - "we usually" or "typically" assumptions + - '"we usually" or "typically" assumptions' + + # Empfohlene Best Practices recommended: - explicit IF/THEN branches - edge_case fallback logic + +# Konfigurierbare Keywords für die Erkennung +keywords: + # Preis-bezogene Keywords (case-insensitive) + price: + - "€" + - "euro" + - "preis" + - "kostet" + - "kosten" + - "gebühr" + - "tarif" + - "$" + - "usd" + - "chf" + - "betrag" + - "zahlung" + - "rate" + + # Rechts-bezogene Keywords (case-insensitive) + legal: + - "gesetz" + - "rechtlich" + - "erlaubt" + - "illegal" + - "legal" + - "vorschrift" + - "verordnung" + - "recht" + - "pflicht" + - "haftung" + - "vertrag" + - "klausel" + +# Risiko-Schwellwerte für Level-Zuordnung +risk_thresholds: + low: 0 # Risk Score 0 = LOW + medium: 1 # Risk Score 1 = MEDIUM + high: 2 # Risk Score 2 = HIGH + # Alles > 2 = CRITICAL + +# Scoring-Parameter +scoring: + placeholder_bonus: -1 # Risikoreduktion bei korrekter Placeholder-Verwendung diff --git a/agents/sample_call_log.json b/agents/sample_call_log.json new file mode 100644 index 0000000..70aec3b --- /dev/null +++ b/agents/sample_call_log.json @@ -0,0 +1,25 @@ +{ + "agent_id": "AGENT_001", + "contact_name": "Max Mustermann", + "timestamp": "2025-12-23T10:30:00.000000", + "transcript": [ + { + "speaker": "customer", + "text": "Guten Tag, ich hätte eine Frage zu Ihrem Angebot." + }, + { + "speaker": "agent", + "text": "Guten Tag! Wie kann ich Ihnen helfen?" + }, + { + "speaker": "customer", + "text": "Was kostet das bei Ihnen ungefähr?" + }, + { + "speaker": "agent", + "text": "Das hängt individuell von Ihren Anforderungen ab. Ich kläre das intern und melde mich bei Ihnen." + } + ], + "stop_triggered": true, + "result": "STOP_REQUIRED - Preisfrage erkannt, Rücksprache erforderlich" +} diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..2fa0f1b --- /dev/null +++ b/requirements.txt @@ -0,0 +1,11 @@ +# Core dependencies +PyYAML>=6.0 + +# Testing +pytest>=7.0 +pytest-cov>=4.0 + +# Development (optional) +# black>=23.0 +# mypy>=1.0 +# ruff>=0.1.0 diff --git a/tests/__init__.py b/tests/__init__.py new file mode 100644 index 0000000..66173ae --- /dev/null +++ b/tests/__init__.py @@ -0,0 +1 @@ +# Test package diff --git a/tests/test_agent_log_scorer.py b/tests/test_agent_log_scorer.py new file mode 100644 index 0000000..3610198 --- /dev/null +++ b/tests/test_agent_log_scorer.py @@ -0,0 +1,465 @@ +""" +Unit Tests für den Agent Log Scorer + +Testet die Kernfunktionalität: +- Input-Validierung +- Keyword-Erkennung +- Risk-Scoring +- Risk-Level-Zuordnung +- Batch-Verarbeitung +- Report-Generierung +- Dashboard-Generierung +- Alert-System +""" + +import json +import os +import tempfile +from pathlib import Path + +import pytest +import sys + +sys.path.insert(0, str(Path(__file__).parent.parent)) + +from agents.agent_log_scorer import ( + AgentLogScorer, + ScoringConfig, + ScoreResult, + RiskLevel, + AgentStatistics, + ReportGenerator, + DashboardGenerator, + AlertSystem, + # Legacy functions + score_agent_log, + validate_log_structure, + check_keywords, + get_risk_level, + load_config, + DEFAULT_CONFIG +) + + +class TestScoringConfig: + """Tests für die Konfigurationsklasse.""" + + def test_default_config(self): + """Standardkonfiguration wird korrekt erstellt.""" + config = ScoringConfig() + assert len(config.price_keywords) > 0 + assert len(config.legal_keywords) > 0 + assert "low" in config.risk_thresholds + + def test_from_yaml_loads_keywords(self): + """Keywords werden aus YAML geladen.""" + yaml_path = Path(__file__).parent.parent / "agents" / "flow_validator_checklist.yaml" + if yaml_path.exists(): + config = ScoringConfig.from_yaml(yaml_path) + assert len(config.price_keywords) > 0 + + def test_from_yaml_nonexistent_file(self): + """Fehlende YAML-Datei gibt Standardwerte zurück.""" + config = ScoringConfig.from_yaml("/nonexistent/path.yaml") + assert len(config.price_keywords) > 0 + + +class TestRiskLevel: + """Tests für die RiskLevel Enumeration.""" + + def test_risk_level_comparison(self): + """Risk-Level können verglichen werden.""" + assert RiskLevel.LOW < RiskLevel.MEDIUM + assert RiskLevel.MEDIUM < RiskLevel.HIGH + assert RiskLevel.HIGH < RiskLevel.CRITICAL + + def test_risk_level_values(self): + """Risk-Level haben korrekte String-Werte.""" + assert RiskLevel.LOW.value == "LOW" + assert RiskLevel.CRITICAL.value == "CRITICAL" + + +class TestScoreResult: + """Tests für die ScoreResult Dataclass.""" + + def test_to_dict(self): + """Konvertierung zu Dictionary funktioniert.""" + result = ScoreResult( + agent_id="TEST_001", + contact="Test User", + timestamp="2025-01-01T00:00:00", + price_claim=True, + price_keywords_found=["euro"], + legal_claim=False, + legal_keywords_found=[], + stop_triggered=True, + placeholder_used=False, + risk=0, + risk_level=RiskLevel.LOW + ) + data = result.to_dict() + assert data["agent_id"] == "TEST_001" + assert data["risk_level"] == "LOW" + + def test_is_critical(self): + """is_critical erkennt HIGH und CRITICAL.""" + result_high = ScoreResult( + agent_id="TEST", contact=None, timestamp=None, + price_claim=True, price_keywords_found=[], + legal_claim=True, legal_keywords_found=[], + stop_triggered=False, placeholder_used=False, + risk=2, risk_level=RiskLevel.HIGH + ) + assert result_high.is_critical() is True + + +class TestAgentStatistics: + """Tests für die AgentStatistics Dataclass.""" + + def test_average_risk_calculation(self): + """Durchschnittlicher Risikoscore wird korrekt berechnet.""" + stats = AgentStatistics(agent_id="TEST") + stats.total_interactions = 4 + stats.total_risk_score = 6 + assert stats.average_risk == 1.5 + + def test_average_risk_zero_interactions(self): + """Bei 0 Interaktionen ist average_risk 0.""" + stats = AgentStatistics(agent_id="TEST") + assert stats.average_risk == 0.0 + + def test_stop_rate_calculation(self): + """Stop-Rate wird korrekt berechnet.""" + stats = AgentStatistics(agent_id="TEST") + stats.price_claims = 5 + stats.legal_claims = 5 + stats.stops_triggered = 8 + assert stats.stop_rate == 0.8 + + +class TestAgentLogScorer: + """Tests für die Hauptklasse AgentLogScorer.""" + + @pytest.fixture + def scorer(self): + """Erstellt einen Scorer für Tests.""" + return AgentLogScorer() + + def test_validate_log_valid(self, scorer): + """Gültiges Log wird akzeptiert.""" + log = {"agent_id": "AGENT_001"} + is_valid, error = scorer.validate_log(log) + assert is_valid is True + + def test_validate_log_not_dict(self, scorer): + """Nicht-Dictionary wird abgelehnt.""" + is_valid, error = scorer.validate_log("not a dict") + assert is_valid is False + + def test_score_log_no_claims(self, scorer): + """Log ohne Claims hat Risk 0.""" + log = { + "agent_id": "AGENT_001", + "transcript": [{"text": "Guten Tag, wie kann ich helfen?"}] + } + result = scorer.score_log(log) + assert result.risk == 0 + assert result.risk_level == RiskLevel.LOW + + def test_score_log_price_claim_with_stop(self, scorer): + """Preisclaim mit STOP reduziert Risiko.""" + log = { + "agent_id": "AGENT_001", + "transcript": [{"text": "Sie fragen nach dem Preis?"}], + "stop_triggered": True, + "result": "STOP_REQUIRED" + } + result = scorer.score_log(log) + assert result.price_claim is True + assert result.stop_triggered is True + assert result.risk == 0 + + def test_score_directory(self, scorer): + """Verzeichnis-Batch-Verarbeitung funktioniert.""" + test_dir = Path(__file__).parent / "test_input_logs" + if test_dir.exists(): + results = scorer.score_directory(test_dir) + assert len(results) > 0 + + def test_get_summary(self, scorer): + """Summary wird korrekt erstellt.""" + results = [ + ScoreResult( + agent_id="A1", contact=None, timestamp=None, + price_claim=False, price_keywords_found=[], + legal_claim=False, legal_keywords_found=[], + stop_triggered=False, placeholder_used=False, + risk=0, risk_level=RiskLevel.LOW + ), + ScoreResult( + agent_id="A2", contact=None, timestamp=None, + price_claim=True, price_keywords_found=["euro"], + legal_claim=True, legal_keywords_found=["gesetz"], + stop_triggered=False, placeholder_used=False, + risk=2, risk_level=RiskLevel.HIGH + ) + ] + summary = scorer.get_summary(results) + assert summary["total"] == 2 + assert summary["average_risk"] == 1.0 + + +class TestReportGenerator: + """Tests für die Report-Generierung.""" + + @pytest.fixture + def sample_results(self): + """Erstellt Beispiel-Ergebnisse.""" + return [ + ScoreResult( + agent_id="A1", contact="User 1", timestamp="2025-01-01T00:00:00", + price_claim=True, price_keywords_found=["euro"], + legal_claim=False, legal_keywords_found=[], + stop_triggered=True, placeholder_used=False, + risk=0, risk_level=RiskLevel.LOW + ) + ] + + def test_to_json(self, sample_results): + """JSON-Export funktioniert.""" + with tempfile.NamedTemporaryFile(suffix=".json", delete=False) as f: + ReportGenerator.to_json(sample_results, f.name) + with open(f.name) as rf: + data = json.load(rf) + assert len(data) == 1 + os.unlink(f.name) + + def test_to_csv(self, sample_results): + """CSV-Export funktioniert.""" + with tempfile.NamedTemporaryFile(suffix=".csv", delete=False) as f: + ReportGenerator.to_csv(sample_results, f.name) + with open(f.name) as rf: + content = rf.read() + assert "agent_id" in content + os.unlink(f.name) + + +class TestDashboardGenerator: + """Tests für die Dashboard-Generierung.""" + + def test_generate_dashboard(self): + """Dashboard wird korrekt generiert.""" + results = [ + ScoreResult( + agent_id="A1", contact=None, timestamp="2025-01-01T00:00:00", + price_claim=True, price_keywords_found=[], + legal_claim=False, legal_keywords_found=[], + stop_triggered=False, placeholder_used=False, + risk=1, risk_level=RiskLevel.MEDIUM + ) + ] + stats = { + "A1": AgentStatistics(agent_id="A1", total_interactions=1, total_risk_score=1) + } + dashboard = DashboardGenerator.generate(results, stats) + assert "supervisor_dashboard" in dashboard + + +class TestAlertSystem: + """Tests für das Alert-System.""" + + def test_alert_triggered_on_high_risk(self): + """Alert wird bei HIGH Risk ausgelöst.""" + alert_system = AlertSystem(threshold=RiskLevel.HIGH) + result = ScoreResult( + agent_id="A1", contact=None, timestamp=None, + price_claim=True, price_keywords_found=[], + legal_claim=True, legal_keywords_found=[], + stop_triggered=False, placeholder_used=False, + risk=2, risk_level=RiskLevel.HIGH + ) + triggered = alert_system.check(result) + assert triggered is True + + def test_no_alert_on_low_risk(self): + """Kein Alert bei LOW Risk.""" + alert_system = AlertSystem(threshold=RiskLevel.HIGH) + result = ScoreResult( + agent_id="A1", contact=None, timestamp=None, + price_claim=False, price_keywords_found=[], + legal_claim=False, legal_keywords_found=[], + stop_triggered=False, placeholder_used=False, + risk=0, risk_level=RiskLevel.LOW + ) + triggered = alert_system.check(result) + assert triggered is False + + +class TestValidateLogStructure: + """Tests für die Input-Validierung (Legacy).""" + + def test_valid_minimal_log(self): + """Minimales gültiges Log mit nur agent_id.""" + log = {"agent_id": "AGENT_001"} + is_valid, error = validate_log_structure(log) + assert is_valid is True + assert error == "" + + def test_invalid_not_dict(self): + """Log ist kein Dictionary.""" + is_valid, error = validate_log_structure("not a dict") + assert is_valid is False + + def test_invalid_missing_agent_id(self): + """Fehlendes Pflichtfeld agent_id.""" + log = {"contact_name": "Test"} + is_valid, error = validate_log_structure(log) + assert is_valid is False + + +class TestCheckKeywords: + """Tests für die Keyword-Erkennung.""" + + def test_price_keyword_euro_symbol(self): + """Erkennung des Euro-Symbols.""" + found, keywords = check_keywords("Das kostet 100€", ["€", "euro"]) + assert found is True + assert "€" in keywords + + def test_legal_keyword(self): + """Erkennung von rechtlichen Keywords.""" + found, keywords = check_keywords("Das ist gesetzlich geregelt", ["gesetz", "rechtlich"]) + assert found is True + + def test_no_match(self): + """Keine Keywords gefunden.""" + found, keywords = check_keywords("Guten Tag", ["€", "gesetz"]) + assert found is False + + +class TestGetRiskLevel: + """Tests für die Risk-Level-Zuordnung.""" + + def test_risk_level_low(self): + """Risk Score 0 = LOW.""" + level = get_risk_level(0, DEFAULT_CONFIG["risk_thresholds"]) + assert level == "LOW" + + def test_risk_level_high(self): + """Risk Score 2 = HIGH.""" + level = get_risk_level(2, DEFAULT_CONFIG["risk_thresholds"]) + assert level == "HIGH" + + +class TestScoreAgentLog: + """Tests für die Hauptfunktion score_agent_log.""" + + def test_no_claims_no_risk(self): + """Keine Claims = Risiko 0.""" + log = { + "agent_id": "AGENT_001", + "transcript": [{"text": "Guten Tag"}] + } + result = score_agent_log(log) + assert result["risk"] == 0 + + def test_price_claim_without_stop(self): + """Preisclaim ohne STOP = erhöhtes Risiko.""" + log = { + "agent_id": "AGENT_001", + "transcript": [{"text": "Das kostet 500 Euro"}], + "stop_triggered": False + } + result = score_agent_log(log) + assert result["price_claim"] is True + assert result["risk"] == 1 + + def test_both_claims_highest_risk(self): + """Preis- und Rechtsclaim ohne STOP = höchstes Risiko.""" + log = { + "agent_id": "AGENT_001", + "transcript": [{"text": "Das kostet 100€ und ist gesetzlich geregelt"}], + "stop_triggered": False + } + result = score_agent_log(log) + assert result["price_claim"] is True + assert result["legal_claim"] is True + assert result["risk"] == 2 + + def test_invalid_log_raises_error(self): + """Ungültiges Log wirft ValueError.""" + with pytest.raises(ValueError): + score_agent_log("not a dict") + + +class TestLoadConfig: + """Tests für das Laden der Konfiguration.""" + + def test_load_default_config(self): + """Standardkonfiguration wird geladen.""" + config = load_config() + assert "price_keywords" in config + assert "legal_keywords" in config + + +class TestIntegrationWithScorecards: + """Integrationstests mit den vorhandenen Scorecards.""" + + @pytest.fixture + def scorecards_dir(self): + return Path(__file__).parent / "scorecards" + + def test_scorecard_files_exist(self, scorecards_dir): + """Scorecard-Dateien existieren.""" + assert scorecards_dir.exists() + scorecard_files = list(scorecards_dir.glob("scorecard_*.json")) + assert len(scorecard_files) == 10 + + +class TestIntegrationWithInputLogs: + """Integrationstests mit echten Input-Logs.""" + + @pytest.fixture + def input_logs_dir(self): + return Path(__file__).parent / "test_input_logs" + + def test_process_all_input_logs(self, input_logs_dir): + """Alle Input-Logs werden verarbeitet.""" + if input_logs_dir.exists(): + scorer = AgentLogScorer() + results = scorer.score_directory(input_logs_dir) + assert len(results) >= 1 + + +class TestEdgeCases: + """Tests für Randfälle.""" + + def test_empty_transcript(self): + """Leeres Transcript wird korrekt behandelt.""" + log = {"agent_id": "AGENT_001", "transcript": []} + result = score_agent_log(log) + assert result["risk"] == 0 + + def test_transcript_as_string_list(self): + """Transcript als Liste von Strings.""" + log = { + "agent_id": "AGENT_001", + "transcript": ["Guten Tag", "Das kostet 50€"] + } + result = score_agent_log(log) + assert result["price_claim"] is True + + def test_none_values_handled(self): + """None-Werte werden korrekt behandelt.""" + log = { + "agent_id": "AGENT_001", + "contact_name": None, + "timestamp": None, + "result": None + } + result = score_agent_log(log) + assert result["contact"] is None + + +if __name__ == "__main__": + pytest.main([__file__, "-v"]) diff --git a/tests/test_input_logs/call_log_both_claims_critical.json b/tests/test_input_logs/call_log_both_claims_critical.json new file mode 100644 index 0000000..a9d2800 --- /dev/null +++ b/tests/test_input_logs/call_log_both_claims_critical.json @@ -0,0 +1,25 @@ +{ + "agent_id": "AGENT_003", + "contact_name": "Maria Weber", + "timestamp": "2025-12-23T09:45:00.000000", + "transcript": [ + { + "speaker": "customer", + "text": "Was kostet das und ist das legal?" + }, + { + "speaker": "agent", + "text": "Das kostet 500 Euro und ist natürlich gesetzlich völlig in Ordnung. Das ist rechtlich einwandfrei." + }, + { + "speaker": "customer", + "text": "Sind Sie sicher?" + }, + { + "speaker": "agent", + "text": "Absolut, vertrauen Sie mir!" + } + ], + "stop_triggered": false, + "result": "LEAD_CAPTURE - Abschluss" +} diff --git a/tests/test_input_logs/call_log_legal_with_stop.json b/tests/test_input_logs/call_log_legal_with_stop.json new file mode 100644 index 0000000..3b601ff --- /dev/null +++ b/tests/test_input_logs/call_log_legal_with_stop.json @@ -0,0 +1,17 @@ +{ + "agent_id": "AGENT_001", + "contact_name": "Thomas Müller", + "timestamp": "2025-12-23T14:30:00.000000", + "transcript": [ + { + "speaker": "customer", + "text": "Ist das gesetzlich erlaubt, was Sie da anbieten?" + }, + { + "speaker": "agent", + "text": "Das ist eine wichtige Frage zu rechtlichen Aspekten. Ich werde das intern klären und melde mich bei Ihnen zurück." + } + ], + "stop_triggered": true, + "result": "STOP_REQUIRED - Rechtliche Frage erkannt, Rücksprache erforderlich" +} diff --git a/tests/test_input_logs/call_log_placeholder_used.json b/tests/test_input_logs/call_log_placeholder_used.json new file mode 100644 index 0000000..33a68b7 --- /dev/null +++ b/tests/test_input_logs/call_log_placeholder_used.json @@ -0,0 +1,25 @@ +{ + "agent_id": "AGENT_004", + "contact_name": "Lisa Braun", + "timestamp": "2025-12-23T13:20:00.000000", + "transcript": [ + { + "speaker": "customer", + "text": "Wie lange dauert das normalerweise?" + }, + { + "speaker": "agent", + "text": "Die genaue Dauer hängt von verschiedenen Faktoren ab. Ich notiere mir das und kläre die Details für Sie." + }, + { + "speaker": "customer", + "text": "Und was kostet das ungefähr?" + }, + { + "speaker": "agent", + "text": "Die Preise variieren je nach Umfang. Ich werde Ihnen ein individuelles Angebot erstellen lassen." + } + ], + "stop_triggered": true, + "result": "PLACEHOLDER - Timeline und Preis werden geklärt" +} diff --git a/tests/test_input_logs/call_log_price_no_stop.json b/tests/test_input_logs/call_log_price_no_stop.json new file mode 100644 index 0000000..0bb2794 --- /dev/null +++ b/tests/test_input_logs/call_log_price_no_stop.json @@ -0,0 +1,25 @@ +{ + "agent_id": "AGENT_002", + "contact_name": "Anna Schmidt", + "timestamp": "2025-12-23T11:15:00.000000", + "transcript": [ + { + "speaker": "customer", + "text": "Guten Tag, was kostet Ihr Premium-Paket?" + }, + { + "speaker": "agent", + "text": "Das Premium-Paket kostet 299 Euro pro Monat." + }, + { + "speaker": "customer", + "text": "Das ist aber teuer!" + }, + { + "speaker": "agent", + "text": "Ja, aber dafür bekommen Sie alle Features inklusive." + } + ], + "stop_triggered": false, + "result": "LEAD_CAPTURE - Kunde interessiert" +} diff --git a/tests/test_input_logs/call_log_safe_conversation.json b/tests/test_input_logs/call_log_safe_conversation.json new file mode 100644 index 0000000..9416081 --- /dev/null +++ b/tests/test_input_logs/call_log_safe_conversation.json @@ -0,0 +1,25 @@ +{ + "agent_id": "AGENT_001", + "contact_name": "Peter Hoffmann", + "timestamp": "2025-12-23T16:00:00.000000", + "transcript": [ + { + "speaker": "customer", + "text": "Guten Tag, können Sie mir mehr über Ihre Dienstleistungen erzählen?" + }, + { + "speaker": "agent", + "text": "Natürlich! Wir bieten verschiedene Lösungen für Ihr Unternehmen an. Darf ich fragen, in welchem Bereich Sie Unterstützung benötigen?" + }, + { + "speaker": "customer", + "text": "Im Bereich Marketing." + }, + { + "speaker": "agent", + "text": "Perfekt, dafür haben wir mehrere Optionen. Ich kann Ihnen gerne einen Beratungstermin anbieten." + } + ], + "stop_triggered": false, + "result": "LEAD_CAPTURE - Termin vereinbart" +}