Files
cleveragents-core/benchmarks/semantic_escalation_bench.py
freemo 583e6b7ea2 feat(correcting-plans): implement Predictive Error Prevention (Layer 4) with Error Pattern Database
Implement spec-mandated Layer 4 Predictive Error Prevention system:

- ErrorPattern domain model with pattern text, historical failures,
  preventive checks, frequency tracking, and keyword-based matching.
- ErrorPatternRepository with in-memory CRUD + context-matching query.
- ErrorPatternService with record_failure(), match_patterns(), and
  get_statistics() methods.
- Wire into plan execution via plan_lifecycle_service pre-execution hook.
- Add error pattern statistics to CLI diagnostics output.

Behave BDD: 11 scenarios covering recording, matching, formatting, stats.
Robot Framework: 3 integration smoke tests.
ASV benchmarks: pattern matching performance.

ISSUES CLOSED: #571
2026-03-08 21:53:21 -04:00

169 lines
5.7 KiB
Python

"""ASV benchmarks for Semantic Escalation confidence computation.
Measures the performance of:
- ConfidenceFactors model construction
- EscalationDecision model construction
- AutonomyController confidence computation
- AutonomyController.should_proceed_automatically end-to-end
- Historical outcome recording and success rate retrieval
"""
from __future__ import annotations
import sys
from pathlib import Path
try:
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.domain.models.core.automation_profile import (
BUILTIN_PROFILES,
)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
HistoricalOutcome,
OperationContext,
)
except ModuleNotFoundError:
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.domain.models.core.automation_profile import (
BUILTIN_PROFILES,
)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
HistoricalOutcome,
OperationContext,
)
class ConfidenceFactorsConstructionSuite:
"""Benchmark ConfidenceFactors model construction."""
def time_factors_construction_defaults(self) -> None:
"""Benchmark default ConfidenceFactors creation."""
ConfidenceFactors()
def time_factors_construction_full(self) -> None:
"""Benchmark fully-populated ConfidenceFactors creation."""
ConfidenceFactors(
past_success_rate=0.85,
codebase_familiarity=0.9,
risk_assessment=0.2,
invariant_complexity=0.3,
)
class ConfidenceComputationSuite:
"""Benchmark confidence score computation throughput."""
def setup(self) -> None:
"""Set up the controller and factors for benchmarking."""
self.controller = AutonomyController()
self.factors_high = ConfidenceFactors(
past_success_rate=0.95,
codebase_familiarity=0.9,
risk_assessment=0.1,
invariant_complexity=0.1,
)
self.factors_low = ConfidenceFactors(
past_success_rate=0.1,
codebase_familiarity=0.2,
risk_assessment=0.9,
invariant_complexity=0.8,
)
self.factors_neutral = ConfidenceFactors(
past_success_rate=0.5,
codebase_familiarity=0.5,
risk_assessment=0.5,
invariant_complexity=0.5,
)
def time_compute_confidence_high(self) -> None:
"""Benchmark confidence computation with high factors."""
self.controller.compute_confidence(self.factors_high)
def time_compute_confidence_low(self) -> None:
"""Benchmark confidence computation with low factors."""
self.controller.compute_confidence(self.factors_low)
def time_compute_confidence_neutral(self) -> None:
"""Benchmark confidence computation with neutral factors."""
self.controller.compute_confidence(self.factors_neutral)
class EscalationDecisionSuite:
"""Benchmark end-to-end escalation decision throughput."""
def setup(self) -> None:
"""Set up controller, factors, and profiles."""
self.controller = AutonomyController()
self.factors = ConfidenceFactors(
past_success_rate=0.8,
codebase_familiarity=0.75,
risk_assessment=0.15,
invariant_complexity=0.2,
)
self.operation = OperationContext(operation_type="auto_execute")
self.profile_cautious = BUILTIN_PROFILES["cautious"]
self.profile_fullauto = BUILTIN_PROFILES["full-auto"]
self.profile_manual = BUILTIN_PROFILES["manual"]
def time_should_proceed_cautious(self) -> None:
"""Benchmark should_proceed_automatically with cautious profile."""
self.controller.should_proceed_automatically(
self.operation,
self.factors,
self.profile_cautious,
)
def time_should_proceed_fullauto(self) -> None:
"""Benchmark should_proceed_automatically with full-auto profile."""
self.controller.should_proceed_automatically(
self.operation,
self.factors,
self.profile_fullauto,
)
def time_should_proceed_manual(self) -> None:
"""Benchmark should_proceed_automatically with manual profile."""
self.controller.should_proceed_automatically(
self.operation,
self.factors,
self.profile_manual,
)
class HistoricalTrackingSuite:
"""Benchmark historical outcome recording and retrieval."""
def setup(self) -> None:
"""Set up controller with pre-populated history."""
self.controller = AutonomyController()
self.outcome_success = HistoricalOutcome(
operation_type="auto_execute",
succeeded=True,
)
self.outcome_failure = HistoricalOutcome(
operation_type="auto_execute",
succeeded=False,
)
# Pre-populate history
for _ in range(100):
self.controller.record_outcome(self.outcome_success)
def time_record_outcome(self) -> None:
"""Benchmark recording a single outcome."""
self.controller.record_outcome(self.outcome_success)
def time_get_success_rate(self) -> None:
"""Benchmark retrieving historical success rate."""
self.controller.get_historical_success_rate("auto_execute")
def time_get_history_count(self) -> None:
"""Benchmark retrieving history count."""
self.controller.get_history_count("auto_execute")