"""ASV benchmarks for Semantic Escalation confidence computation. Measures the performance of: - ConfidenceFactors model construction - EscalationDecision model construction - AutonomyController confidence computation - AutonomyController.should_proceed_automatically end-to-end - Historical outcome recording and success rate retrieval """ from __future__ import annotations import sys from pathlib import Path try: from cleveragents.application.services.autonomy_controller import ( AutonomyController, ) from cleveragents.domain.models.core.automation_profile import ( BUILTIN_PROFILES, ) from cleveragents.domain.models.core.escalation import ( ConfidenceFactors, HistoricalOutcome, OperationContext, ) except ModuleNotFoundError: sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src")) from cleveragents.application.services.autonomy_controller import ( AutonomyController, ) from cleveragents.domain.models.core.automation_profile import ( BUILTIN_PROFILES, ) from cleveragents.domain.models.core.escalation import ( ConfidenceFactors, HistoricalOutcome, OperationContext, ) class ConfidenceFactorsConstructionSuite: """Benchmark ConfidenceFactors model construction.""" def time_factors_construction_defaults(self) -> None: """Benchmark default ConfidenceFactors creation.""" ConfidenceFactors() def time_factors_construction_full(self) -> None: """Benchmark fully-populated ConfidenceFactors creation.""" ConfidenceFactors( past_success_rate=0.85, codebase_familiarity=0.9, risk_assessment=0.2, invariant_complexity=0.3, ) class ConfidenceComputationSuite: """Benchmark confidence score computation throughput.""" def setup(self) -> None: """Set up the controller and factors for benchmarking.""" self.controller = AutonomyController() self.factors_high = ConfidenceFactors( past_success_rate=0.95, codebase_familiarity=0.9, risk_assessment=0.1, invariant_complexity=0.1, ) self.factors_low = ConfidenceFactors( past_success_rate=0.1, codebase_familiarity=0.2, risk_assessment=0.9, invariant_complexity=0.8, ) self.factors_neutral = ConfidenceFactors( past_success_rate=0.5, codebase_familiarity=0.5, risk_assessment=0.5, invariant_complexity=0.5, ) def time_compute_confidence_high(self) -> None: """Benchmark confidence computation with high factors.""" self.controller.compute_confidence(self.factors_high) def time_compute_confidence_low(self) -> None: """Benchmark confidence computation with low factors.""" self.controller.compute_confidence(self.factors_low) def time_compute_confidence_neutral(self) -> None: """Benchmark confidence computation with neutral factors.""" self.controller.compute_confidence(self.factors_neutral) class EscalationDecisionSuite: """Benchmark end-to-end escalation decision throughput.""" def setup(self) -> None: """Set up controller, factors, and profiles.""" self.controller = AutonomyController() self.factors = ConfidenceFactors( past_success_rate=0.8, codebase_familiarity=0.75, risk_assessment=0.15, invariant_complexity=0.2, ) self.operation = OperationContext(operation_type="create_tool") self.profile_cautious = BUILTIN_PROFILES["cautious"] self.profile_fullauto = BUILTIN_PROFILES["full-auto"] self.profile_manual = BUILTIN_PROFILES["manual"] def time_should_proceed_cautious(self) -> None: """Benchmark should_proceed_automatically with cautious profile.""" self.controller.should_proceed_automatically( self.operation, self.factors, self.profile_cautious, ) def time_should_proceed_fullauto(self) -> None: """Benchmark should_proceed_automatically with full-auto profile.""" self.controller.should_proceed_automatically( self.operation, self.factors, self.profile_fullauto, ) def time_should_proceed_manual(self) -> None: """Benchmark should_proceed_automatically with manual profile.""" self.controller.should_proceed_automatically( self.operation, self.factors, self.profile_manual, ) class HistoricalTrackingSuite: """Benchmark historical outcome recording and retrieval.""" def setup(self) -> None: """Set up controller with pre-populated history.""" self.controller = AutonomyController() self.outcome_success = HistoricalOutcome( operation_type="create_tool", succeeded=True, ) self.outcome_failure = HistoricalOutcome( operation_type="create_tool", succeeded=False, ) # Pre-populate history for _ in range(100): self.controller.record_outcome(self.outcome_success) def time_record_outcome(self) -> None: """Benchmark recording a single outcome.""" self.controller.record_outcome(self.outcome_success) def time_get_success_rate(self) -> None: """Benchmark retrieving historical success rate.""" self.controller.get_historical_success_rate("create_tool") def time_get_history_count(self) -> None: """Benchmark retrieving history count.""" self.controller.get_history_count("create_tool")