feat(autonomy): implement semantic escalation with confidence scoring and threshold comparison
CI / lint (pull_request) Successful in 14s
CI / typecheck (pull_request) Successful in 31s
CI / security (pull_request) Successful in 30s
CI / quality (pull_request) Successful in 15s
CI / benchmark-publish (pull_request) Has been skipped
CI / build (pull_request) Successful in 15s
CI / unit_tests (pull_request) Successful in 1m58s
CI / integration_tests (pull_request) Successful in 3m24s
CI / docker (pull_request) Successful in 40s
CI / coverage (pull_request) Successful in 4m3s
CI / lint (push) Successful in 12s
CI / typecheck (push) Successful in 33s
CI / quality (push) Successful in 15s
CI / security (push) Successful in 30s
CI / unit_tests (push) Successful in 2m7s
CI / build (push) Successful in 15s
CI / benchmark-regression (push) Has been skipped
CI / integration_tests (push) Successful in 2m57s
CI / docker (push) Successful in 43s
CI / coverage (push) Successful in 4m14s
CI / benchmark-publish (push) Successful in 14m12s
CI / benchmark-regression (pull_request) Successful in 31m33s

This commit was merged in pull request #558.
This commit is contained in:
2026-03-04 04:56:31 +00:00
parent 296431cfd6
commit db5e5c974f
12 changed files with 1866 additions and 0 deletions
+9
View File
@@ -11,6 +11,15 @@
`execute_revert`, `execute_correction`, `generate_dry_run_report`) now accept optional
`influence_edges` parameter. Includes Behave BDD scenarios, Robot Framework integration
tests, and ASV benchmarks. (#542)
- Added Semantic Escalation system with `AutonomyController` class implementing
`should_proceed_automatically()` that computes confidence scores from weighted factors
(past_success_rate, codebase_familiarity, risk_assessment, invariant_complexity) and
compares them against automation profile thresholds. Includes `EscalationDecision`,
`ConfidenceFactors`, `OperationContext`, and `HistoricalOutcome` domain models.
Historical success tracking records outcomes for future confidence computation.
Integrates with all 8 built-in automation profiles. DI-wired as singleton
`autonomy_controller`. Includes 48 Behave BDD scenarios, 10 Robot Framework
integration tests, and ASV benchmarks. (#546)
- Fixed `context inspect` to display project-scoped tier fragment counts instead of global
counts. Added `ContextTierService.get_scoped_metrics()` which returns fragment population
counts filtered to the target project while keeping hit/miss counters as global service
+170
View File
@@ -0,0 +1,170 @@
"""ASV benchmarks for Semantic Escalation confidence computation.
Measures the performance of:
- ConfidenceFactors model construction
- EscalationDecision model construction
- AutonomyController confidence computation
- AutonomyController.should_proceed_automatically end-to-end
- Historical outcome recording and success rate retrieval
"""
from __future__ import annotations
import sys
from pathlib import Path
try:
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.domain.models.core.automation_profile import (
BUILTIN_PROFILES,
AutomationProfile,
)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
HistoricalOutcome,
OperationContext,
)
except ModuleNotFoundError:
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.domain.models.core.automation_profile import (
BUILTIN_PROFILES,
AutomationProfile,
)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
HistoricalOutcome,
OperationContext,
)
class ConfidenceFactorsConstructionSuite:
"""Benchmark ConfidenceFactors model construction."""
def time_factors_construction_defaults(self) -> None:
"""Benchmark default ConfidenceFactors creation."""
ConfidenceFactors()
def time_factors_construction_full(self) -> None:
"""Benchmark fully-populated ConfidenceFactors creation."""
ConfidenceFactors(
past_success_rate=0.85,
codebase_familiarity=0.9,
risk_assessment=0.2,
invariant_complexity=0.3,
)
class ConfidenceComputationSuite:
"""Benchmark confidence score computation throughput."""
def setup(self) -> None:
"""Set up the controller and factors for benchmarking."""
self.controller = AutonomyController()
self.factors_high = ConfidenceFactors(
past_success_rate=0.95,
codebase_familiarity=0.9,
risk_assessment=0.1,
invariant_complexity=0.1,
)
self.factors_low = ConfidenceFactors(
past_success_rate=0.1,
codebase_familiarity=0.2,
risk_assessment=0.9,
invariant_complexity=0.8,
)
self.factors_neutral = ConfidenceFactors(
past_success_rate=0.5,
codebase_familiarity=0.5,
risk_assessment=0.5,
invariant_complexity=0.5,
)
def time_compute_confidence_high(self) -> None:
"""Benchmark confidence computation with high factors."""
self.controller.compute_confidence(self.factors_high)
def time_compute_confidence_low(self) -> None:
"""Benchmark confidence computation with low factors."""
self.controller.compute_confidence(self.factors_low)
def time_compute_confidence_neutral(self) -> None:
"""Benchmark confidence computation with neutral factors."""
self.controller.compute_confidence(self.factors_neutral)
class EscalationDecisionSuite:
"""Benchmark end-to-end escalation decision throughput."""
def setup(self) -> None:
"""Set up controller, factors, and profiles."""
self.controller = AutonomyController()
self.factors = ConfidenceFactors(
past_success_rate=0.8,
codebase_familiarity=0.75,
risk_assessment=0.15,
invariant_complexity=0.2,
)
self.operation = OperationContext(operation_type="auto_execute")
self.profile_cautious = BUILTIN_PROFILES["cautious"]
self.profile_fullauto = BUILTIN_PROFILES["full-auto"]
self.profile_manual = BUILTIN_PROFILES["manual"]
def time_should_proceed_cautious(self) -> None:
"""Benchmark should_proceed_automatically with cautious profile."""
self.controller.should_proceed_automatically(
self.operation,
self.factors,
self.profile_cautious,
)
def time_should_proceed_fullauto(self) -> None:
"""Benchmark should_proceed_automatically with full-auto profile."""
self.controller.should_proceed_automatically(
self.operation,
self.factors,
self.profile_fullauto,
)
def time_should_proceed_manual(self) -> None:
"""Benchmark should_proceed_automatically with manual profile."""
self.controller.should_proceed_automatically(
self.operation,
self.factors,
self.profile_manual,
)
class HistoricalTrackingSuite:
"""Benchmark historical outcome recording and retrieval."""
def setup(self) -> None:
"""Set up controller with pre-populated history."""
self.controller = AutonomyController()
self.outcome_success = HistoricalOutcome(
operation_type="auto_execute",
succeeded=True,
)
self.outcome_failure = HistoricalOutcome(
operation_type="auto_execute",
succeeded=False,
)
# Pre-populate history
for _ in range(100):
self.controller.record_outcome(self.outcome_success)
def time_record_outcome(self) -> None:
"""Benchmark recording a single outcome."""
self.controller.record_outcome(self.outcome_success)
def time_get_success_rate(self) -> None:
"""Benchmark retrieving historical success rate."""
self.controller.get_historical_success_rate("auto_execute")
def time_get_history_count(self) -> None:
"""Benchmark retrieving history count."""
self.controller.get_history_count("auto_execute")
+223
View File
@@ -0,0 +1,223 @@
Feature: Semantic Escalation with Confidence Scoring
As an automation framework
I want to compute confidence scores and compare them against profile thresholds
So that the system can decide whether to proceed automatically or escalate to the user
Background:
Given a default autonomy controller
# ---- Confidence computation ----
Scenario: Confidence with all factors at maximum
When I compute confidence with past_success_rate 1.0 codebase_familiarity 1.0 risk_assessment 0.0 invariant_complexity 0.0
Then the confidence score should be 1.0
Scenario: Confidence with all factors at minimum
When I compute confidence with past_success_rate 0.0 codebase_familiarity 0.0 risk_assessment 1.0 invariant_complexity 1.0
Then the confidence score should be 0.0
Scenario: Confidence with neutral factors
When I compute confidence with past_success_rate 0.5 codebase_familiarity 0.5 risk_assessment 0.5 invariant_complexity 0.5
Then the confidence score should be 0.5
Scenario: Confidence with mixed high and low factors
When I compute confidence with past_success_rate 0.9 codebase_familiarity 0.8 risk_assessment 0.2 invariant_complexity 0.1
Then the confidence score should be approximately 0.84
Scenario: Risk inversion in confidence computation
When I compute confidence with past_success_rate 0.5 codebase_familiarity 0.5 risk_assessment 1.0 invariant_complexity 0.5
Then the confidence score should be less than 0.5
Scenario: Invariant complexity inversion in confidence computation
When I compute confidence with past_success_rate 0.5 codebase_familiarity 0.5 risk_assessment 0.5 invariant_complexity 1.0
Then the confidence score should be less than 0.5
# ---- Threshold comparison for each of the 8 profiles ----
Scenario Outline: Threshold comparison for built-in profile <profile_name>
Given the automation profile "<profile_name>"
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.85 codebase_familiarity 0.9 risk_assessment 0.1 invariant_complexity 0.1
Then the escalation decision should match the "<profile_name>" profile threshold for "auto_execute"
Examples:
| profile_name |
| manual |
| review |
| supervised |
| cautious |
| trusted |
| auto |
| ci |
| full-auto |
Scenario: Threshold 0.0 always proceeds automatically
Given a profile with auto_execute threshold 0.0
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.0 codebase_familiarity 0.0 risk_assessment 1.0 invariant_complexity 1.0
Then the escalation decision proceed should be true
Scenario: Threshold 1.0 escalates with realistic confidence
Given a profile with auto_execute threshold 1.0
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.95 codebase_familiarity 0.95 risk_assessment 0.05 invariant_complexity 0.05
Then the escalation decision proceed should be false
Scenario: Exactly at threshold proceeds
Given a profile with auto_execute threshold 0.5
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.5 codebase_familiarity 0.5 risk_assessment 0.5 invariant_complexity 0.5
Then the escalation decision proceed should be true
Scenario: Just below threshold escalates
Given a profile with auto_execute threshold 0.51
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.5 codebase_familiarity 0.5 risk_assessment 0.5 invariant_complexity 0.5
Then the escalation decision proceed should be false
# ---- Historical success tracking ----
Scenario: Historical success rate starts at neutral
Then the historical success rate for "auto_execute" should be 0.5
Scenario: Recording successes increases success rate
When I record 8 successes and 2 failures for "auto_execute"
Then the historical success rate for "auto_execute" should be 0.8
Scenario: Recording all failures gives zero success rate
When I record 0 successes and 5 failures for "auto_execute"
Then the historical success rate for "auto_execute" should be 0.0
Scenario: Recording all successes gives perfect success rate
When I record 10 successes and 0 failures for "auto_execute"
Then the historical success rate for "auto_execute" should be 1.0
Scenario: Success rate evolves with new outcomes
When I record 5 successes and 5 failures for "auto_execute"
Then the historical success rate for "auto_execute" should be 0.5
When I record 5 additional successes for "auto_execute"
Then the historical success rate for "auto_execute" should be approximately 0.667
Scenario: Clearing history resets to neutral
When I record 10 successes and 0 failures for "auto_execute"
And I clear history for "auto_execute"
Then the historical success rate for "auto_execute" should be 0.5
Scenario: History count tracks recorded outcomes
When I record 3 successes and 2 failures for "auto_execute"
Then the history count for "auto_execute" should be 5
# ---- Escalation explanation generation ----
Scenario: Proceeding decision includes explanation
Given a profile with auto_execute threshold 0.3
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.9 codebase_familiarity 0.9 risk_assessment 0.1 invariant_complexity 0.1
Then the escalation explanation should contain "Proceeding automatically"
And the escalation explanation should contain "auto_execute"
Scenario: Escalating decision includes explanation
Given a profile with auto_execute threshold 0.99
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.5 codebase_familiarity 0.5 risk_assessment 0.5 invariant_complexity 0.5
Then the escalation explanation should contain "Escalating to user"
And the escalation explanation should contain "auto_execute"
# ---- Edge cases ----
Scenario: Zero confidence with zero threshold proceeds
Given a profile with auto_execute threshold 0.0
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.0 codebase_familiarity 0.0 risk_assessment 1.0 invariant_complexity 1.0
Then the escalation decision proceed should be true
And the confidence should be 0.0
Scenario: Perfect confidence with threshold below 1.0 proceeds
Given a profile with auto_execute threshold 0.99
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 1.0 codebase_familiarity 1.0 risk_assessment 0.0 invariant_complexity 0.0
Then the escalation decision proceed should be true
And the confidence should be 1.0
Scenario: Unknown operation type defaults to manual threshold
Given the automation profile "full-auto"
When I evaluate escalation for operation "nonexistent_operation" with confidence factors past_success_rate 0.95 codebase_familiarity 0.95 risk_assessment 0.05 invariant_complexity 0.05
Then the escalation threshold should be 1.0
And the escalation decision proceed should be false
# ---- Custom weights ----
Scenario: Custom weights are validated
When I try to create a controller with invalid weights
Then a weight validation error should be raised
Scenario: Custom equal weights produce different score
Given a controller with equal weights
When I compute confidence with past_success_rate 1.0 codebase_familiarity 0.0 risk_assessment 0.0 invariant_complexity 0.0
Then the confidence score should be 0.75
# ---- Factor model validation ----
Scenario: ConfidenceFactors rejects out-of-range values
When I try to create confidence factors with past_success_rate 1.5
Then a factor validation error should be raised
Scenario: OperationContext requires non-empty operation type
When I try to create an operation context with empty operation type
Then an operation context validation error should be raised
# ---- EscalationDecision model ----
Scenario: EscalationDecision captures all fields
Given a profile with auto_execute threshold 0.5
When I evaluate escalation for operation "auto_execute" with confidence factors past_success_rate 0.8 codebase_familiarity 0.7 risk_assessment 0.2 invariant_complexity 0.3
Then the escalation decision should have operation_type "auto_execute"
And the escalation decision should have 4 factors
And the escalation decision threshold should be 0.5
# ---- Argument validation coverage ----
Scenario: should_proceed_automatically rejects None operation
When I call should_proceed_automatically with None operation
Then a weight validation error should be raised
Scenario: should_proceed_automatically rejects None factors
When I call should_proceed_automatically with None factors
Then a weight validation error should be raised
Scenario: should_proceed_automatically rejects None profile
When I call should_proceed_automatically with None profile
Then a weight validation error should be raised
Scenario: compute_confidence rejects None factors
When I call compute_confidence with None factors
Then a weight validation error should be raised
Scenario: record_outcome rejects None outcome
When I call record_outcome with None outcome
Then a weight validation error should be raised
Scenario: get_historical_success_rate rejects empty operation type
When I call get_historical_success_rate with empty string
Then a weight validation error should be raised
Scenario: get_history_count rejects empty operation type
When I call get_history_count with empty string
Then a weight validation error should be raised
Scenario: Clearing all history resets everything
When I record 5 successes and 0 failures for "auto_execute"
And I record 5 successes and 0 failures for "auto_apply"
And I clear all history
Then the historical success rate for "auto_execute" should be 0.5
And the historical success rate for "auto_apply" should be 0.5
Scenario: Weights property returns current weights
Then the controller weights should have 4 entries
Scenario: History eviction at max capacity
When I record max history outcomes for "auto_execute"
Then the history count for "auto_execute" should be 10000
Scenario: Controller rejects extra weight keys
When I try to create a controller with extra weight keys
Then a weight validation error should be raised
Scenario: Controller rejects negative weight
When I try to create a controller with negative weight
Then a weight validation error should be raised
Scenario: Controller rejects weights not summing to 1
When I try to create a controller with weights not summing to 1
Then a weight validation error should be raised
+576
View File
@@ -0,0 +1,576 @@
"""Step definitions for Semantic Escalation with Confidence Scoring."""
from __future__ import annotations
from behave import given, then, when
from behave.runner import Context
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.domain.models.core.automation_profile import (
BUILTIN_PROFILES,
AutomationProfile,
)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
HistoricalOutcome,
OperationContext,
)
# ---- Background / Setup steps ----
@given("a default autonomy controller")
def step_default_controller(context: Context) -> None:
"""Create a default AutonomyController instance."""
context.controller = AutonomyController()
context.decision = None
context.error = None
# ---- Confidence computation steps ----
@when(
"I compute confidence with past_success_rate {psr:g} "
"codebase_familiarity {cf:g} risk_assessment {ra:g} "
"invariant_complexity {ic:g}"
)
def step_compute_confidence(
context: Context,
psr: float,
cf: float,
ra: float,
ic: float,
) -> None:
"""Compute confidence from provided factors."""
factors = ConfidenceFactors(
past_success_rate=psr,
codebase_familiarity=cf,
risk_assessment=ra,
invariant_complexity=ic,
)
context.confidence = context.controller.compute_confidence(factors)
@then("the confidence score should be {expected:g}")
def step_confidence_should_be(context: Context, expected: float) -> None:
"""Assert exact confidence score."""
assert abs(context.confidence - expected) < 1e-6, (
f"Expected confidence {expected}, got {context.confidence}"
)
@then("the confidence score should be approximately {expected:g}")
def step_confidence_approximately(
context: Context,
expected: float,
) -> None:
"""Assert confidence within tolerance."""
assert abs(context.confidence - expected) < 0.02, (
f"Expected confidence ~{expected}, got {context.confidence}"
)
@then("the confidence score should be less than {bound:g}")
def step_confidence_less_than(context: Context, bound: float) -> None:
"""Assert confidence is strictly less than bound."""
assert context.confidence < bound, (
f"Expected confidence < {bound}, got {context.confidence}"
)
# ---- Threshold / Profile steps ----
@given('the automation profile "{profile_name}"')
def step_given_profile(context: Context, profile_name: str) -> None:
"""Load a built-in automation profile by name."""
context.profile = BUILTIN_PROFILES[profile_name]
@given("a profile with auto_execute threshold {threshold:g}")
def step_given_custom_threshold_profile(
context: Context,
threshold: float,
) -> None:
"""Create a minimal profile with a specific auto_execute threshold."""
context.profile = AutomationProfile(
name="test-profile",
auto_execute=threshold,
)
@when(
'I evaluate escalation for operation "{op}" with confidence factors '
"past_success_rate {psr:g} codebase_familiarity {cf:g} "
"risk_assessment {ra:g} invariant_complexity {ic:g}"
)
def step_evaluate_escalation(
context: Context,
op: str,
psr: float,
cf: float,
ra: float,
ic: float,
) -> None:
"""Evaluate should_proceed_automatically with given factors."""
factors = ConfidenceFactors(
past_success_rate=psr,
codebase_familiarity=cf,
risk_assessment=ra,
invariant_complexity=ic,
)
operation = OperationContext(operation_type=op)
context.decision = context.controller.should_proceed_automatically(
operation=operation,
factors=factors,
profile=context.profile,
)
context.confidence = context.decision.confidence
@then(
'the escalation decision should match the "{profile_name}" '
'profile threshold for "{op}"'
)
def step_decision_matches_profile(
context: Context,
profile_name: str,
op: str,
) -> None:
"""Verify the decision matches the profile's threshold for the op."""
profile = BUILTIN_PROFILES[profile_name]
threshold = getattr(profile, op, 1.0)
decision = context.decision
assert decision is not None
if threshold == 0.0:
assert decision.proceed is True, (
f"Profile '{profile_name}' has threshold 0.0 for '{op}', "
f"expected proceed=True but got False"
)
elif threshold == 1.0:
assert decision.proceed is False, (
f"Profile '{profile_name}' has threshold 1.0 for '{op}', "
f"expected proceed=False but got True"
)
else:
expected = decision.confidence >= threshold
assert decision.proceed == expected, (
f"Profile '{profile_name}': confidence={decision.confidence:.3f} "
f"threshold={threshold:.3f}, expected proceed={expected} "
f"but got {decision.proceed}"
)
@then("the escalation decision proceed should be true")
def step_decision_proceed_true(context: Context) -> None:
"""Assert the escalation decision is to proceed."""
assert context.decision is not None
assert context.decision.proceed is True, (
f"Expected proceed=True, got False "
f"(confidence={context.decision.confidence:.3f}, "
f"threshold={context.decision.threshold:.3f})"
)
@then("the escalation decision proceed should be false")
def step_decision_proceed_false(context: Context) -> None:
"""Assert the escalation decision is to escalate."""
assert context.decision is not None
assert context.decision.proceed is False, (
f"Expected proceed=False, got True "
f"(confidence={context.decision.confidence:.3f}, "
f"threshold={context.decision.threshold:.3f})"
)
@then("the escalation threshold should be {expected:g}")
def step_escalation_threshold(context: Context, expected: float) -> None:
"""Assert the threshold used in the decision."""
assert context.decision is not None
assert abs(context.decision.threshold - expected) < 1e-6, (
f"Expected threshold {expected}, got {context.decision.threshold}"
)
@then("the confidence should be {expected:g}")
def step_confidence_exact(context: Context, expected: float) -> None:
"""Assert confidence from the decision."""
assert context.decision is not None
assert abs(context.decision.confidence - expected) < 1e-6, (
f"Expected confidence {expected}, got {context.decision.confidence}"
)
# ---- Historical success tracking steps ----
@then('the historical success rate for "{op}" should be {expected:g}')
def step_historical_rate(
context: Context,
op: str,
expected: float,
) -> None:
"""Assert historical success rate."""
rate = context.controller.get_historical_success_rate(op)
assert abs(rate - expected) < 1e-6, (
f"Expected rate {expected} for '{op}', got {rate}"
)
@then('the historical success rate for "{op}" should be approximately {expected:g}')
def step_historical_rate_approx(
context: Context,
op: str,
expected: float,
) -> None:
"""Assert historical success rate within tolerance."""
rate = context.controller.get_historical_success_rate(op)
assert abs(rate - expected) < 0.02, (
f"Expected rate ~{expected} for '{op}', got {rate}"
)
@when('I record {successes:d} successes and {failures:d} failures for "{op}"')
def step_record_outcomes(
context: Context,
successes: int,
failures: int,
op: str,
) -> None:
"""Record a batch of successes and failures."""
for _ in range(successes):
context.controller.record_outcome(
HistoricalOutcome(operation_type=op, succeeded=True)
)
for _ in range(failures):
context.controller.record_outcome(
HistoricalOutcome(operation_type=op, succeeded=False)
)
@when('I record {count:d} additional successes for "{op}"')
def step_record_additional_successes(
context: Context,
count: int,
op: str,
) -> None:
"""Record additional successes."""
for _ in range(count):
context.controller.record_outcome(
HistoricalOutcome(operation_type=op, succeeded=True)
)
@when('I clear history for "{op}"')
def step_clear_history(context: Context, op: str) -> None:
"""Clear history for a specific operation type."""
context.controller.clear_history(op)
@then('the history count for "{op}" should be {expected:d}')
def step_history_count(
context: Context,
op: str,
expected: int,
) -> None:
"""Assert the number of recorded outcomes."""
count = context.controller.get_history_count(op)
assert count == expected, f"Expected {expected} outcomes for '{op}', got {count}"
# ---- Explanation steps ----
@then('the escalation explanation should contain "{text}"')
def step_explanation_contains(context: Context, text: str) -> None:
"""Assert the explanation contains expected text."""
assert context.decision is not None
assert text in context.decision.explanation, (
f"Expected '{text}' in explanation: {context.decision.explanation}"
)
# ---- Custom weights steps ----
@when("I try to create a controller with invalid weights")
def step_try_invalid_weights(context: Context) -> None:
"""Attempt to create a controller with bad weights."""
try:
AutonomyController(
weights={
"past_success_rate": 0.5,
"codebase_familiarity": 0.5,
# missing risk_assessment and invariant_complexity
}
)
context.error = None
except ValueError as exc:
context.error = exc
@then("a weight validation error should be raised")
def step_weight_validation_error_raised(context: Context) -> None:
"""Assert a ValueError was raised for weight validation."""
assert context.error is not None
assert isinstance(context.error, ValueError), (
f"Expected ValueError, got {type(context.error).__name__}"
)
@given("a controller with equal weights")
def step_equal_weights_controller(context: Context) -> None:
"""Create a controller where all 4 weights are 0.25."""
context.controller = AutonomyController(
weights={
"past_success_rate": 0.25,
"codebase_familiarity": 0.25,
"risk_assessment": 0.25,
"invariant_complexity": 0.25,
}
)
# ---- Model validation steps ----
@when("I try to create confidence factors with past_success_rate {value:g}")
def step_try_bad_confidence_factors(
context: Context,
value: float,
) -> None:
"""Attempt to create ConfidenceFactors with out-of-range value."""
try:
ConfidenceFactors(past_success_rate=value)
context.error = None
except Exception as exc:
context.error = exc
@then("a factor validation error should be raised")
def step_factor_validation_error_raised(context: Context) -> None:
"""Assert a validation error was raised for factors."""
assert context.error is not None, "Expected an error but none was raised"
@then("an operation context validation error should be raised")
def step_op_context_validation_error_raised(context: Context) -> None:
"""Assert a validation error was raised for operation context."""
assert context.error is not None, "Expected an error but none was raised"
@when("I try to create an operation context with empty operation type")
def step_try_empty_operation_context(context: Context) -> None:
"""Attempt to create OperationContext with empty operation_type."""
try:
OperationContext(operation_type="")
context.error = None
except Exception as exc:
context.error = exc
# ---- EscalationDecision field steps ----
@then('the escalation decision should have operation_type "{op}"')
def step_decision_operation_type(context: Context, op: str) -> None:
"""Assert the decision's operation_type."""
assert context.decision is not None
assert context.decision.operation_type == op
@then("the escalation decision should have {count:d} factors")
def step_decision_factor_count(context: Context, count: int) -> None:
"""Assert the number of factors in the decision."""
assert context.decision is not None
assert len(context.decision.factors) == count, (
f"Expected {count} factors, got {len(context.decision.factors)}"
)
@then("the escalation decision threshold should be {expected:g}")
def step_decision_threshold_value(
context: Context,
expected: float,
) -> None:
"""Assert the decision's threshold."""
assert context.decision is not None
assert abs(context.decision.threshold - expected) < 1e-6, (
f"Expected threshold {expected}, got {context.decision.threshold}"
)
# ---- Argument validation coverage steps ----
@when("I call should_proceed_automatically with None operation")
def step_call_proceed_none_operation(context: Context) -> None:
"""Call should_proceed_automatically with None operation."""
try:
factors = ConfidenceFactors()
profile = AutomationProfile(name="test")
context.controller.should_proceed_automatically(
None,
factors,
profile, # type: ignore[arg-type]
)
context.error = None
except ValueError as exc:
context.error = exc
@when("I call should_proceed_automatically with None factors")
def step_call_proceed_none_factors(context: Context) -> None:
"""Call should_proceed_automatically with None factors."""
try:
op = OperationContext(operation_type="auto_execute")
profile = AutomationProfile(name="test")
context.controller.should_proceed_automatically(
op,
None,
profile, # type: ignore[arg-type]
)
context.error = None
except ValueError as exc:
context.error = exc
@when("I call should_proceed_automatically with None profile")
def step_call_proceed_none_profile(context: Context) -> None:
"""Call should_proceed_automatically with None profile."""
try:
op = OperationContext(operation_type="auto_execute")
factors = ConfidenceFactors()
context.controller.should_proceed_automatically(
op,
factors,
None, # type: ignore[arg-type]
)
context.error = None
except ValueError as exc:
context.error = exc
@when("I call compute_confidence with None factors")
def step_call_compute_confidence_none(context: Context) -> None:
"""Call compute_confidence with None factors."""
try:
context.controller.compute_confidence(
None, # type: ignore[arg-type]
)
context.error = None
except ValueError as exc:
context.error = exc
@when("I call record_outcome with None outcome")
def step_call_record_outcome_none(context: Context) -> None:
"""Call record_outcome with None outcome."""
try:
context.controller.record_outcome(
None, # type: ignore[arg-type]
)
context.error = None
except ValueError as exc:
context.error = exc
@when("I call get_historical_success_rate with empty string")
def step_call_get_rate_empty(context: Context) -> None:
"""Call get_historical_success_rate with empty string."""
try:
context.controller.get_historical_success_rate("")
context.error = None
except ValueError as exc:
context.error = exc
@when("I call get_history_count with empty string")
def step_call_get_count_empty(context: Context) -> None:
"""Call get_history_count with empty string."""
try:
context.controller.get_history_count("")
context.error = None
except ValueError as exc:
context.error = exc
@when("I clear all history")
def step_clear_all_history(context: Context) -> None:
"""Clear all history."""
context.controller.clear_history()
@then("the controller weights should have {count:d} entries")
def step_weights_count(context: Context, count: int) -> None:
"""Assert the number of weight entries."""
weights = context.controller.weights
assert len(weights) == count, f"Expected {count} weights, got {len(weights)}"
@when('I record max history outcomes for "{op}"')
def step_record_max_history(context: Context, op: str) -> None:
"""Record exactly _MAX_HISTORY_PER_TYPE outcomes to trigger eviction."""
from cleveragents.application.services.autonomy_controller import (
_MAX_HISTORY_PER_TYPE,
)
for _ in range(_MAX_HISTORY_PER_TYPE + 1):
context.controller.record_outcome(
HistoricalOutcome(operation_type=op, succeeded=True)
)
@when("I try to create a controller with extra weight keys")
def step_try_extra_weight_keys(context: Context) -> None:
"""Attempt to create controller with extra weight keys."""
try:
AutonomyController(
weights={
"past_success_rate": 0.2,
"codebase_familiarity": 0.2,
"risk_assessment": 0.2,
"invariant_complexity": 0.2,
"extra_key": 0.2,
}
)
context.error = None
except ValueError as exc:
context.error = exc
@when("I try to create a controller with negative weight")
def step_try_negative_weight(context: Context) -> None:
"""Attempt to create controller with negative weight."""
try:
AutonomyController(
weights={
"past_success_rate": -0.1,
"codebase_familiarity": 0.4,
"risk_assessment": 0.4,
"invariant_complexity": 0.3,
}
)
context.error = None
except ValueError as exc:
context.error = exc
@when("I try to create a controller with weights not summing to 1")
def step_try_bad_sum_weights(context: Context) -> None:
"""Attempt to create controller with weights not summing to 1."""
try:
AutonomyController(
weights={
"past_success_rate": 0.3,
"codebase_familiarity": 0.3,
"risk_assessment": 0.3,
"invariant_complexity": 0.3,
}
)
context.error = None
except ValueError as exc:
context.error = exc
+236
View File
@@ -0,0 +1,236 @@
"""Helper script for Robot Framework semantic escalation tests."""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.domain.models.core.automation_profile import (
BUILTIN_PROFILES,
AutomationProfile,
)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
HistoricalOutcome,
OperationContext,
)
def _test_confidence_high() -> None:
"""All positive factors produce high confidence."""
ctrl = AutonomyController()
factors = ConfidenceFactors(
past_success_rate=1.0,
codebase_familiarity=1.0,
risk_assessment=0.0,
invariant_complexity=0.0,
)
score = ctrl.compute_confidence(factors)
assert abs(score - 1.0) < 1e-6, f"Expected 1.0, got {score}"
print("confidence-high-ok")
def _test_confidence_low() -> None:
"""All negative factors produce zero confidence."""
ctrl = AutonomyController()
factors = ConfidenceFactors(
past_success_rate=0.0,
codebase_familiarity=0.0,
risk_assessment=1.0,
invariant_complexity=1.0,
)
score = ctrl.compute_confidence(factors)
assert abs(score - 0.0) < 1e-6, f"Expected 0.0, got {score}"
print("confidence-low-ok")
def _test_threshold_proceed() -> None:
"""Proceed when confidence >= threshold."""
ctrl = AutonomyController()
profile = AutomationProfile(name="test", auto_execute=0.5)
factors = ConfidenceFactors(
past_success_rate=0.9,
codebase_familiarity=0.9,
risk_assessment=0.1,
invariant_complexity=0.1,
)
op = OperationContext(operation_type="auto_execute")
decision = ctrl.should_proceed_automatically(op, factors, profile)
assert decision.proceed is True, f"Expected proceed=True, got {decision.proceed}"
assert decision.confidence >= 0.5
print("threshold-proceed-ok")
def _test_threshold_escalate() -> None:
"""Escalate when confidence < threshold."""
ctrl = AutonomyController()
profile = AutomationProfile(name="test", auto_execute=0.99)
factors = ConfidenceFactors(
past_success_rate=0.3,
codebase_familiarity=0.3,
risk_assessment=0.7,
invariant_complexity=0.7,
)
op = OperationContext(operation_type="auto_execute")
decision = ctrl.should_proceed_automatically(op, factors, profile)
assert decision.proceed is False, f"Expected proceed=False, got {decision.proceed}"
print("threshold-escalate-ok")
def _test_history_tracking() -> None:
"""Success rate updates after recording outcomes."""
ctrl = AutonomyController()
assert abs(ctrl.get_historical_success_rate("auto_execute") - 0.5) < 1e-6
for _ in range(8):
ctrl.record_outcome(
HistoricalOutcome(operation_type="auto_execute", succeeded=True)
)
for _ in range(2):
ctrl.record_outcome(
HistoricalOutcome(operation_type="auto_execute", succeeded=False)
)
rate = ctrl.get_historical_success_rate("auto_execute")
assert abs(rate - 0.8) < 1e-6, f"Expected 0.8, got {rate}"
assert ctrl.get_history_count("auto_execute") == 10
print("history-tracking-ok")
def _test_profile_manual() -> None:
"""Manual profile always escalates (with realistic confidence < 1.0)."""
ctrl = AutonomyController()
profile = BUILTIN_PROFILES["manual"]
factors = ConfidenceFactors(
past_success_rate=0.95,
codebase_familiarity=0.95,
risk_assessment=0.05,
invariant_complexity=0.05,
)
op = OperationContext(operation_type="auto_execute")
decision = ctrl.should_proceed_automatically(op, factors, profile)
assert decision.proceed is False
print("profile-manual-ok")
def _test_profile_fullauto() -> None:
"""Full-auto profile always proceeds."""
ctrl = AutonomyController()
profile = BUILTIN_PROFILES["full-auto"]
factors = ConfidenceFactors(
past_success_rate=0.0,
codebase_familiarity=0.0,
risk_assessment=1.0,
invariant_complexity=1.0,
)
op = OperationContext(operation_type="auto_execute")
decision = ctrl.should_proceed_automatically(op, factors, profile)
assert decision.proceed is True
print("profile-fullauto-ok")
def _test_profile_cautious() -> None:
"""Cautious profile uses intermediate thresholds."""
ctrl = AutonomyController()
profile = BUILTIN_PROFILES["cautious"]
# High confidence should proceed
high_factors = ConfidenceFactors(
past_success_rate=0.95,
codebase_familiarity=0.9,
risk_assessment=0.05,
invariant_complexity=0.05,
)
op = OperationContext(operation_type="auto_execute")
high_decision = ctrl.should_proceed_automatically(
op,
high_factors,
profile,
)
assert high_decision.proceed is True
# Low confidence should escalate
low_factors = ConfidenceFactors(
past_success_rate=0.3,
codebase_familiarity=0.3,
risk_assessment=0.8,
invariant_complexity=0.8,
)
low_decision = ctrl.should_proceed_automatically(
op,
low_factors,
profile,
)
assert low_decision.proceed is False
print("profile-cautious-ok")
def _test_decision_fields() -> None:
"""Decision model includes all required fields."""
ctrl = AutonomyController()
profile = AutomationProfile(name="test", auto_execute=0.5)
factors = ConfidenceFactors(
past_success_rate=0.8,
codebase_familiarity=0.7,
risk_assessment=0.2,
invariant_complexity=0.3,
)
op = OperationContext(operation_type="auto_execute")
decision = ctrl.should_proceed_automatically(op, factors, profile)
assert decision.operation_type == "auto_execute"
assert len(decision.factors) == 4
assert "past_success_rate" in decision.factors
assert abs(decision.threshold - 0.5) < 1e-6
assert isinstance(decision.explanation, str)
assert len(decision.explanation) > 0
print("decision-fields-ok")
def _test_zero_threshold() -> None:
"""Threshold 0.0 means always automatic, even zero confidence."""
ctrl = AutonomyController()
profile = AutomationProfile(name="test", auto_execute=0.0)
factors = ConfidenceFactors(
past_success_rate=0.0,
codebase_familiarity=0.0,
risk_assessment=1.0,
invariant_complexity=1.0,
)
op = OperationContext(operation_type="auto_execute")
decision = ctrl.should_proceed_automatically(op, factors, profile)
assert decision.proceed is True
assert abs(decision.confidence - 0.0) < 1e-6
print("zero-threshold-ok")
_TESTS = {
"confidence-high": _test_confidence_high,
"confidence-low": _test_confidence_low,
"threshold-proceed": _test_threshold_proceed,
"threshold-escalate": _test_threshold_escalate,
"history-tracking": _test_history_tracking,
"profile-manual": _test_profile_manual,
"profile-fullauto": _test_profile_fullauto,
"profile-cautious": _test_profile_cautious,
"decision-fields": _test_decision_fields,
"zero-threshold": _test_zero_threshold,
}
if __name__ == "__main__":
if len(sys.argv) < 2:
print(f"Usage: {sys.argv[0]} <test-name>", file=sys.stderr)
print(f"Available tests: {', '.join(sorted(_TESTS))}", file=sys.stderr)
sys.exit(1)
test_name = sys.argv[1]
if test_name not in _TESTS:
print(f"Unknown test: {test_name}", file=sys.stderr)
sys.exit(1)
_TESTS[test_name]()
+69
View File
@@ -0,0 +1,69 @@
*** Settings ***
Documentation End-to-end tests for Semantic Escalation confidence scoring
Resource ${CURDIR}/common.resource
Suite Setup Setup Test Environment
Suite Teardown Cleanup Test Environment
*** Variables ***
${HELPER_SCRIPT} robot/helper_semantic_escalation.py
*** Test Cases ***
Confidence Computation With All Factors High
[Documentation] All positive factors produce high confidence
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} confidence-high cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} confidence-high-ok
Confidence Computation With All Factors Low
[Documentation] All negative factors produce zero confidence
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} confidence-low cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} confidence-low-ok
Threshold Comparison Proceeds When Above
[Documentation] Proceed when confidence >= threshold
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} threshold-proceed cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} threshold-proceed-ok
Threshold Comparison Escalates When Below
[Documentation] Escalate when confidence < threshold
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} threshold-escalate cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} threshold-escalate-ok
Historical Success Tracking Records And Retrieves
[Documentation] Success rate updates after recording outcomes
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} history-tracking cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} history-tracking-ok
Escalation With Manual Profile Always Escalates
[Documentation] Manual profile (threshold 1.0) always escalates
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} profile-manual cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} profile-manual-ok
Escalation With FullAuto Profile Always Proceeds
[Documentation] Full-auto profile (threshold 0.0) always proceeds
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} profile-fullauto cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} profile-fullauto-ok
Escalation With Cautious Profile Uses Intermediate Thresholds
[Documentation] Cautious profile proceeds only with sufficient confidence
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} profile-cautious cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} profile-cautious-ok
EscalationDecision Model Captures All Fields
[Documentation] Decision model includes all required fields
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} decision-fields cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} decision-fields-ok
Zero Threshold Always Proceeds Even With Zero Confidence
[Documentation] Spec: threshold 0.0 means always automatic
${result}= Run Process ${PYTHON} ${HELPER_SCRIPT} zero-threshold cwd=${WORKSPACE}
Should Be Equal As Integers ${result.rc} 0
Should Contain ${result.stdout} zero-threshold-ok
@@ -11,6 +11,9 @@ from dependency_injector import containers, providers
from cleveragents.actor.registry import ActorRegistry
from cleveragents.application.reactive_registry_adapter import register_registry_agents
from cleveragents.application.services.actor_service import ActorService
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.application.services.autonomy_guardrail_service import (
AutonomyGuardrailService,
)
@@ -366,6 +369,11 @@ class Container(containers.DeclarativeContainer):
AutonomyGuardrailService,
)
# Autonomy Controller — Semantic Escalation (spec § Semantic Escalation)
autonomy_controller = providers.Singleton(
AutonomyController,
)
# Execution Environment Resolver - Singleton (stateless)
execution_environment_resolver = providers.Singleton(
ExecutionEnvironmentResolver,
@@ -3,6 +3,9 @@
Contains service classes that orchestrate business operations.
"""
from cleveragents.application.services.autonomy_controller import (
AutonomyController,
)
from cleveragents.application.services.autonomy_guardrail_service import (
AutonomyGuardrailService,
)
@@ -141,6 +144,7 @@ __all__ = [
"ApplyValidationResult",
"ApplyValidationSummary",
"AttachmentScope",
"AutonomyController",
"AutonomyGuardrailService",
"BrokenReferenceRule",
"ClusterStrategy",
@@ -0,0 +1,349 @@
"""Autonomy Controller — Semantic Escalation engine.
Implements the ``AutonomyController`` class described in the
specification (§ Automation & Safety > Semantic Escalation, lines
28176--28206). Computes a confidence score from multiple weighted
factors and compares it against the active automation profile's
threshold to decide whether to proceed automatically or escalate
to the user.
Thread-safe: all mutable state (historical outcomes) is protected
by a reentrant lock.
"""
from __future__ import annotations
import logging
import threading
from collections import defaultdict
from cleveragents.domain.models.core.automation_profile import (
AutomationProfile,
)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
EscalationDecision,
HistoricalOutcome,
OperationContext,
)
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Default factor weights (sum to 1.0)
# ---------------------------------------------------------------------------
_DEFAULT_WEIGHTS: dict[str, float] = {
"past_success_rate": 0.30,
"codebase_familiarity": 0.20,
"risk_assessment": 0.30,
"invariant_complexity": 0.20,
}
# Hard ceiling for historical outcomes per operation type to prevent
# unbounded memory growth.
_MAX_HISTORY_PER_TYPE = 10_000
class AutonomyController:
"""Semantic Escalation controller.
Computes a confidence score from weighted factors and compares
it to the automation profile threshold. Records outcomes of
autonomous decisions for future ``past_success_rate`` computation.
Spec reference: ``docs/specification.md`` lines 28176--28206.
"""
def __init__(
self,
weights: dict[str, float] | None = None,
) -> None:
"""Initialise the controller.
Args:
weights: Optional mapping of factor name to weight.
Must sum to 1.0. Falls back to built-in defaults
when ``None``.
Raises:
ValueError: If *weights* keys are invalid or do not
sum to 1.0.
"""
resolved = dict(weights) if weights is not None else dict(_DEFAULT_WEIGHTS)
self._validate_weights(resolved)
self._weights: dict[str, float] = resolved
# operation_type -> list of booleans (True = success)
self._history: dict[str, list[bool]] = defaultdict(list)
self._lock = threading.RLock()
# ------------------------------------------------------------------
# Public API
# ------------------------------------------------------------------
def should_proceed_automatically(
self,
operation: OperationContext,
factors: ConfidenceFactors,
profile: AutomationProfile,
) -> EscalationDecision:
"""Determine whether to proceed automatically or escalate.
Args:
operation: Context about the operation being evaluated.
factors: Individual confidence factors.
profile: The active automation profile.
Returns:
An ``EscalationDecision`` capturing the outcome.
Raises:
ValueError: If any argument is ``None`` or invalid.
"""
if operation is None:
raise ValueError("operation must not be None")
if factors is None:
raise ValueError("factors must not be None")
if profile is None:
raise ValueError("profile must not be None")
threshold = self._get_threshold(profile, operation.operation_type)
confidence = self.compute_confidence(factors)
proceed = confidence >= threshold
factor_dict = {
"past_success_rate": factors.past_success_rate,
"codebase_familiarity": factors.codebase_familiarity,
"risk_assessment": factors.risk_assessment,
"invariant_complexity": factors.invariant_complexity,
}
explanation = self._build_explanation(
proceed=proceed,
confidence=confidence,
threshold=threshold,
operation_type=operation.operation_type,
factor_dict=factor_dict,
)
decision = EscalationDecision(
proceed=proceed,
confidence=confidence,
factors=factor_dict,
explanation=explanation,
threshold=threshold,
operation_type=operation.operation_type,
)
logger.info(
"Escalation decision for %s: proceed=%s confidence=%.3f threshold=%.3f",
operation.operation_type,
proceed,
confidence,
threshold,
)
return decision
def compute_confidence(
self,
factors: ConfidenceFactors,
) -> float:
"""Compute a weighted confidence score from factors.
Risk and invariant complexity are *inverted*: higher values
reduce the confidence score.
Args:
factors: The confidence factors.
Returns:
A float in [0.0, 1.0].
Raises:
ValueError: If *factors* is ``None``.
"""
if factors is None:
raise ValueError("factors must not be None")
score = (
self._weights["past_success_rate"] * factors.past_success_rate
+ self._weights["codebase_familiarity"] * factors.codebase_familiarity
+ self._weights["risk_assessment"] * (1.0 - factors.risk_assessment)
+ self._weights["invariant_complexity"]
* (1.0 - factors.invariant_complexity)
)
return max(0.0, min(1.0, score))
def record_outcome(self, outcome: HistoricalOutcome) -> None:
"""Record the outcome of an autonomous decision.
Args:
outcome: The outcome to record.
Raises:
ValueError: If *outcome* is ``None``.
"""
if outcome is None:
raise ValueError("outcome must not be None")
with self._lock:
history = self._history[outcome.operation_type]
if len(history) >= _MAX_HISTORY_PER_TYPE:
history.pop(0)
history.append(outcome.succeeded)
logger.debug(
"Recorded outcome for %s: %s",
outcome.operation_type,
"success" if outcome.succeeded else "failure",
)
def get_historical_success_rate(
self,
operation_type: str,
) -> float:
"""Return the historical success rate for an operation type.
Returns 0.5 (neutral) when no history exists.
Args:
operation_type: The operation type to query.
Returns:
A float in [0.0, 1.0].
Raises:
ValueError: If *operation_type* is empty.
"""
if not operation_type:
raise ValueError("operation_type must not be empty")
with self._lock:
history = self._history.get(operation_type)
if not history:
return 0.5
return sum(1 for h in history if h) / len(history)
def get_history_count(self, operation_type: str) -> int:
"""Return the number of recorded outcomes for an operation type.
Args:
operation_type: The operation type to query.
Returns:
The number of recorded outcomes.
Raises:
ValueError: If *operation_type* is empty.
"""
if not operation_type:
raise ValueError("operation_type must not be empty")
with self._lock:
return len(self._history.get(operation_type, []))
def clear_history(self, operation_type: str | None = None) -> None:
"""Clear historical outcomes.
Args:
operation_type: If given, clear only that type.
Otherwise clear all history.
"""
with self._lock:
if operation_type is not None:
self._history.pop(operation_type, None)
else:
self._history.clear()
@property
def weights(self) -> dict[str, float]:
"""Return a copy of the current factor weights."""
return dict(self._weights)
# ------------------------------------------------------------------
# Private helpers
# ------------------------------------------------------------------
@staticmethod
def _validate_weights(weights: dict[str, float]) -> None:
"""Validate that weights cover all factors and sum to 1.0."""
required_keys = {
"past_success_rate",
"codebase_familiarity",
"risk_assessment",
"invariant_complexity",
}
if set(weights.keys()) != required_keys:
missing = required_keys - set(weights.keys())
extra = set(weights.keys()) - required_keys
parts: list[str] = []
if missing:
parts.append(f"missing: {sorted(missing)}")
if extra:
parts.append(f"extra: {sorted(extra)}")
raise ValueError(
f"weights must contain exactly {sorted(required_keys)}: "
+ ", ".join(parts)
)
for key, value in weights.items():
if value < 0.0:
raise ValueError(f"Weight for '{key}' must be >= 0.0, got {value}")
total = sum(weights.values())
if abs(total - 1.0) > 1e-6:
raise ValueError(f"Weights must sum to 1.0, got {total:.6f}")
@staticmethod
def _get_threshold(
profile: AutomationProfile,
operation_type: str,
) -> float:
"""Look up the threshold for an operation type on a profile.
Falls back to 1.0 (always manual) if the operation type
is not a recognised threshold field.
Args:
profile: The automation profile.
operation_type: The flag name (e.g. ``auto_execute``).
Returns:
The threshold float in [0.0, 1.0].
"""
if hasattr(profile, operation_type):
value = getattr(profile, operation_type)
if isinstance(value, (int, float)):
return float(value)
return 1.0
@staticmethod
def _build_explanation(
*,
proceed: bool,
confidence: float,
threshold: float,
operation_type: str,
factor_dict: dict[str, float],
) -> str:
"""Build a human-readable explanation of the decision."""
action = "Proceeding automatically" if proceed else "Escalating to user"
factor_parts = [
f"{name}={value:.2f}" for name, value in sorted(factor_dict.items())
]
factors_str = ", ".join(factor_parts)
return (
f"{action} for '{operation_type}': "
f"confidence={confidence:.3f} "
f"{'>='} threshold={threshold:.3f}. "
f"Factors: {factors_str}."
if proceed
else f"{action} for '{operation_type}': "
f"confidence={confidence:.3f} < "
f"threshold={threshold:.3f}. "
f"Factors: {factors_str}."
)
@@ -97,6 +97,14 @@ from cleveragents.domain.models.core.error_recovery import (
get_recovery_hints,
)
# Semantic escalation models (spec § Semantic Escalation)
from cleveragents.domain.models.core.escalation import (
ConfidenceFactors,
EscalationDecision,
HistoricalOutcome,
OperationContext,
)
# Invariant domain models
from cleveragents.domain.models.core.invariant import (
Invariant,
@@ -281,6 +289,7 @@ __all__ = [
"CheckpointRetentionPolicy",
"CheckpointScope",
"CloudBillingFields",
"ConfidenceFactors",
"Context",
"ContextConfig",
"ContextFile",
@@ -311,12 +320,14 @@ __all__ = [
"ErrorHistory",
"ErrorRecord",
"ErrorRecoveryPolicy",
"EscalationDecision",
"ExecutionEnvironment",
"GuardResult",
"GuardrailAuditEntry",
"GuardrailAuditTrail",
"GuardrailEventType",
"GuardrailResult",
"HistoricalOutcome",
"InMemoryChangeSetStore",
"InMemoryInvocationTracker",
"Invariant",
@@ -336,6 +347,7 @@ __all__ = [
"NamespacedName",
"NamespacedProject",
"Operation",
"OperationContext",
"OperationType",
"Org",
"OrgRole",
@@ -0,0 +1,202 @@
"""Semantic Escalation domain models.
Defines the ``EscalationDecision``, ``OperationContext``, and
``ConfidenceFactors`` models used by the Semantic Escalation system
to decide whether an operation should proceed automatically or
escalate to the user for approval.
The confidence score (0.0--1.0) is compared against the active
automation profile's threshold for the relevant flag. When the
score meets or exceeds the threshold the operation proceeds
autonomously; otherwise the user is prompted.
Based on ``docs/specification.md`` Section "Semantic Escalation"
(lines 28176--28206).
"""
from __future__ import annotations
from pydantic import BaseModel, ConfigDict, Field, field_validator
class ConfidenceFactors(BaseModel):
"""Individual factors contributing to the confidence score.
Each factor is a float in [0.0, 1.0].
Attributes:
past_success_rate: Historical success rate for similar ops.
codebase_familiarity: Agent knowledge of relevant code area.
risk_assessment: Estimated risk (0.0 = no risk, 1.0 = max).
invariant_complexity: Complexity of invariants to preserve
(0.0 = trivial, 1.0 = extremely complex).
"""
past_success_rate: float = Field(
0.5,
ge=0.0,
le=1.0,
description="Historical success rate for similar operations",
)
codebase_familiarity: float = Field(
0.5,
ge=0.0,
le=1.0,
description="Agent familiarity with relevant code area",
)
risk_assessment: float = Field(
0.5,
ge=0.0,
le=1.0,
description=(
"Estimated risk of the operation (0.0 = no risk, 1.0 = maximum risk)"
),
)
invariant_complexity: float = Field(
0.5,
ge=0.0,
le=1.0,
description=(
"Complexity of invariants that must be preserved "
"(0.0 = trivial, 1.0 = extremely complex)"
),
)
@field_validator(
"past_success_rate",
"codebase_familiarity",
"risk_assessment",
"invariant_complexity",
)
@classmethod
def validate_factor_range(
cls: type[ConfidenceFactors],
v: float,
) -> float:
"""Validate factor is in [0.0, 1.0]."""
if v < 0.0 or v > 1.0:
raise ValueError(f"Factor must be between 0.0 and 1.0, got {v}")
return v
model_config = ConfigDict(
str_strip_whitespace=True,
validate_assignment=True,
)
class OperationContext(BaseModel):
"""Context for an operation being evaluated for escalation.
Attributes:
operation_type: Type/name of the operation (e.g. the
automation profile flag like ``auto_execute``).
project_id: Identifier of the project being operated on.
file_count: Number of files affected by the operation.
test_coverage: Test coverage ratio for affected code.
blast_radius: Estimated blast radius (0.0--1.0).
"""
operation_type: str = Field(
...,
min_length=1,
description="Automation profile flag name (e.g. 'auto_execute')",
)
project_id: str = Field(
default="",
description="Identifier of the project being operated on",
)
file_count: int = Field(
default=0,
ge=0,
description="Number of files affected by the operation",
)
test_coverage: float = Field(
default=0.5,
ge=0.0,
le=1.0,
description="Test coverage ratio for affected code area",
)
blast_radius: float = Field(
default=0.5,
ge=0.0,
le=1.0,
description="Estimated blast radius (0.0 = none, 1.0 = full)",
)
model_config = ConfigDict(
str_strip_whitespace=True,
validate_assignment=True,
)
class EscalationDecision(BaseModel):
"""Result of a semantic escalation check.
Returned by ``AutonomyController.should_proceed_automatically()``.
Attributes:
proceed: Whether the operation should proceed automatically.
confidence: Computed confidence score (0.0--1.0).
factors: Individual factor values that produced the score.
explanation: Human-readable explanation of the decision.
threshold: The automation profile threshold that was compared.
operation_type: The operation type that was evaluated.
"""
proceed: bool = Field(
...,
description="Whether to proceed automatically",
)
confidence: float = Field(
...,
ge=0.0,
le=1.0,
description="Computed confidence score (0.0--1.0)",
)
factors: dict[str, float] = Field(
...,
description="Individual factor values used in computation",
)
explanation: str = Field(
...,
min_length=1,
description="Human-readable explanation of the decision",
)
threshold: float = Field(
...,
ge=0.0,
le=1.0,
description="Automation profile threshold used for comparison",
)
operation_type: str = Field(
...,
min_length=1,
description="The operation type that was evaluated",
)
model_config = ConfigDict(
str_strip_whitespace=True,
)
class HistoricalOutcome(BaseModel):
"""A single recorded outcome of an autonomous decision.
Attributes:
operation_type: The operation type that was performed.
succeeded: Whether the autonomous operation succeeded.
"""
operation_type: str = Field(
...,
min_length=1,
description="The operation type that was performed",
)
succeeded: bool = Field(
...,
description="Whether the autonomous operation succeeded",
)
model_config = ConfigDict(
str_strip_whitespace=True,
)
+8
View File
@@ -721,3 +721,11 @@ get_or_create_sandbox_for_resource # noqa: B018, F821
clear_boundary_cache # noqa: B018, F821
boundary_cache_size # noqa: B018, F821
_MAX_DAG_DEPTH # noqa: B018, F821
# Semantic Escalation — public API used by AutonomyController and tests
AutonomyController # noqa: B018, F821
ConfidenceFactors # noqa: B018, F821
EscalationDecision # noqa: B018, F821
HistoricalOutcome # noqa: B018, F821
OperationContext # noqa: B018, F821
autonomy_controller # noqa: B018, F821