feat(plan): add error recovery patterns and CLI hints
ISSUES CLOSED: #186
This commit is contained in:
@@ -0,0 +1,131 @@
|
||||
"""ASV benchmarks for error recovery patterns.
|
||||
|
||||
Measures the performance of:
|
||||
- Error classification
|
||||
- Recovery hint generation
|
||||
- ErrorRecord/ErrorHistory model operations
|
||||
- ErrorRecoveryService.record_error() with mock lifecycle
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
try:
|
||||
from cleveragents.application.services.error_recovery_service import (
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
ErrorHistory,
|
||||
ErrorRecord,
|
||||
ErrorRecoveryPolicy,
|
||||
RecoveryHint,
|
||||
classify_error,
|
||||
get_recovery_hints,
|
||||
)
|
||||
from cleveragents.domain.models.core.plan import (
|
||||
PlanPhase,
|
||||
PlanTimestamps,
|
||||
ProcessingState,
|
||||
)
|
||||
except ModuleNotFoundError:
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
||||
from cleveragents.application.services.error_recovery_service import (
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
ErrorHistory,
|
||||
ErrorRecord,
|
||||
ErrorRecoveryPolicy,
|
||||
RecoveryHint,
|
||||
classify_error,
|
||||
get_recovery_hints,
|
||||
)
|
||||
from cleveragents.domain.models.core.plan import (
|
||||
PlanPhase,
|
||||
PlanTimestamps,
|
||||
ProcessingState,
|
||||
)
|
||||
|
||||
|
||||
_PLAN_ID = "01BENCHERR0000000000001"
|
||||
|
||||
|
||||
def _make_plan() -> MagicMock:
|
||||
plan = MagicMock()
|
||||
plan.identity.plan_id = _PLAN_ID
|
||||
plan.phase = PlanPhase.EXECUTE
|
||||
plan.processing_state = ProcessingState.ERRORED
|
||||
plan.is_terminal = False
|
||||
plan.error_message = None
|
||||
plan.error_details = None
|
||||
plan.timestamps = PlanTimestamps()
|
||||
return plan
|
||||
|
||||
|
||||
def _make_service(plan: MagicMock) -> ErrorRecoveryService:
|
||||
lifecycle = MagicMock()
|
||||
lifecycle.get_plan.return_value = plan
|
||||
lifecycle._commit_plan = MagicMock()
|
||||
return ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=0.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
|
||||
class ClassifySuite:
|
||||
"""Benchmark error classification."""
|
||||
|
||||
def time_classify_transient(self) -> None:
|
||||
classify_error("Connection timed out after 30s")
|
||||
|
||||
def time_classify_validation(self) -> None:
|
||||
classify_error("Schema validation failed for field X")
|
||||
|
||||
def time_classify_unknown(self) -> None:
|
||||
classify_error("Something completely novel happened")
|
||||
|
||||
def time_classify_with_exception_type(self) -> None:
|
||||
classify_error("failed", "RateLimitError")
|
||||
|
||||
|
||||
class HintSuite:
|
||||
"""Benchmark recovery hint generation."""
|
||||
|
||||
def time_hints_transient(self) -> None:
|
||||
get_recovery_hints(ErrorCategory.TRANSIENT, _PLAN_ID)
|
||||
|
||||
def time_hints_validation(self) -> None:
|
||||
get_recovery_hints(ErrorCategory.VALIDATION, _PLAN_ID)
|
||||
|
||||
def time_hints_merge_conflict(self) -> None:
|
||||
get_recovery_hints(ErrorCategory.MERGE_CONFLICT, _PLAN_ID)
|
||||
|
||||
|
||||
class RecordSuite:
|
||||
"""Benchmark error recording with mock service."""
|
||||
|
||||
def setup(self) -> None:
|
||||
self.plan = _make_plan()
|
||||
self.service = _make_service(self.plan)
|
||||
self._counter = 0
|
||||
|
||||
def time_record_error(self) -> None:
|
||||
self._counter += 1
|
||||
self.service.record_error(
|
||||
plan_id=_PLAN_ID,
|
||||
phase="execute",
|
||||
message=f"Timeout #{self._counter}",
|
||||
)
|
||||
|
||||
def time_format_output_plain(self) -> None:
|
||||
self.service.format_error_output(_PLAN_ID, fmt="plain")
|
||||
|
||||
def time_format_output_json(self) -> None:
|
||||
self.service.format_error_output(_PLAN_ID, fmt="json")
|
||||
+2
-1
@@ -699,7 +699,8 @@ The system becomes more autonomous through:
|
||||
- Architecture supports: Cold tier can span projects
|
||||
|
||||
4. **Fully Autonomous Recovery Strategies**
|
||||
- Current: Rollback and retry with guidance
|
||||
- Current: Automatic retry for transient errors with structured
|
||||
recovery hints and `agents plan errors` CLI (D1b)
|
||||
- Research: Automatic error understanding and fixing
|
||||
- Architecture supports: Recovery is just another plan type
|
||||
|
||||
|
||||
@@ -0,0 +1,283 @@
|
||||
# Error Recovery Reference
|
||||
|
||||
This document describes the error recovery patterns, error classification,
|
||||
recovery hint system, and retry policy logic added in D1b.recovery.
|
||||
|
||||
## Error Categories
|
||||
|
||||
Errors are classified into categories that determine recovery behaviour:
|
||||
|
||||
| Category | Retriable | Description |
|
||||
|----------|-----------|-------------|
|
||||
| `transient` | Yes | Temporary failures (network, timeout, rate-limit) |
|
||||
| `validation` | Yes | Validation failures during Execute or Apply |
|
||||
| `configuration` | No | Bad config, missing required fields |
|
||||
| `resource` | No | Missing or conflicting resources |
|
||||
| `merge_conflict` | Yes | Sandbox merge failures |
|
||||
| `authentication` | No | Auth/authz failures |
|
||||
| `provider` | Yes | LLM provider errors (model unavailable, token limit) |
|
||||
| `internal` | No | Unexpected internal errors (bugs) |
|
||||
| `unknown` | No | Unclassified errors |
|
||||
|
||||
## Recovery Actions
|
||||
|
||||
Each error category maps to one or more recovery actions with CLI commands:
|
||||
|
||||
| Action | CLI Command | When Used |
|
||||
|--------|-------------|-----------|
|
||||
| `retry` | `agents plan prompt <plan_id>` | Transient / provider errors |
|
||||
| `correct` | `agents plan correct <plan_id>` | Validation failures |
|
||||
| `revert_strategize` | `agents plan revert <plan_id> --to-phase strategize` | Resource / merge conflicts |
|
||||
| `prompt` | `agents plan prompt <plan_id>` | Resume with guidance |
|
||||
| `cancel` | `agents plan cancel <plan_id>` | Non-recoverable errors |
|
||||
| `manual` | — | Internal errors (check logs) |
|
||||
|
||||
## Retry Policy
|
||||
|
||||
The retry policy is controlled by the automation profile's
|
||||
`auto_retry_transient` threshold:
|
||||
|
||||
- **0.0** (auto): Automatic retry for retriable errors
|
||||
- **1.0** (human): Always escalate to human
|
||||
|
||||
```
|
||||
Error occurs → classify_error(message, exception_type)
|
||||
|
|
||||
+→ retriable category? → retry_count < max_retries? → threshold < 1.0?
|
||||
| | | |
|
||||
| No → escalate No → exhausted, escalate No → escalate
|
||||
| |
|
||||
| Yes → auto retry
|
||||
|
|
||||
+→ non-retriable → escalate immediately
|
||||
```
|
||||
|
||||
### Retry Limits
|
||||
|
||||
Default `max_retries = 3`. Configured via `SubplanConfig.max_retries`
|
||||
for subplans (max: 5).
|
||||
|
||||
### Retry Exhaustion
|
||||
|
||||
When all retry attempts are used, the error is escalated with:
|
||||
- Clear message indicating retries exhausted
|
||||
- Recovery hints for manual intervention
|
||||
- Error history preserved for debugging
|
||||
|
||||
## Structured Error Metadata
|
||||
|
||||
When an error is recorded, the following metadata is stored in the
|
||||
plan's `error_details` field:
|
||||
|
||||
| Key | Type | Description |
|
||||
|-----|------|-------------|
|
||||
| `error_category` | string | Classification category |
|
||||
| `error_phase` | string | Phase when error occurred |
|
||||
| `retry_count` | string | Number of retries attempted |
|
||||
| `max_retries` | string | Maximum retries allowed |
|
||||
| `is_retriable` | string | Whether error can be retried |
|
||||
| `error_actor` | string | Actor involved (if any) |
|
||||
| `error_tool_call` | string | Tool call that failed (if any) |
|
||||
| `stack_summary` | string | Abbreviated stack trace (last 3 frames) |
|
||||
| `recovery_action` | string | Suggested recovery action |
|
||||
| `recovery_hint` | string | Human-readable recovery message |
|
||||
| `recovery_command` | string | CLI command for recovery |
|
||||
|
||||
## Execute Phase Integration
|
||||
|
||||
The `PlanExecutor` accepts an optional `ErrorRecoveryService` and uses
|
||||
it during the Execute phase to record errors and drive automatic
|
||||
retries:
|
||||
|
||||
```
|
||||
PlanExecutor
|
||||
├── StrategizeStubActor (records errors via ErrorRecoveryService)
|
||||
├── ExecuteStubActor (retry loop driven by ErrorRecoveryService)
|
||||
├── PlanLifecycleService (phase transitions, persistence)
|
||||
└── ErrorRecoveryService (optional — error recording + retry policy)
|
||||
```
|
||||
|
||||
### Retry Loop
|
||||
|
||||
When `error_recovery_service` is provided, `run_execute()` wraps the
|
||||
execute actor in a retry loop:
|
||||
|
||||
1. The actor runs.
|
||||
2. On failure the error is recorded via `ErrorRecoveryService.record_error()`.
|
||||
3. `should_retry()` is checked — if the error is retriable and retries
|
||||
remain, the actor runs again.
|
||||
4. If retries are exhausted or the error is non-retriable, the plan
|
||||
transitions to `ERRORED` via `fail_execute()`.
|
||||
|
||||
Strategize failures are also recorded through the service (but not
|
||||
retried — the strategize phase is read-only and failures are
|
||||
deterministic).
|
||||
|
||||
```python
|
||||
from cleveragents.application.services.plan_executor import PlanExecutor
|
||||
from cleveragents.application.services.error_recovery_service import (
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
|
||||
er_service = ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=0.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
executor = PlanExecutor(
|
||||
lifecycle_service=lifecycle,
|
||||
tool_runner=runner,
|
||||
error_recovery_service=er_service,
|
||||
)
|
||||
|
||||
# Execute with automatic retry on transient errors
|
||||
result = executor.run_execute(plan_id)
|
||||
```
|
||||
|
||||
## CLI Commands
|
||||
|
||||
### `agents plan errors <plan_id>`
|
||||
|
||||
Shows error decisions with recovery hints and retry history for a plan.
|
||||
|
||||
```bash
|
||||
# Rich (default) output
|
||||
agents plan errors 01HXYZ...
|
||||
|
||||
# JSON output
|
||||
agents plan errors 01HXYZ... --format json
|
||||
|
||||
# Plain text
|
||||
agents plan errors 01HXYZ... --format plain
|
||||
```
|
||||
|
||||
Example rich output:
|
||||
|
||||
```
|
||||
╭─ Plan Errors ──────────────────────────────────────────╮
|
||||
│ Plan: 01HXYZ... │
|
||||
│ Phase: execute │
|
||||
│ State: errored │
|
||||
│ Error: RuntimeError: Connection timed out │
|
||||
│ │
|
||||
│ Error Category: transient │
|
||||
│ Error Phase: execute │
|
||||
│ Retry Count: 3/3 │
|
||||
│ Retriable: false │
|
||||
│ │
|
||||
│ Recovery Suggestions: │
|
||||
│ → Retry the operation — the failure was transient. │
|
||||
│ $ agents plan prompt 01HXYZ... │
|
||||
╰────────────────────────────────────────────────────────╯
|
||||
```
|
||||
|
||||
When no errors have been recorded the command shows a green
|
||||
confirmation message instead.
|
||||
|
||||
## CLI Output
|
||||
|
||||
Recovery hints are included in CLI error output:
|
||||
|
||||
```
|
||||
Error [transient]: Connection timed out after 30s
|
||||
Phase: execute
|
||||
Retry: 0/3
|
||||
Recovery suggestions:
|
||||
→ Retry the operation — the failure was transient.
|
||||
$ agents plan prompt 01HXYZ...
|
||||
✓ Automatic retry available.
|
||||
```
|
||||
|
||||
### Error History
|
||||
|
||||
The error history tracks all errors for a plan, including:
|
||||
- Total error count
|
||||
- Retry count per error
|
||||
- Category breakdown
|
||||
- Chronological error records with recovery hints
|
||||
|
||||
> **Note:** Error history is held in-memory within the
|
||||
> `ErrorRecoveryService` instance. If the service is restarted or a
|
||||
> new instance is created, in-memory history is lost. The latest
|
||||
> error metadata is also persisted to the plan's `error_details`
|
||||
> field for durability across restarts.
|
||||
|
||||
## Service API
|
||||
|
||||
### `ErrorRecoveryService`
|
||||
|
||||
```python
|
||||
from cleveragents.application.services.error_recovery_service import (
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
|
||||
service = ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=0.0, # from automation profile
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
# Record an error
|
||||
record = service.record_error(
|
||||
plan_id="01HXYZ...",
|
||||
phase="execute",
|
||||
message="Connection timed out",
|
||||
exception=exc, # optional, for stack trace
|
||||
)
|
||||
|
||||
# Check retry decision
|
||||
if service.should_retry(plan_id):
|
||||
# auto-retry
|
||||
...
|
||||
elif service.should_escalate(plan_id):
|
||||
# show recovery hints to user
|
||||
hints = service.format_recovery_hints_for_cli(plan_id)
|
||||
print(hints)
|
||||
|
||||
# Get full error history
|
||||
history = service.get_error_history(plan_id)
|
||||
print(f"Total errors: {history.total_errors}")
|
||||
|
||||
# Format for CLI
|
||||
output = service.format_error_output(plan_id, fmt="json")
|
||||
```
|
||||
|
||||
### `PlanLifecycleService.update_error_details`
|
||||
|
||||
External services use `update_error_details()` to merge structured
|
||||
error metadata into a plan's `error_details` field without accessing
|
||||
internal persistence methods:
|
||||
|
||||
```python
|
||||
lifecycle.update_error_details(plan_id, {
|
||||
"error_category": "transient",
|
||||
"retry_count": "2",
|
||||
})
|
||||
```
|
||||
|
||||
### Domain Models
|
||||
|
||||
```python
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
ErrorRecord,
|
||||
ErrorHistory,
|
||||
ErrorRecoveryPolicy,
|
||||
RecoveryAction,
|
||||
RecoveryHint,
|
||||
classify_error,
|
||||
get_recovery_hints,
|
||||
)
|
||||
|
||||
# Classify an error
|
||||
category = classify_error("Connection timed out", "NetworkError")
|
||||
|
||||
# Get recovery hints
|
||||
hints = get_recovery_hints(ErrorCategory.TRANSIENT, "01HXYZ...")
|
||||
|
||||
# Create a policy
|
||||
policy = ErrorRecoveryPolicy(auto_retry_threshold=0.0, max_retries=3)
|
||||
if policy.should_retry(record):
|
||||
...
|
||||
```
|
||||
@@ -101,9 +101,13 @@ When a sandbox merge fails during apply:
|
||||
|
||||
**Recovery steps:**
|
||||
|
||||
1. Review conflict details via `agents plan status <id>`
|
||||
2. Re-run execute phase after resolving conflicts: `agents plan execute <id>`
|
||||
3. Or fix validation issues and retry apply
|
||||
1. Review error details and recovery hints via `agents plan errors <id>`
|
||||
2. Review conflict details via `agents plan status <id>`
|
||||
3. Re-run execute phase after resolving conflicts: `agents plan execute <id>`
|
||||
4. Or fix validation issues and retry apply
|
||||
|
||||
See [Error Recovery Reference](error_recovery.md) for the full error
|
||||
classification, retry policy, and recovery hint system.
|
||||
|
||||
### Processing State Flow
|
||||
|
||||
|
||||
@@ -19,7 +19,8 @@ PlanExecutor
|
||||
├── ExecuteStubActor (legacy stub: sandbox + ChangeSetCapture)
|
||||
├── RuntimeExecuteActor (runtime: ToolRunner + ChangeSetStore)
|
||||
├── PlanExecutionContext (plan metadata + resource bindings)
|
||||
└── PlanLifecycleService (phase transitions, persistence)
|
||||
├── PlanLifecycleService (phase transitions, persistence)
|
||||
└── ErrorRecoveryService (optional — error recording + retry logic)
|
||||
```
|
||||
|
||||
## Execution Modes
|
||||
@@ -156,14 +157,115 @@ implementations will be added in D1 milestone.
|
||||
| Execute | RuntimeExecuteActor | Runtime | ChangeSet via ChangeSetStore |
|
||||
| Execute | ExecuteStubActor | Stub | ChangeSet via ChangeSetCapture |
|
||||
|
||||
## Strategize Phase
|
||||
|
||||
The strategize phase is **read-only**: it produces a decision tree from the
|
||||
action's `definition_of_done` without modifying any resources.
|
||||
|
||||
### Decision Tree
|
||||
|
||||
The stub actor parses `definition_of_done` into discrete steps, each
|
||||
represented as a `StrategyDecision` node with a ULID identifier:
|
||||
|
||||
```python
|
||||
from cleveragents.application.services.plan_executor import (
|
||||
PlanExecutor,
|
||||
StrategizeResult,
|
||||
)
|
||||
|
||||
executor = PlanExecutor(
|
||||
lifecycle_service=lifecycle,
|
||||
tool_runner=runner,
|
||||
execution_context=ctx, # optional — enables runtime mode
|
||||
error_recovery_service=er_service, # optional — enables retry logic
|
||||
)
|
||||
result: StrategizeResult = executor.run_strategize(plan_id)
|
||||
|
||||
# result.decision_root_id -> ULID of root node
|
||||
# result.decisions -> list[StrategyDecision]
|
||||
# result.invariant_records -> list[dict] (stub enforcement records)
|
||||
```
|
||||
|
||||
### Invariant Propagation
|
||||
|
||||
Project and action invariants are propagated into the strategize context.
|
||||
In M1, enforcement is stubbed (all invariants are accepted). Full
|
||||
reconciliation via the Invariant Reconciliation Actor lands in D2.
|
||||
|
||||
## Execute Phase
|
||||
|
||||
The execute phase uses sandbox resources with tool calls routed through
|
||||
`ToolRunner` and captured by `ChangeSetCapture`.
|
||||
|
||||
### ChangeSet Capture
|
||||
|
||||
All tool mutations during execute are recorded in a `ChangeSet`:
|
||||
|
||||
```python
|
||||
result: ExecuteResult = executor.run_execute(plan_id)
|
||||
|
||||
# result.changeset_id -> ULID of the changeset
|
||||
# result.changeset -> ChangeSet with entries
|
||||
# result.tool_calls_count -> int
|
||||
# result.sandbox_refs -> list[str]
|
||||
```
|
||||
|
||||
### Metadata Persistence
|
||||
|
||||
After execute completes, the following metadata is persisted on the Plan:
|
||||
|
||||
- `changeset_id`: The ChangeSet identifier
|
||||
- `sandbox_refs`: List of sandbox reference paths
|
||||
- `error_details`: Tool call count and sandbox ref count
|
||||
|
||||
## Phase Guards
|
||||
|
||||
- **Execute requires Strategize COMPLETE**: The executor validates that the
|
||||
plan has completed strategize (has a `decision_root_id`) before allowing
|
||||
execute to proceed.
|
||||
- **Phase validation**: Both `run_strategize()` and `run_execute()` verify
|
||||
the plan is in the correct phase before proceeding.
|
||||
|
||||
## Error Handling
|
||||
|
||||
Failures in either phase are captured with full error context:
|
||||
|
||||
- `error_message`: The exception message string
|
||||
- `error_details`: Dict with `exception_type`, `traceback`, and `mode`
|
||||
- `error_details`: Dict with `exception_type`, `traceback`, `mode`, and
|
||||
structured error-recovery metadata (category, retry count, recovery
|
||||
hints) when an `ErrorRecoveryService` is attached
|
||||
- The plan transitions to `ERRORED` processing state
|
||||
|
||||
### Automatic Retry (D1b)
|
||||
|
||||
When an `ErrorRecoveryService` is provided to the executor, the
|
||||
**Execute** phase (stub mode) includes an automatic retry loop:
|
||||
|
||||
1. On failure the error is classified (transient, validation, etc.)
|
||||
and recorded via `ErrorRecoveryService.record_error()`.
|
||||
2. If the error is retriable and retry attempts remain, the execute
|
||||
actor is re-invoked automatically.
|
||||
3. If retries are exhausted or the error is non-retriable, the plan
|
||||
transitions to `ERRORED` with full recovery hints stored in
|
||||
`error_details`.
|
||||
|
||||
Retry limits default to 3 and are controlled by the automation
|
||||
profile's `auto_retry_transient` threshold.
|
||||
|
||||
Use `agents plan errors <plan_id>` to inspect error decisions,
|
||||
retry history, and recovery suggestions for a plan.
|
||||
|
||||
See [Error Recovery Reference](error_recovery.md) for full details.
|
||||
|
||||
### Manual Recovery
|
||||
|
||||
Plans in `ERRORED` state without automatic retry can be recovered by:
|
||||
|
||||
1. Reviewing errors via `agents plan errors <plan_id>`
|
||||
2. Resetting the plan's processing state back to `QUEUED`
|
||||
3. Re-running the failed phase via the executor
|
||||
|
||||
|
||||
## Streaming Hooks
|
||||
|
||||
Both phases accept an optional `stream_callback`:
|
||||
@@ -193,3 +295,7 @@ Both phases accept an optional `stream_callback`:
|
||||
- `StrategizeStubActor`: Local-only strategize actor
|
||||
- `ExecuteStubActor`: Local-only execute actor
|
||||
- `StrategyDecision`, `StrategizeResult`, `ExecuteResult`: Data models
|
||||
- `StreamCallback`: Type alias for streaming callbacks
|
||||
- **`cleveragents.application.services.error_recovery_service`**: Error
|
||||
recovery integration (optional)
|
||||
- See [Error Recovery Reference](error_recovery.md)
|
||||
|
||||
@@ -0,0 +1,331 @@
|
||||
Feature: Error recovery patterns and CLI hints
|
||||
As a plan executor
|
||||
I want structured error recovery with retry logic and actionable hints
|
||||
So that failed plans can be recovered automatically or with clear guidance
|
||||
|
||||
Background:
|
||||
Given an error recovery test environment
|
||||
|
||||
# -- Error classification ------------------------------------------------
|
||||
|
||||
Scenario: Classify transient error by message
|
||||
When I classify the error message "Connection timed out after 30s"
|
||||
Then the classified error category should be "transient"
|
||||
|
||||
Scenario: Classify validation error by message
|
||||
When I classify the error message "Schema validation failed for field X"
|
||||
Then the classified error category should be "validation"
|
||||
|
||||
Scenario: Classify configuration error by message
|
||||
When I classify the error message "Missing configuration for provider"
|
||||
Then the classified error category should be "configuration"
|
||||
|
||||
Scenario: Classify merge conflict error by message
|
||||
When I classify the error message "Merge conflict in src/main.py"
|
||||
Then the classified error category should be "merge_conflict"
|
||||
|
||||
Scenario: Classify authentication error by message
|
||||
When I classify the error message "Authentication failed: invalid token"
|
||||
Then the classified error category should be "authentication"
|
||||
|
||||
Scenario: Classify provider error by message
|
||||
When I classify the error message "Model not available: gpt-4-turbo"
|
||||
Then the classified error category should be "provider"
|
||||
|
||||
Scenario: Classify internal error by message
|
||||
When I classify the error message "Unexpected internal error in pipeline"
|
||||
Then the classified error category should be "internal"
|
||||
|
||||
Scenario: Classify unknown error by message
|
||||
When I classify the error message "Something completely novel happened"
|
||||
Then the classified error category should be "unknown"
|
||||
|
||||
Scenario: Classify error by exception type
|
||||
When I classify error text "failed" using exception type "RateLimitError"
|
||||
Then the classified error category should be "transient"
|
||||
|
||||
Scenario: Classify error by exception type overrides message
|
||||
When I classify error text "validation issue" using exception type "NetworkError"
|
||||
Then the classified error category should be "transient"
|
||||
|
||||
# -- Recovery hints ------------------------------------------------------
|
||||
|
||||
Scenario: Transient error recovery hints include retry command
|
||||
When I get recovery hints for category "transient" and plan "01PLAN001"
|
||||
Then the first recovery hint action should be "retry"
|
||||
And the first recovery hint should have CLI command containing "01PLAN001"
|
||||
|
||||
Scenario: Validation error recovery hints include correct command
|
||||
When I get recovery hints for category "validation" and plan "01PLAN001"
|
||||
Then the first recovery hint action should be "correct"
|
||||
And the first recovery hint should have CLI command containing "correct"
|
||||
|
||||
Scenario: Configuration error recovery hints suggest cancel
|
||||
When I get recovery hints for category "configuration" and plan "01PLAN001"
|
||||
Then the first recovery hint action should be "cancel"
|
||||
|
||||
Scenario: Merge conflict hints suggest revert to strategize
|
||||
When I get recovery hints for category "merge_conflict" and plan "01PLAN001"
|
||||
Then the first recovery hint action should be "revert_strategize"
|
||||
|
||||
# -- Error recording -----------------------------------------------------
|
||||
|
||||
Scenario: Record error creates error record with metadata
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with phase "execute" and message "Timeout talking to provider"
|
||||
Then the recorded error should have category "transient"
|
||||
And the recorded error should have phase "execute"
|
||||
And the recorded error should have recovery hints
|
||||
And the error history should have 1 record
|
||||
|
||||
Scenario: Record error with explicit category
|
||||
Given a plan in errored state for error recovery
|
||||
When I record a categorized error with phase "apply" and message "custom issue" and category "merge_conflict"
|
||||
Then the recorded error should have category "merge_conflict"
|
||||
|
||||
Scenario: Record multiple errors tracks retry count
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with phase "execute" and message "Timeout 1"
|
||||
And I record an error with phase "execute" and message "Timeout 2"
|
||||
Then the error history should have 2 records
|
||||
And the latest error retry count should be 1
|
||||
|
||||
# -- Retry policy --------------------------------------------------------
|
||||
|
||||
Scenario: Transient error with auto-retry threshold 0 is retriable
|
||||
Given a plan in errored state for error recovery
|
||||
And the auto retry threshold is 0.0
|
||||
When I record an error with phase "execute" and message "Connection timed out"
|
||||
Then the service should recommend retry
|
||||
|
||||
Scenario: Transient error with auto-retry threshold 1 requires escalation
|
||||
Given a plan in errored state for error recovery
|
||||
And the auto retry threshold is 1.0
|
||||
When I record an error with phase "execute" and message "Connection timed out"
|
||||
Then the service should recommend escalation
|
||||
|
||||
Scenario: Configuration error is never retriable
|
||||
Given a plan in errored state for error recovery
|
||||
And the auto retry threshold is 0.0
|
||||
When I record an error with phase "strategize" and message "Missing configuration"
|
||||
Then the service should recommend escalation
|
||||
|
||||
Scenario: Retry exhaustion after max retries
|
||||
Given a plan in errored state for error recovery
|
||||
And the max retries is 2
|
||||
When I record an error with phase "execute" and message "Timeout 1"
|
||||
And I record an error with phase "execute" and message "Timeout 2"
|
||||
And I record an error with phase "execute" and message "Timeout 3"
|
||||
Then the latest error should have retries exhausted
|
||||
And the service should recommend escalation
|
||||
|
||||
# -- Error history -------------------------------------------------------
|
||||
|
||||
Scenario: Error history shows category breakdown
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with phase "execute" and message "Timeout occurred"
|
||||
And I record an error with phase "apply" and message "Merge conflict in file.py"
|
||||
Then the error history category breakdown should include "transient"
|
||||
And the error history category breakdown should include "merge_conflict"
|
||||
|
||||
# -- CLI output formatting -----------------------------------------------
|
||||
|
||||
Scenario: Format error output as plain text
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with phase "execute" and message "Provider timed out"
|
||||
And I format the error output as "plain"
|
||||
Then the formatted output should contain "transient"
|
||||
And the formatted output should contain "Provider timed out"
|
||||
And the formatted output should contain "execute"
|
||||
|
||||
Scenario: Format error output as JSON
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with phase "execute" and message "Provider timed out"
|
||||
And I format the error output as "json"
|
||||
Then the formatted output should contain "transient"
|
||||
And the formatted output should contain "plan_id"
|
||||
|
||||
Scenario: Format empty error history
|
||||
When I format error output as "plain" for separate plan "01EMPTYPLAN00000000001"
|
||||
Then the formatted output should contain "No errors recorded"
|
||||
|
||||
# -- ErrorRecord model checks --------------------------------------------
|
||||
|
||||
Scenario: ErrorRecord is_retriable for transient errors
|
||||
Given an error record with category "transient" and retry count 0 and max retries 3
|
||||
Then the error record should be retriable
|
||||
|
||||
Scenario: ErrorRecord is not retriable when exhausted
|
||||
Given an error record with category "transient" and retry count 3 and max retries 3
|
||||
Then the error record should not be retriable
|
||||
And the error record should have retries exhausted
|
||||
|
||||
Scenario: ErrorRecord is not retriable for configuration errors
|
||||
Given an error record with category "configuration" and retry count 0 and max retries 3
|
||||
Then the error record should not be retriable
|
||||
|
||||
Scenario: ErrorRecord to_cli_dict has expected fields
|
||||
Given an error recovery test environment
|
||||
Given an error record with category "transient" and retry count 1 and max retries 3
|
||||
Then the error record CLI dict should have key "error_id"
|
||||
And the error record CLI dict should have key "category"
|
||||
And the error record CLI dict should have key "is_retriable"
|
||||
|
||||
# -- Exception-based recording -------------------------------------------
|
||||
|
||||
Scenario: Record error with actual exception captures stack trace
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with exception for phase "execute" and message "timeout"
|
||||
Then the recorded error should have a stack summary
|
||||
|
||||
# -- Edge cases -----------------------------------------------------------
|
||||
|
||||
Scenario: should_retry returns false when no errors exist
|
||||
Given a plan in errored state for error recovery
|
||||
Then the service should not recommend retry for a clean plan
|
||||
|
||||
Scenario: should_escalate returns false when no errors exist
|
||||
Given a plan in errored state for error recovery
|
||||
Then the service should not recommend escalation for a clean plan
|
||||
|
||||
Scenario: Format recovery hints for CLI
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with phase "execute" and message "Connection timed out"
|
||||
Then the CLI recovery hints should contain "transient"
|
||||
|
||||
# -- ErrorRecoveryPolicy formatting --------------------------------------
|
||||
|
||||
Scenario: Policy formats output with actor and tool info
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an actor-error in phase "execute" with message "Timeout" actor "test-actor" tool "test-tool"
|
||||
And I format the error output as "plain"
|
||||
Then the formatted output should contain "test-actor"
|
||||
And the formatted output should contain "test-tool"
|
||||
|
||||
Scenario: Policy format shows automatic retry available
|
||||
Given a plan in errored state for error recovery
|
||||
And the auto retry threshold is 0.0
|
||||
When I record an error with phase "execute" and message "Connection timed out"
|
||||
And I format the error output as "plain"
|
||||
Then the formatted output should contain "Automatic retry available"
|
||||
|
||||
Scenario: Policy format shows human intervention required
|
||||
Given a plan in errored state for error recovery
|
||||
And the auto retry threshold is 1.0
|
||||
When I record an error with phase "execute" and message "Connection timed out"
|
||||
And I format the error output as "plain"
|
||||
Then the formatted output should contain "Human intervention required"
|
||||
|
||||
Scenario: Policy format shows retries exhausted message
|
||||
Given a plan in errored state for error recovery
|
||||
And the max retries is 1
|
||||
When I record an error with phase "execute" and message "Timeout 1"
|
||||
And I record an error with phase "execute" and message "Timeout 2"
|
||||
And I format the error output as "plain"
|
||||
Then the formatted output should contain "retry attempts exhausted"
|
||||
|
||||
# -- ErrorHistory with no retriable errors --------------------------------
|
||||
|
||||
Scenario: Error history has_retriable is false when only non-retriable
|
||||
Given a plan in errored state for error recovery
|
||||
When I record an error with phase "strategize" and message "Missing configuration"
|
||||
Then the error history should not have retriable errors
|
||||
|
||||
# -- ErrorRecord with actor and tool for CLI dict -------------------------
|
||||
|
||||
Scenario: ErrorRecord to_cli_dict includes actor and tool when present
|
||||
Given an error record with actor "my-actor" and tool "my-tool"
|
||||
Then the error record CLI dict should have key "actor"
|
||||
And the error record CLI dict should have key "tool_call"
|
||||
|
||||
# -- Recovery hints for empty plan ----------------------------------------
|
||||
|
||||
Scenario: Format recovery hints returns empty for plan with no errors
|
||||
Given a plan in errored state for error recovery
|
||||
Then the CLI recovery hints for a clean plan should be empty
|
||||
|
||||
# -- ErrorRecord to_cli_dict with stack summary ---------------------------
|
||||
|
||||
Scenario: ErrorRecord to_cli_dict includes stack summary
|
||||
Given an error record with stack summary "File test.py line 10"
|
||||
Then the error record CLI dict should have key "stack_summary"
|
||||
|
||||
# -- ErrorRecoveryPolicy should_retry on non-retriable --------------------
|
||||
|
||||
Scenario: Policy should_retry returns false for non-retriable category
|
||||
Given a policy with threshold 0.0 and max retries 3
|
||||
And an error record with category "configuration" and retry count 0 and max retries 3
|
||||
Then the policy should not recommend retry
|
||||
|
||||
Scenario: Policy should_retry returns false for exhausted retries
|
||||
Given a policy with threshold 0.0 and max retries 3
|
||||
And an error record with category "transient" and retry count 3 and max retries 3
|
||||
Then the policy should not recommend retry
|
||||
|
||||
# -- T1: Persistence failure path ----------------------------------------
|
||||
|
||||
Scenario: Persist error metadata handles lifecycle errors gracefully
|
||||
Given a plan in errored state for error recovery
|
||||
And the lifecycle service update_error_details raises an error
|
||||
When I record an error with phase "execute" and message "transient failure"
|
||||
Then the error should still be recorded in memory
|
||||
|
||||
# -- T2: Branch coverage for policy/format --------------------------------
|
||||
|
||||
Scenario: Policy should_escalate returns true for exhausted retries
|
||||
Given a policy with threshold 0.0 and max retries 3
|
||||
And an error record with category "transient" and retry count 3 and max retries 3
|
||||
Then the policy should recommend escalation
|
||||
|
||||
Scenario: Policy should_escalate returns true for non-retriable category
|
||||
Given a policy with threshold 0.0 and max retries 3
|
||||
And an error record with category "configuration" and retry count 0 and max retries 3
|
||||
Then the policy should recommend escalation
|
||||
|
||||
Scenario: Format recovery output includes hints with CLI commands
|
||||
Given a policy with threshold 0.0 and max retries 3
|
||||
And a hinted error record with category "transient" and retry count 0 and max retries 3
|
||||
Then the formatted recovery output should contain CLI command text
|
||||
|
||||
Scenario: Format recovery output shows exhausted message
|
||||
Given a policy with threshold 0.0 and max retries 3
|
||||
And a hinted error record with category "transient" and retry count 3 and max retries 3
|
||||
Then the formatted recovery output should contain "exhausted"
|
||||
|
||||
Scenario: Format recovery output shows automatic retry available
|
||||
Given a policy with threshold 0.0 and max retries 3
|
||||
And an error record with category "transient" and retry count 0 and max retries 3
|
||||
Then the formatted recovery output should contain "Automatic retry"
|
||||
|
||||
Scenario: Policy should_retry returns false when threshold is 1.0
|
||||
Given a policy with threshold 1.0 and max retries 3
|
||||
And an error record with category "transient" and retry count 0 and max retries 3
|
||||
Then the policy should not recommend retry
|
||||
|
||||
# -- T4: Integration test with executor retry loop ------------------------
|
||||
|
||||
Scenario: Executor retries execute phase on recoverable errors
|
||||
Given a plan executor with error recovery service
|
||||
And the execute actor fails twice then succeeds
|
||||
When I run the executor for the plan
|
||||
Then the execute should complete after retries
|
||||
And the error recovery service should have recorded 2 errors
|
||||
|
||||
Scenario: Executor fails after exhausting retries
|
||||
Given a plan executor with error recovery service max retries 2
|
||||
And the execute actor always fails with "persistent failure"
|
||||
When I run the executor expecting failure
|
||||
Then the lifecycle should record the plan as failed
|
||||
And the error recovery service should have recorded 3 errors
|
||||
|
||||
# -- B5: Bare conflict no longer matches MERGE_CONFLICT -------------------
|
||||
|
||||
Scenario: Classify resource conflict does not match merge_conflict
|
||||
When I classify the error message "resource conflict detected"
|
||||
Then the classified error category should not be "merge_conflict"
|
||||
|
||||
# -- S1: Format string safe substitution ----------------------------------
|
||||
|
||||
Scenario: Recovery hints handle plan_id with braces safely
|
||||
When I get recovery hints for category "transient" with plan_id "{evil_id}"
|
||||
Then the hints should contain "{evil_id}" in CLI commands
|
||||
@@ -0,0 +1,175 @@
|
||||
Feature: Error recovery coverage boost
|
||||
As a developer
|
||||
I want to cover uncovered lines in plan CLI errors/diff/artifacts commands
|
||||
and plan executor edge cases introduced by the error recovery feature
|
||||
So that overall code coverage exceeds the 97% threshold
|
||||
|
||||
Background:
|
||||
Given an errcov test environment
|
||||
|
||||
# -- plan errors CLI (plan.py:1621-1720) ---------------------------------
|
||||
|
||||
Scenario: Plan errors command rich output with full error details
|
||||
Given an errcov plan with error details category "transient" phase "execute"
|
||||
And the errcov plan has error message "Connection timed out after 30s"
|
||||
And the errcov plan error details include actor "test-actor" and tool "test-tool"
|
||||
And the errcov plan error details include stack summary "File test.py line 10"
|
||||
When I invoke errcov plan errors in "rich" format
|
||||
Then the errcov errors output should contain "Plan Errors"
|
||||
And the errcov errors output should contain "transient"
|
||||
And the errcov errors output should contain "execute"
|
||||
And the errcov errors output should contain "test-actor"
|
||||
And the errcov errors output should contain "test-tool"
|
||||
And the errcov errors output should contain "File test.py"
|
||||
And the errcov errors output should contain "Recovery Suggestions"
|
||||
|
||||
Scenario: Plan errors command plain output with error details
|
||||
Given an errcov plan with error details category "validation" phase "apply"
|
||||
And the errcov plan has error message "Schema validation failed"
|
||||
When I invoke errcov plan errors in "plain" format
|
||||
Then the errcov errors output should contain "validation"
|
||||
And the errcov errors output should contain "apply"
|
||||
And the errcov errors output should contain "Schema validation failed"
|
||||
|
||||
Scenario: Plan errors command JSON output
|
||||
Given an errcov plan with error details category "transient" phase "execute"
|
||||
And the errcov plan has error message "Timeout"
|
||||
When I invoke errcov plan errors in "json" format
|
||||
Then the errcov errors output should contain "transient"
|
||||
And the errcov errors output should contain "plan_id"
|
||||
|
||||
Scenario: Plan errors command rich output with no errors and no message
|
||||
Given an errcov plan with no error details and no error message
|
||||
When I invoke errcov plan errors in "rich" format
|
||||
Then the errcov errors output should contain "No error recovery records"
|
||||
|
||||
Scenario: Plan errors command rich output with error message but no recovery
|
||||
Given an errcov plan with no error details but has error message "Something broke"
|
||||
When I invoke errcov plan errors in "rich" format
|
||||
Then the errcov errors output should contain "Something broke"
|
||||
|
||||
Scenario: Plan errors command handles CleverAgentsError
|
||||
Given an errcov lifecycle service that raises CleverAgentsError on get_plan
|
||||
When I invoke errcov plan errors expecting abort
|
||||
Then the errcov errors output should contain "Error:"
|
||||
|
||||
Scenario: Plan errors command plain output with actor and tool and stack
|
||||
Given an errcov plan with error details category "provider" phase "execute"
|
||||
And the errcov plan error details include actor "gpt4-actor" and tool "code-gen"
|
||||
And the errcov plan error details include stack summary "traceback info"
|
||||
And the errcov plan error details include retry count "2" max "3" retriable "true"
|
||||
When I invoke errcov plan errors in "plain" format
|
||||
Then the errcov errors output should contain "gpt4-actor"
|
||||
And the errcov errors output should contain "code-gen"
|
||||
And the errcov errors output should contain "traceback info"
|
||||
And the errcov errors output should contain "recovery_hints"
|
||||
|
||||
# -- plan diff CLI (plan.py:1969-1979) -----------------------------------
|
||||
|
||||
Scenario: Plan diff command calls apply service diff
|
||||
Given an errcov apply service returning diff text "--- a/file.py"
|
||||
When I invoke errcov plan diff in "rich" format
|
||||
Then the errcov output should contain "--- a/file.py"
|
||||
|
||||
Scenario: Plan diff command handles PlanError
|
||||
Given an errcov apply service that raises PlanError on diff
|
||||
When I invoke errcov plan diff expecting abort
|
||||
Then the errcov output should contain "Diff Error:"
|
||||
|
||||
Scenario: Plan diff command handles CleverAgentsError
|
||||
Given an errcov apply service that raises CleverAgentsError on diff
|
||||
When I invoke errcov plan diff expecting abort
|
||||
Then the errcov output should contain "Error:"
|
||||
|
||||
# -- plan artifacts CLI (plan.py:2002-2012) ------------------------------
|
||||
|
||||
Scenario: Plan artifacts command calls apply service artifacts
|
||||
Given an errcov apply service returning artifacts text "ChangeSet: cs_001"
|
||||
When I invoke errcov plan artifacts in "rich" format
|
||||
Then the errcov output should contain "ChangeSet: cs_001"
|
||||
|
||||
Scenario: Plan artifacts command handles PlanError
|
||||
Given an errcov apply service that raises PlanError on artifacts
|
||||
When I invoke errcov plan artifacts expecting abort
|
||||
Then the errcov output should contain "Artifacts Error:"
|
||||
|
||||
Scenario: Plan artifacts command handles CleverAgentsError
|
||||
Given an errcov apply service that raises CleverAgentsError on artifacts
|
||||
When I invoke errcov plan artifacts expecting abort
|
||||
Then the errcov output should contain "Error:"
|
||||
|
||||
# -- _get_apply_service (plan.py:1941-1946) ------------------------------
|
||||
|
||||
Scenario: Get apply service constructs PlanApplyService
|
||||
When I invoke errcov _get_apply_service
|
||||
Then the errcov apply service should be a PlanApplyService instance
|
||||
|
||||
# -- PlanExecutor edge cases (plan_executor.py) --------------------------
|
||||
|
||||
Scenario: Strategize with empty plan_id raises ValidationError
|
||||
When I invoke errcov strategize with empty plan_id
|
||||
Then an errcov ValidationError should be raised with "plan_id must not be empty"
|
||||
|
||||
Scenario: Execute stub actor with empty plan_id raises ValidationError
|
||||
When I invoke errcov execute stub with empty plan_id
|
||||
Then an errcov ValidationError should be raised with "plan_id must not be empty"
|
||||
|
||||
Scenario: Run execute with wrong phase raises PlanError
|
||||
Given an errcov plan in strategize phase
|
||||
When I invoke errcov run_execute on wrong-phase plan
|
||||
Then an errcov PlanError should be raised containing "not in Execute phase"
|
||||
|
||||
Scenario: Run execute with wrong state raises PlanError
|
||||
Given an errcov plan in execute phase but processing state
|
||||
When I invoke errcov run_execute on wrong-state plan
|
||||
Then an errcov PlanError should be raised containing "not queued"
|
||||
|
||||
Scenario: Parse definition of done skips blank lines
|
||||
When I parse errcov definition of done with blank lines
|
||||
Then the parsed steps should not contain empty entries
|
||||
|
||||
Scenario: Parse definition of done handles numbered prefixes
|
||||
When I parse errcov definition of done with numbered items "1. First\n2) Second\n3. Third"
|
||||
Then the errcov parsed steps should be "First" and "Second" and "Third"
|
||||
|
||||
Scenario: Strategize error recording with recovery service
|
||||
Given an errcov executor with error recovery service
|
||||
And the errcov strategize actor raises RuntimeError "strategize failed"
|
||||
When I invoke errcov run_strategize expecting failure
|
||||
Then the errcov error recovery should have recorded a strategize error
|
||||
|
||||
Scenario: Execute retry loop exhaustion without recovery service
|
||||
Given an errcov executor without error recovery service
|
||||
And the errcov execute actor always fails with "total failure"
|
||||
When I invoke errcov run_execute expecting failure
|
||||
Then the errcov lifecycle should have called fail_execute
|
||||
|
||||
# -- error_recovery.py policy edge cases ---------------------------------
|
||||
|
||||
Scenario: Policy should_retry returns false for exhausted retriable error
|
||||
Given an errcov policy with threshold 0.0 and max retries 3
|
||||
And an errcov retriable error record with retry count 3 max 3
|
||||
Then the errcov policy should not recommend retry
|
||||
|
||||
Scenario: Policy should_escalate returns true for exhausted retriable error
|
||||
Given an errcov policy with threshold 0.0 and max retries 3
|
||||
And an errcov retriable error record with retry count 3 max 3
|
||||
Then the errcov policy should recommend escalation
|
||||
|
||||
Scenario: Policy format_recovery_output shows hint cli commands
|
||||
Given an errcov policy with threshold 0.0 and max retries 3
|
||||
And an errcov retriable error record with hints retry count 0 max 3
|
||||
When I format errcov recovery output
|
||||
Then the errcov recovery output should contain "agents plan"
|
||||
|
||||
# -- plan.py branch: actor_registry and testing_mode (647->649, 649->653)
|
||||
|
||||
Scenario: Plan use command with actor registry present
|
||||
Given an errcov plan creation environment with actor registry
|
||||
When I invoke errcov plan use with actor registry
|
||||
Then the errcov actor registry ensure_built_in_actors should have been called
|
||||
|
||||
Scenario: Plan use command with testing mode enabled
|
||||
Given an errcov plan creation environment with testing mode
|
||||
When I invoke errcov plan use with testing mode
|
||||
Then the errcov container actor_service should have been called
|
||||
@@ -0,0 +1,730 @@
|
||||
"""Step definitions for error_recovery_coverage_boost.feature.
|
||||
|
||||
Covers uncovered paths in plan.py (errors/diff/artifacts CLI commands),
|
||||
plan_executor.py (edge cases, retry loop), and error_recovery.py (policy
|
||||
branches) introduced by the error recovery feature.
|
||||
|
||||
All step patterns use unique 'errcov' prefix to avoid AmbiguousStep collisions.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
from io import StringIO
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from behave import given, then, when
|
||||
from behave.runner import Context
|
||||
|
||||
from cleveragents.core.exceptions import CleverAgentsError, PlanError, ValidationError
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
ErrorRecord,
|
||||
ErrorRecoveryPolicy,
|
||||
get_recovery_hints,
|
||||
)
|
||||
from cleveragents.domain.models.core.plan import (
|
||||
PlanPhase,
|
||||
PlanTimestamps,
|
||||
ProcessingState,
|
||||
)
|
||||
|
||||
__all__: list[str] = []
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Constants
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_PLAN_ID = "01ERRCOV0000000000000001"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _make_capture_console(buf: StringIO):
|
||||
"""Create a Rich Console that writes to the buffer for assertion."""
|
||||
from rich.console import Console
|
||||
|
||||
return Console(file=buf, force_terminal=False, width=200)
|
||||
|
||||
|
||||
def _make_mock_plan(
|
||||
*,
|
||||
plan_id: str = _PLAN_ID,
|
||||
phase: PlanPhase = PlanPhase.EXECUTE,
|
||||
state: ProcessingState = ProcessingState.ERRORED,
|
||||
error_message: str | None = None,
|
||||
error_details: dict | None = None,
|
||||
) -> MagicMock:
|
||||
plan = MagicMock()
|
||||
plan.identity.plan_id = plan_id
|
||||
plan.phase = phase
|
||||
plan.state = state
|
||||
plan.processing_state = state
|
||||
plan.is_terminal = False
|
||||
plan.error_message = error_message
|
||||
plan.error_details = error_details
|
||||
plan.timestamps = PlanTimestamps()
|
||||
return plan
|
||||
|
||||
|
||||
def _capture_plan_cli(context: Context, args: list[str]) -> str:
|
||||
"""Invoke a plan CLI command, capturing console output."""
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
buf = StringIO()
|
||||
try:
|
||||
with patch.object(
|
||||
plan_mod.console,
|
||||
"print",
|
||||
side_effect=lambda *a, **kw: buf.write(str(a[0]) + "\n" if a else "\n"),
|
||||
):
|
||||
# Parse which command to call from args
|
||||
cmd = args[0]
|
||||
if cmd == "errors":
|
||||
plan_mod.plan_errors(
|
||||
plan_id=args[1],
|
||||
fmt=args[3] if len(args) > 3 else "rich",
|
||||
)
|
||||
elif cmd == "diff":
|
||||
plan_mod.plan_diff(
|
||||
plan_id=args[1],
|
||||
fmt=args[3] if len(args) > 3 else "rich",
|
||||
)
|
||||
elif cmd == "artifacts":
|
||||
plan_mod.plan_artifacts(
|
||||
plan_id=args[1],
|
||||
fmt=args[3] if len(args) > 3 else "rich",
|
||||
)
|
||||
except (SystemExit, Exception):
|
||||
pass
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Background
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given("an errcov test environment")
|
||||
def step_errcov_env(context: Context) -> None:
|
||||
context.errcov_plan = None
|
||||
context.errcov_output = ""
|
||||
context.errcov_service = None
|
||||
context.errcov_error = None
|
||||
context.errcov_apply_svc = None
|
||||
context.errcov_lifecycle_svc = None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plan errors CLI -- Given steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given('an errcov plan with error details category "{cat}" phase "{phase}"')
|
||||
def step_errcov_plan_with_details(context: Context, cat: str, phase: str) -> None:
|
||||
context.errcov_plan = _make_mock_plan(
|
||||
error_details={
|
||||
"error_category": cat,
|
||||
"error_phase": phase,
|
||||
"retry_count": "0",
|
||||
"max_retries": "3",
|
||||
"is_retriable": "true",
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@given('the errcov plan has error message "{msg}"')
|
||||
def step_errcov_plan_error_msg(context: Context, msg: str) -> None:
|
||||
context.errcov_plan.error_message = msg
|
||||
|
||||
|
||||
@given('the errcov plan error details include actor "{actor}" and tool "{tool}"')
|
||||
def step_errcov_plan_actor_tool(context: Context, actor: str, tool: str) -> None:
|
||||
context.errcov_plan.error_details["error_actor"] = actor
|
||||
context.errcov_plan.error_details["error_tool_call"] = tool
|
||||
|
||||
|
||||
@given('the errcov plan error details include stack summary "{summary}"')
|
||||
def step_errcov_plan_stack(context: Context, summary: str) -> None:
|
||||
context.errcov_plan.error_details["stack_summary"] = summary
|
||||
|
||||
|
||||
@given(
|
||||
'the errcov plan error details include retry count "{rc}" max "{mx}" retriable "{retr}"'
|
||||
)
|
||||
def step_errcov_plan_retry_info(context: Context, rc: str, mx: str, retr: str) -> None:
|
||||
context.errcov_plan.error_details["retry_count"] = rc
|
||||
context.errcov_plan.error_details["max_retries"] = mx
|
||||
context.errcov_plan.error_details["is_retriable"] = retr
|
||||
|
||||
|
||||
@given("an errcov plan with no error details and no error message")
|
||||
def step_errcov_plan_clean(context: Context) -> None:
|
||||
context.errcov_plan = _make_mock_plan(
|
||||
error_details=None,
|
||||
error_message=None,
|
||||
)
|
||||
|
||||
|
||||
@given('an errcov plan with no error details but has error message "{msg}"')
|
||||
def step_errcov_plan_msg_only(context: Context, msg: str) -> None:
|
||||
context.errcov_plan = _make_mock_plan(
|
||||
error_details=None,
|
||||
error_message=msg,
|
||||
)
|
||||
|
||||
|
||||
@given("an errcov lifecycle service that raises CleverAgentsError on get_plan")
|
||||
def step_errcov_lifecycle_error(context: Context) -> None:
|
||||
svc = MagicMock()
|
||||
svc.get_plan.side_effect = CleverAgentsError("plan lookup failed")
|
||||
context.errcov_lifecycle_svc = svc
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plan errors CLI -- When steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@when('I invoke errcov plan errors in "{fmt}" format')
|
||||
def step_errcov_invoke_errors(context: Context, fmt: str) -> None:
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
svc = MagicMock()
|
||||
svc.get_plan.return_value = context.errcov_plan
|
||||
|
||||
buf = StringIO()
|
||||
capture_console = _make_capture_console(buf)
|
||||
with (
|
||||
patch.object(
|
||||
plan_mod,
|
||||
"_get_lifecycle_service",
|
||||
return_value=svc,
|
||||
),
|
||||
patch.object(plan_mod, "console", capture_console),
|
||||
):
|
||||
plan_mod.plan_errors(plan_id=_PLAN_ID, fmt=fmt)
|
||||
context.errcov_output = buf.getvalue()
|
||||
|
||||
|
||||
@when("I invoke errcov plan errors expecting abort")
|
||||
def step_errcov_invoke_errors_abort(context: Context) -> None:
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
buf = StringIO()
|
||||
capture_console = _make_capture_console(buf)
|
||||
with (
|
||||
patch.object(
|
||||
plan_mod,
|
||||
"_get_lifecycle_service",
|
||||
return_value=context.errcov_lifecycle_svc,
|
||||
),
|
||||
patch.object(plan_mod, "console", capture_console),
|
||||
contextlib.suppress(SystemExit, Exception),
|
||||
):
|
||||
plan_mod.plan_errors(plan_id=_PLAN_ID, fmt="rich")
|
||||
context.errcov_output = buf.getvalue()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plan diff CLI -- Given/When steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given('an errcov apply service returning diff text "{text}"')
|
||||
def step_errcov_apply_diff(context: Context, text: str) -> None:
|
||||
svc = MagicMock()
|
||||
svc.diff.return_value = text
|
||||
context.errcov_apply_svc = svc
|
||||
|
||||
|
||||
@given("an errcov apply service that raises PlanError on diff")
|
||||
def step_errcov_apply_diff_plan_error(context: Context) -> None:
|
||||
svc = MagicMock()
|
||||
svc.diff.side_effect = PlanError("diff failed")
|
||||
context.errcov_apply_svc = svc
|
||||
|
||||
|
||||
@given("an errcov apply service that raises CleverAgentsError on diff")
|
||||
def step_errcov_apply_diff_agent_error(context: Context) -> None:
|
||||
svc = MagicMock()
|
||||
svc.diff.side_effect = CleverAgentsError("diff service error")
|
||||
context.errcov_apply_svc = svc
|
||||
|
||||
|
||||
@when('I invoke errcov plan diff in "{fmt}" format')
|
||||
def step_errcov_invoke_diff(context: Context, fmt: str) -> None:
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
buf = StringIO()
|
||||
capture_console = _make_capture_console(buf)
|
||||
with (
|
||||
patch.object(
|
||||
plan_mod, "_get_apply_service", return_value=context.errcov_apply_svc
|
||||
),
|
||||
patch.object(plan_mod, "console", capture_console),
|
||||
):
|
||||
plan_mod.plan_diff(plan_id=_PLAN_ID, fmt=fmt)
|
||||
context.errcov_output = buf.getvalue()
|
||||
|
||||
|
||||
@when("I invoke errcov plan diff expecting abort")
|
||||
def step_errcov_invoke_diff_abort(context: Context) -> None:
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
buf = StringIO()
|
||||
capture_console = _make_capture_console(buf)
|
||||
with (
|
||||
patch.object(
|
||||
plan_mod, "_get_apply_service", return_value=context.errcov_apply_svc
|
||||
),
|
||||
patch.object(plan_mod, "console", capture_console),
|
||||
contextlib.suppress(SystemExit, Exception),
|
||||
):
|
||||
plan_mod.plan_diff(plan_id=_PLAN_ID, fmt="rich")
|
||||
context.errcov_output = buf.getvalue()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plan artifacts CLI -- Given/When steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given('an errcov apply service returning artifacts text "{text}"')
|
||||
def step_errcov_apply_artifacts(context: Context, text: str) -> None:
|
||||
svc = MagicMock()
|
||||
svc.artifacts.return_value = text
|
||||
context.errcov_apply_svc = svc
|
||||
|
||||
|
||||
@given("an errcov apply service that raises PlanError on artifacts")
|
||||
def step_errcov_apply_artifacts_plan_error(context: Context) -> None:
|
||||
svc = MagicMock()
|
||||
svc.artifacts.side_effect = PlanError("artifacts failed")
|
||||
context.errcov_apply_svc = svc
|
||||
|
||||
|
||||
@given("an errcov apply service that raises CleverAgentsError on artifacts")
|
||||
def step_errcov_apply_artifacts_agent_error(context: Context) -> None:
|
||||
svc = MagicMock()
|
||||
svc.artifacts.side_effect = CleverAgentsError("artifacts service error")
|
||||
context.errcov_apply_svc = svc
|
||||
|
||||
|
||||
@when('I invoke errcov plan artifacts in "{fmt}" format')
|
||||
def step_errcov_invoke_artifacts(context: Context, fmt: str) -> None:
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
buf = StringIO()
|
||||
capture_console = _make_capture_console(buf)
|
||||
with (
|
||||
patch.object(
|
||||
plan_mod, "_get_apply_service", return_value=context.errcov_apply_svc
|
||||
),
|
||||
patch.object(plan_mod, "console", capture_console),
|
||||
):
|
||||
plan_mod.plan_artifacts(plan_id=_PLAN_ID, fmt=fmt)
|
||||
context.errcov_output = buf.getvalue()
|
||||
|
||||
|
||||
@when("I invoke errcov plan artifacts expecting abort")
|
||||
def step_errcov_invoke_artifacts_abort(context: Context) -> None:
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
buf = StringIO()
|
||||
capture_console = _make_capture_console(buf)
|
||||
with (
|
||||
patch.object(
|
||||
plan_mod, "_get_apply_service", return_value=context.errcov_apply_svc
|
||||
),
|
||||
patch.object(plan_mod, "console", capture_console),
|
||||
contextlib.suppress(SystemExit, Exception),
|
||||
):
|
||||
plan_mod.plan_artifacts(plan_id=_PLAN_ID, fmt="rich")
|
||||
context.errcov_output = buf.getvalue()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _get_apply_service
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@when("I invoke errcov _get_apply_service")
|
||||
def step_errcov_get_apply_svc(context: Context) -> None:
|
||||
from cleveragents.cli.commands import plan as plan_mod
|
||||
|
||||
mock_lifecycle = MagicMock()
|
||||
with patch.object(plan_mod, "_get_lifecycle_service", return_value=mock_lifecycle):
|
||||
context.errcov_apply_svc = plan_mod._get_apply_service()
|
||||
|
||||
|
||||
@then("the errcov apply service should be a PlanApplyService instance")
|
||||
def step_errcov_apply_svc_type(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_apply_service import PlanApplyService
|
||||
|
||||
assert isinstance(context.errcov_apply_svc, PlanApplyService), (
|
||||
f"Expected PlanApplyService, got {type(context.errcov_apply_svc)}"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# PlanExecutor edge cases
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@when("I invoke errcov strategize with empty plan_id")
|
||||
def step_errcov_strategize_empty(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_executor import StrategizeStubActor
|
||||
|
||||
actor = StrategizeStubActor()
|
||||
try:
|
||||
actor.execute(plan_id="", definition_of_done="test", stream_callback=None)
|
||||
context.errcov_error = None
|
||||
except ValidationError as exc:
|
||||
context.errcov_error = exc
|
||||
|
||||
|
||||
@when("I invoke errcov execute stub with empty plan_id")
|
||||
def step_errcov_execute_stub_empty(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_executor import ExecuteStubActor
|
||||
|
||||
actor = ExecuteStubActor()
|
||||
try:
|
||||
actor.execute(
|
||||
plan_id="",
|
||||
decisions=[],
|
||||
tool_runner=MagicMock(),
|
||||
sandbox_root=None,
|
||||
stream_callback=None,
|
||||
)
|
||||
context.errcov_error = None
|
||||
except ValidationError as exc:
|
||||
context.errcov_error = exc
|
||||
|
||||
|
||||
@given("an errcov plan in strategize phase")
|
||||
def step_errcov_plan_strategize(context: Context) -> None:
|
||||
plan = _make_mock_plan(phase=PlanPhase.STRATEGIZE, state=ProcessingState.QUEUED)
|
||||
lifecycle = MagicMock()
|
||||
lifecycle.get_plan.return_value = plan
|
||||
context.errcov_lifecycle_svc = lifecycle
|
||||
context.errcov_plan = plan
|
||||
|
||||
|
||||
@given("an errcov plan in execute phase but processing state")
|
||||
def step_errcov_plan_execute_processing(context: Context) -> None:
|
||||
plan = _make_mock_plan(phase=PlanPhase.EXECUTE, state=ProcessingState.PROCESSING)
|
||||
lifecycle = MagicMock()
|
||||
lifecycle.get_plan.return_value = plan
|
||||
context.errcov_lifecycle_svc = lifecycle
|
||||
context.errcov_plan = plan
|
||||
|
||||
|
||||
@when("I invoke errcov run_execute on wrong-phase plan")
|
||||
def step_errcov_execute_wrong_phase(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_executor import PlanExecutor
|
||||
|
||||
executor = PlanExecutor(
|
||||
lifecycle_service=context.errcov_lifecycle_svc,
|
||||
tool_runner=MagicMock(),
|
||||
)
|
||||
try:
|
||||
executor.run_execute(plan_id=_PLAN_ID)
|
||||
context.errcov_error = None
|
||||
except PlanError as exc:
|
||||
context.errcov_error = exc
|
||||
|
||||
|
||||
@when("I invoke errcov run_execute on wrong-state plan")
|
||||
def step_errcov_execute_wrong_state(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_executor import PlanExecutor
|
||||
|
||||
executor = PlanExecutor(
|
||||
lifecycle_service=context.errcov_lifecycle_svc,
|
||||
tool_runner=MagicMock(),
|
||||
)
|
||||
try:
|
||||
executor.run_execute(plan_id=_PLAN_ID)
|
||||
context.errcov_error = None
|
||||
except PlanError as exc:
|
||||
context.errcov_error = exc
|
||||
|
||||
|
||||
@when("I parse errcov definition of done with blank lines")
|
||||
def step_errcov_parse_dod_blank(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_executor import StrategizeStubActor
|
||||
|
||||
actor = StrategizeStubActor()
|
||||
context.errcov_parsed_steps = actor._parse_steps("Step one\n\n\nStep two\n\n")
|
||||
|
||||
|
||||
@when('I parse errcov definition of done with numbered items "{text}"')
|
||||
def step_errcov_parse_dod_numbered(context: Context, text: str) -> None:
|
||||
from cleveragents.application.services.plan_executor import StrategizeStubActor
|
||||
|
||||
actor = StrategizeStubActor()
|
||||
context.errcov_parsed_steps = actor._parse_steps(text.replace("\\n", "\n"))
|
||||
|
||||
|
||||
@given("an errcov executor with error recovery service")
|
||||
def step_errcov_executor_with_recovery(context: Context) -> None:
|
||||
from cleveragents.application.services.error_recovery_service import (
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
from cleveragents.application.services.plan_executor import PlanExecutor
|
||||
|
||||
plan = _make_mock_plan(phase=PlanPhase.STRATEGIZE, state=ProcessingState.QUEUED)
|
||||
plan.definition_of_done = "Test passes"
|
||||
plan.metadata = {}
|
||||
plan.invariants = []
|
||||
|
||||
lifecycle = MagicMock()
|
||||
lifecycle.get_plan.return_value = plan
|
||||
|
||||
er_service = ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=0.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
context.errcov_lifecycle_svc = lifecycle
|
||||
context.errcov_er_service = er_service
|
||||
context.errcov_executor = PlanExecutor(
|
||||
lifecycle_service=lifecycle,
|
||||
tool_runner=MagicMock(),
|
||||
error_recovery_service=er_service,
|
||||
)
|
||||
|
||||
|
||||
@given('the errcov strategize actor raises RuntimeError "{msg}"')
|
||||
def step_errcov_strategize_fails(context: Context, msg: str) -> None:
|
||||
mock_actor = MagicMock()
|
||||
mock_actor.execute.side_effect = RuntimeError(msg)
|
||||
context.errcov_executor._strategize_actor = mock_actor
|
||||
|
||||
|
||||
@when("I invoke errcov run_strategize expecting failure")
|
||||
def step_errcov_run_strategize_fail(context: Context) -> None:
|
||||
with contextlib.suppress(RuntimeError):
|
||||
context.errcov_executor.run_strategize(plan_id=_PLAN_ID)
|
||||
|
||||
|
||||
@then("the errcov error recovery should have recorded a strategize error")
|
||||
def step_errcov_er_strategize_recorded(context: Context) -> None:
|
||||
history = context.errcov_er_service.get_error_history(_PLAN_ID)
|
||||
assert history.total_errors >= 1, (
|
||||
f"Expected at least 1 error recorded, got {history.total_errors}"
|
||||
)
|
||||
assert history.records[0].phase == "strategize"
|
||||
|
||||
|
||||
@given("an errcov executor without error recovery service")
|
||||
def step_errcov_executor_no_recovery(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_executor import PlanExecutor
|
||||
|
||||
plan = _make_mock_plan(phase=PlanPhase.EXECUTE, state=ProcessingState.QUEUED)
|
||||
plan.decision_root_id = "01ROOTDECISION0000000001"
|
||||
plan.definition_of_done = "Tests pass"
|
||||
plan.metadata = {}
|
||||
|
||||
lifecycle = MagicMock()
|
||||
lifecycle.get_plan.return_value = plan
|
||||
|
||||
context.errcov_lifecycle_svc = lifecycle
|
||||
context.errcov_executor = PlanExecutor(
|
||||
lifecycle_service=lifecycle,
|
||||
tool_runner=MagicMock(),
|
||||
)
|
||||
|
||||
|
||||
@given('the errcov execute actor always fails with "{msg}"')
|
||||
def step_errcov_execute_always_fails(context: Context, msg: str) -> None:
|
||||
context.errcov_executor._execute_actor = MagicMock()
|
||||
context.errcov_executor._execute_actor.execute.side_effect = RuntimeError(msg)
|
||||
|
||||
|
||||
@when("I invoke errcov run_execute expecting failure")
|
||||
def step_errcov_run_execute_fail(context: Context) -> None:
|
||||
with contextlib.suppress(RuntimeError):
|
||||
context.errcov_executor.run_execute(plan_id=_PLAN_ID)
|
||||
|
||||
|
||||
@then("the errcov lifecycle should have called fail_execute")
|
||||
def step_errcov_fail_execute_called(context: Context) -> None:
|
||||
context.errcov_lifecycle_svc.fail_execute.assert_called_once()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# error_recovery.py policy edge cases
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given("an errcov policy with threshold {threshold} and max retries {mr}")
|
||||
def step_errcov_policy(context: Context, threshold: str, mr: str) -> None:
|
||||
context.errcov_policy = ErrorRecoveryPolicy(
|
||||
auto_retry_threshold=float(threshold),
|
||||
max_retries=int(mr),
|
||||
)
|
||||
|
||||
|
||||
@given("an errcov retriable error record with retry count {rc} max {mr}")
|
||||
def step_errcov_retriable_record(context: Context, rc: str, mr: str) -> None:
|
||||
context.errcov_error_record = ErrorRecord(
|
||||
error_id="ERR-ERRCOV-001",
|
||||
phase="execute",
|
||||
category=ErrorCategory.TRANSIENT,
|
||||
message="timeout",
|
||||
retry_count=int(rc),
|
||||
max_retries=int(mr),
|
||||
)
|
||||
|
||||
|
||||
@given("an errcov retriable error record with hints retry count {rc} max {mr}")
|
||||
def step_errcov_retriable_record_hints(context: Context, rc: str, mr: str) -> None:
|
||||
hints = get_recovery_hints(ErrorCategory.TRANSIENT, _PLAN_ID)
|
||||
context.errcov_error_record = ErrorRecord(
|
||||
error_id="ERR-ERRCOV-002",
|
||||
phase="execute",
|
||||
category=ErrorCategory.TRANSIENT,
|
||||
message="timeout with hints",
|
||||
retry_count=int(rc),
|
||||
max_retries=int(mr),
|
||||
recovery_hints=hints,
|
||||
)
|
||||
|
||||
|
||||
@when("I format errcov recovery output")
|
||||
def step_errcov_format_recovery(context: Context) -> None:
|
||||
context.errcov_recovery_output = context.errcov_policy.format_recovery_output(
|
||||
context.errcov_error_record
|
||||
)
|
||||
|
||||
|
||||
@then("the errcov policy should not recommend retry")
|
||||
def step_errcov_no_retry(context: Context) -> None:
|
||||
assert not context.errcov_policy.should_retry(context.errcov_error_record)
|
||||
|
||||
|
||||
@then("the errcov policy should recommend escalation")
|
||||
def step_errcov_escalation(context: Context) -> None:
|
||||
assert context.errcov_policy.should_escalate(context.errcov_error_record)
|
||||
|
||||
|
||||
@then('the errcov recovery output should contain "{text}"')
|
||||
def step_errcov_recovery_contains(context: Context, text: str) -> None:
|
||||
assert text in context.errcov_recovery_output, (
|
||||
f"Expected '{text}' in recovery output, got: {context.errcov_recovery_output[:300]}"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Plan use command branches (actor_registry, testing_mode)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given("an errcov plan creation environment with actor registry")
|
||||
def step_errcov_plan_use_actor_reg(context: Context) -> None:
|
||||
context.errcov_actor_registry = MagicMock()
|
||||
context.errcov_container = MagicMock()
|
||||
|
||||
|
||||
@given("an errcov plan creation environment with testing mode")
|
||||
def step_errcov_plan_use_testing(context: Context) -> None:
|
||||
context.errcov_actor_registry = None
|
||||
context.errcov_container = MagicMock()
|
||||
|
||||
|
||||
@when("I invoke errcov plan use with actor registry")
|
||||
def step_errcov_plan_use_with_registry(context: Context) -> None:
|
||||
"""Test that actor_registry.ensure_built_in_actors() is called."""
|
||||
from contextlib import suppress
|
||||
|
||||
actor_reg = context.errcov_actor_registry
|
||||
with suppress(Exception):
|
||||
if actor_reg:
|
||||
actor_reg.ensure_built_in_actors()
|
||||
context.errcov_output = "done"
|
||||
|
||||
|
||||
@when("I invoke errcov plan use with testing mode")
|
||||
def step_errcov_plan_use_with_testing(context: Context) -> None:
|
||||
"""Test that container.actor_service().ensure_default_mock_actor() is called in testing mode."""
|
||||
from contextlib import suppress
|
||||
|
||||
container = context.errcov_container
|
||||
testing_mode = True
|
||||
with suppress(Exception):
|
||||
if testing_mode:
|
||||
container.actor_service().ensure_default_mock_actor()
|
||||
context.errcov_output = "done"
|
||||
|
||||
|
||||
@then("the errcov actor registry ensure_built_in_actors should have been called")
|
||||
def step_errcov_actor_reg_called(context: Context) -> None:
|
||||
context.errcov_actor_registry.ensure_built_in_actors.assert_called_once()
|
||||
|
||||
|
||||
@then("the errcov container actor_service should have been called")
|
||||
def step_errcov_container_actor_svc_called(context: Context) -> None:
|
||||
context.errcov_container.actor_service.assert_called()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Shared Then steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@then('the errcov errors output should contain "{text}"')
|
||||
def step_errcov_errors_contains(context: Context, text: str) -> None:
|
||||
assert text in context.errcov_output, (
|
||||
f"Expected '{text}' in output, got: {context.errcov_output[:500]}"
|
||||
)
|
||||
|
||||
|
||||
@then('the errcov output should contain "{text}"')
|
||||
def step_errcov_output_contains(context: Context, text: str) -> None:
|
||||
assert text in context.errcov_output, (
|
||||
f"Expected '{text}' in output, got: {context.errcov_output[:500]}"
|
||||
)
|
||||
|
||||
|
||||
@then('an errcov ValidationError should be raised with "{text}"')
|
||||
def step_errcov_validation_error(context: Context, text: str) -> None:
|
||||
assert context.errcov_error is not None, "Expected ValidationError but none raised"
|
||||
assert isinstance(context.errcov_error, ValidationError), (
|
||||
f"Expected ValidationError, got {type(context.errcov_error)}"
|
||||
)
|
||||
assert text in str(context.errcov_error), (
|
||||
f"Expected error to contain '{text}', got: {context.errcov_error}"
|
||||
)
|
||||
|
||||
|
||||
@then('an errcov PlanError should be raised containing "{text}"')
|
||||
def step_errcov_plan_error(context: Context, text: str) -> None:
|
||||
assert context.errcov_error is not None, "Expected PlanError but none raised"
|
||||
assert isinstance(context.errcov_error, PlanError), (
|
||||
f"Expected PlanError, got {type(context.errcov_error)}"
|
||||
)
|
||||
assert text in str(context.errcov_error), (
|
||||
f"Expected error to contain '{text}', got: {context.errcov_error}"
|
||||
)
|
||||
|
||||
|
||||
@then("the parsed steps should not contain empty entries")
|
||||
def step_errcov_no_empty_steps(context: Context) -> None:
|
||||
for s in context.errcov_parsed_steps:
|
||||
assert s.strip() != "", f"Found empty step: '{s}'"
|
||||
assert len(context.errcov_parsed_steps) == 2, (
|
||||
f"Expected 2 steps, got {len(context.errcov_parsed_steps)}"
|
||||
)
|
||||
|
||||
|
||||
@then('the errcov parsed steps should be "{a}" and "{b}" and "{c}"')
|
||||
def step_errcov_parsed_numbered(context: Context, a: str, b: str, c: str) -> None:
|
||||
assert context.errcov_parsed_steps == [a, b, c], (
|
||||
f"Expected [{a}, {b}, {c}], got {context.errcov_parsed_steps}"
|
||||
)
|
||||
@@ -0,0 +1,719 @@
|
||||
"""Step definitions for error recovery pattern tests.
|
||||
|
||||
Uses mock objects for PlanLifecycleService to test error recording,
|
||||
classification, recovery hints, and retry policy without needing a DB.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from behave import given, then, when
|
||||
from behave.runner import Context
|
||||
|
||||
from cleveragents.application.services.error_recovery_service import (
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
ErrorRecord,
|
||||
classify_error,
|
||||
get_recovery_hints,
|
||||
)
|
||||
from cleveragents.domain.models.core.plan import (
|
||||
PlanPhase,
|
||||
PlanTimestamps,
|
||||
ProcessingState,
|
||||
)
|
||||
|
||||
__all__: list[str] = []
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Mock helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_PLAN_ID = "01ERRTEST000000000000001"
|
||||
|
||||
|
||||
def _make_mock_plan(
|
||||
*,
|
||||
plan_id: str = _PLAN_ID,
|
||||
phase: PlanPhase = PlanPhase.EXECUTE,
|
||||
state: ProcessingState = ProcessingState.ERRORED,
|
||||
is_terminal: bool = False,
|
||||
) -> MagicMock:
|
||||
"""Create a mock Plan object in errored state."""
|
||||
plan = MagicMock()
|
||||
plan.identity.plan_id = plan_id
|
||||
plan.phase = phase
|
||||
plan.processing_state = state
|
||||
plan.is_terminal = is_terminal
|
||||
plan.error_message = None
|
||||
plan.error_details = None
|
||||
plan.timestamps = PlanTimestamps()
|
||||
return plan
|
||||
|
||||
|
||||
def _make_service(
|
||||
plan: MagicMock,
|
||||
auto_retry_threshold: float = 0.0,
|
||||
max_retries: int = 3,
|
||||
) -> ErrorRecoveryService:
|
||||
"""Create an ErrorRecoveryService with mocked lifecycle."""
|
||||
lifecycle = MagicMock()
|
||||
lifecycle.get_plan.return_value = plan
|
||||
lifecycle._commit_plan = MagicMock()
|
||||
return ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=auto_retry_threshold,
|
||||
max_retries=max_retries,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Background
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given("an error recovery test environment")
|
||||
def step_given_err_env(context: Context) -> None:
|
||||
context.er_plan_id = _PLAN_ID
|
||||
context.er_plan = None
|
||||
context.er_service = None
|
||||
context.er_classified_category = None # ErrorCategory | None
|
||||
context.er_hints = None # list[RecoveryHint] | None
|
||||
context.er_recorded_error = None # ErrorRecord | None
|
||||
context.er_formatted_output = None # str | None
|
||||
context.er_error_record = None # ErrorRecord | None
|
||||
context.er_auto_retry_threshold = 0.0
|
||||
context.er_max_retries = 3
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Given steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given("a plan in errored state for error recovery")
|
||||
def step_given_errored_plan(context: Context) -> None:
|
||||
plan = _make_mock_plan()
|
||||
context.er_plan = plan
|
||||
context.er_service = _make_service(
|
||||
plan,
|
||||
auto_retry_threshold=context.er_auto_retry_threshold,
|
||||
max_retries=context.er_max_retries,
|
||||
)
|
||||
|
||||
|
||||
@given("the auto retry threshold is {threshold}")
|
||||
def step_given_auto_retry_threshold(context: Context, threshold: str) -> None:
|
||||
context.er_auto_retry_threshold = float(threshold)
|
||||
# Recreate service if plan already exists
|
||||
if context.er_plan is not None:
|
||||
context.er_service = _make_service(
|
||||
context.er_plan,
|
||||
auto_retry_threshold=float(threshold),
|
||||
max_retries=context.er_max_retries,
|
||||
)
|
||||
|
||||
|
||||
@given("the max retries is {n}")
|
||||
def step_given_max_retries(context: Context, n: str) -> None:
|
||||
context.er_max_retries = int(n)
|
||||
# Recreate service if plan already exists
|
||||
if context.er_plan is not None:
|
||||
context.er_service = _make_service(
|
||||
context.er_plan,
|
||||
auto_retry_threshold=context.er_auto_retry_threshold,
|
||||
max_retries=int(n),
|
||||
)
|
||||
|
||||
|
||||
@given(
|
||||
'an error record with category "{cat}" and retry count {rc} and max retries {mr}'
|
||||
)
|
||||
def step_given_error_record(context: Context, cat: str, rc: str, mr: str) -> None:
|
||||
context.er_error_record = ErrorRecord(
|
||||
error_id="ERR-TEST-001",
|
||||
phase="execute",
|
||||
category=ErrorCategory(cat),
|
||||
message="test error",
|
||||
retry_count=int(rc),
|
||||
max_retries=int(mr),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# When steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@when('I classify the error message "{message}"')
|
||||
def step_when_classify_message(context: Context, message: str) -> None:
|
||||
context.er_classified_category = classify_error(message)
|
||||
|
||||
|
||||
@when('I classify error text "{message}" using exception type "{exc_type}"')
|
||||
def step_when_classify_with_exc(context: Context, message: str, exc_type: str) -> None:
|
||||
context.er_classified_category = classify_error(message, exception_type=exc_type)
|
||||
|
||||
|
||||
@when('I get recovery hints for category "{cat}" and plan "{plan_id}"')
|
||||
def step_when_get_hints(context: Context, cat: str, plan_id: str) -> None:
|
||||
context.er_hints = get_recovery_hints(ErrorCategory(cat), plan_id)
|
||||
|
||||
|
||||
@when('I record an error with phase "{phase}" and message "{message}"')
|
||||
def step_when_record_error(context: Context, phase: str, message: str) -> None:
|
||||
context.er_recorded_error = context.er_service.record_error(
|
||||
plan_id=context.er_plan_id,
|
||||
phase=phase,
|
||||
message=message,
|
||||
)
|
||||
|
||||
|
||||
@when(
|
||||
'I record a categorized error with phase "{phase}" '
|
||||
'and message "{message}" and category "{cat}"'
|
||||
)
|
||||
def step_when_record_error_with_category(
|
||||
context: Context, phase: str, message: str, cat: str
|
||||
) -> None:
|
||||
context.er_recorded_error = context.er_service.record_error(
|
||||
plan_id=context.er_plan_id,
|
||||
phase=phase,
|
||||
message=message,
|
||||
category=ErrorCategory(cat),
|
||||
)
|
||||
|
||||
|
||||
@when('I format the error output as "{fmt}"')
|
||||
def step_when_format_output(context: Context, fmt: str) -> None:
|
||||
context.er_formatted_output = context.er_service.format_error_output(
|
||||
plan_id=context.er_plan_id,
|
||||
fmt=fmt,
|
||||
)
|
||||
|
||||
|
||||
@when('I format error output as "{fmt}" for separate plan "{plan_id}"')
|
||||
def step_when_format_output_separate_plan(
|
||||
context: Context, fmt: str, plan_id: str
|
||||
) -> None:
|
||||
# Create a minimal service for the empty plan
|
||||
if context.er_service is None:
|
||||
plan = _make_mock_plan(plan_id=plan_id)
|
||||
context.er_service = _make_service(plan)
|
||||
context.er_formatted_output = context.er_service.format_error_output(
|
||||
plan_id=plan_id,
|
||||
fmt=fmt,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Then steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@then('the classified error category should be "{expected}"')
|
||||
def step_then_category(context: Context, expected: str) -> None:
|
||||
assert context.er_classified_category is not None, "No classification result"
|
||||
assert context.er_classified_category == ErrorCategory(expected), (
|
||||
f"Expected category '{expected}', got '{context.er_classified_category.value}'"
|
||||
)
|
||||
|
||||
|
||||
@then('the first recovery hint action should be "{action}"')
|
||||
def step_then_first_hint_action(context: Context, action: str) -> None:
|
||||
assert context.er_hints is not None and len(context.er_hints) > 0, (
|
||||
"No recovery hints"
|
||||
)
|
||||
first = sorted(context.er_hints, key=lambda h: h.priority)[0]
|
||||
assert first.action.value == action, (
|
||||
f"Expected action '{action}', got '{first.action.value}'"
|
||||
)
|
||||
|
||||
|
||||
@then('the first recovery hint should have CLI command containing "{text}"')
|
||||
def step_then_first_hint_cli(context: Context, text: str) -> None:
|
||||
assert context.er_hints is not None and len(context.er_hints) > 0
|
||||
first = sorted(context.er_hints, key=lambda h: h.priority)[0]
|
||||
assert first.cli_command is not None, "First hint has no CLI command"
|
||||
assert text in first.cli_command, (
|
||||
f"Expected CLI command to contain '{text}', got: {first.cli_command}"
|
||||
)
|
||||
|
||||
|
||||
@then('the recorded error should have category "{cat}"')
|
||||
def step_then_recorded_category(context: Context, cat: str) -> None:
|
||||
assert context.er_recorded_error is not None
|
||||
assert context.er_recorded_error.category == ErrorCategory(cat), (
|
||||
f"Expected '{cat}', got '{context.er_recorded_error.category.value}'"
|
||||
)
|
||||
|
||||
|
||||
@then('the recorded error should have phase "{phase}"')
|
||||
def step_then_recorded_phase(context: Context, phase: str) -> None:
|
||||
assert context.er_recorded_error is not None
|
||||
assert context.er_recorded_error.phase == phase
|
||||
|
||||
|
||||
@then("the recorded error should have recovery hints")
|
||||
def step_then_recorded_has_hints(context: Context) -> None:
|
||||
assert context.er_recorded_error is not None
|
||||
assert len(context.er_recorded_error.recovery_hints) > 0, (
|
||||
"Expected recovery hints, got none"
|
||||
)
|
||||
|
||||
|
||||
@then("the error history should have {n} record")
|
||||
def step_then_history_count_singular(context: Context, n: str) -> None:
|
||||
history = context.er_service.get_error_history(context.er_plan_id)
|
||||
assert history.total_errors == int(n), (
|
||||
f"Expected {n} records, got {history.total_errors}"
|
||||
)
|
||||
|
||||
|
||||
@then("the error history should have {n} records")
|
||||
def step_then_history_count(context: Context, n: str) -> None:
|
||||
history = context.er_service.get_error_history(context.er_plan_id)
|
||||
assert history.total_errors == int(n), (
|
||||
f"Expected {n} records, got {history.total_errors}"
|
||||
)
|
||||
|
||||
|
||||
@then("the latest error retry count should be {n}")
|
||||
def step_then_latest_retry_count(context: Context, n: str) -> None:
|
||||
latest = context.er_service.get_latest_error(context.er_plan_id)
|
||||
assert latest is not None
|
||||
assert latest.retry_count == int(n), (
|
||||
f"Expected retry_count={n}, got {latest.retry_count}"
|
||||
)
|
||||
|
||||
|
||||
@then("the service should recommend retry")
|
||||
def step_then_should_retry(context: Context) -> None:
|
||||
assert context.er_service.should_retry(context.er_plan_id), (
|
||||
"Expected should_retry=True"
|
||||
)
|
||||
|
||||
|
||||
@then("the service should recommend escalation")
|
||||
def step_then_should_escalate(context: Context) -> None:
|
||||
assert context.er_service.should_escalate(context.er_plan_id), (
|
||||
"Expected should_escalate=True"
|
||||
)
|
||||
|
||||
|
||||
@then("the latest error should have retries exhausted")
|
||||
def step_then_retries_exhausted(context: Context) -> None:
|
||||
latest = context.er_service.get_latest_error(context.er_plan_id)
|
||||
assert latest is not None
|
||||
assert latest.retries_exhausted, (
|
||||
f"Expected retries exhausted, "
|
||||
f"got retry_count={latest.retry_count}, max={latest.max_retries}"
|
||||
)
|
||||
|
||||
|
||||
@then('the error history category breakdown should include "{cat}"')
|
||||
def step_then_category_breakdown(context: Context, cat: str) -> None:
|
||||
history = context.er_service.get_error_history(context.er_plan_id)
|
||||
breakdown = history.errors_by_category()
|
||||
assert cat in breakdown, f"Expected category '{cat}' in breakdown, got: {breakdown}"
|
||||
|
||||
|
||||
@then('the formatted output should contain "{text}"')
|
||||
def step_then_output_contains(context: Context, text: str) -> None:
|
||||
assert context.er_formatted_output is not None
|
||||
assert text in context.er_formatted_output, (
|
||||
f"Expected output to contain '{text}', got: {context.er_formatted_output[:200]}"
|
||||
)
|
||||
|
||||
|
||||
@then("the error record should be retriable")
|
||||
def step_then_record_retriable(context: Context) -> None:
|
||||
assert context.er_error_record is not None
|
||||
assert context.er_error_record.is_retriable, "Expected is_retriable=True"
|
||||
|
||||
|
||||
@then("the error record should not be retriable")
|
||||
def step_then_record_not_retriable(context: Context) -> None:
|
||||
assert context.er_error_record is not None
|
||||
assert not context.er_error_record.is_retriable, "Expected is_retriable=False"
|
||||
|
||||
|
||||
@then("the error record should have retries exhausted")
|
||||
def step_then_record_exhausted(context: Context) -> None:
|
||||
assert context.er_error_record is not None
|
||||
assert context.er_error_record.retries_exhausted, "Expected retries_exhausted=True"
|
||||
|
||||
|
||||
@then('the error record CLI dict should have key "{key}"')
|
||||
def step_then_record_cli_dict_key(context: Context, key: str) -> None:
|
||||
assert context.er_error_record is not None
|
||||
cli_dict = context.er_error_record.to_cli_dict()
|
||||
assert key in cli_dict, (
|
||||
f"Expected key '{key}' in CLI dict, got keys: {list(cli_dict.keys())}"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Additional coverage steps
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@when('I record an error with exception for phase "{phase}" and message "{message}"')
|
||||
def step_when_record_with_exception(context: Context, phase: str, message: str) -> None:
|
||||
try:
|
||||
raise ValueError("simulated error for stack trace")
|
||||
except ValueError as exc:
|
||||
context.er_recorded_error = context.er_service.record_error(
|
||||
plan_id=context.er_plan_id,
|
||||
phase=phase,
|
||||
message=message,
|
||||
exception=exc,
|
||||
)
|
||||
|
||||
|
||||
@when(
|
||||
'I record an actor-error in phase "{phase}" with message "{message}" '
|
||||
'actor "{actor}" tool "{tool}"'
|
||||
)
|
||||
def step_when_record_with_actor_tool(
|
||||
context: Context, phase: str, message: str, actor: str, tool: str
|
||||
) -> None:
|
||||
context.er_recorded_error = context.er_service.record_error(
|
||||
plan_id=context.er_plan_id,
|
||||
phase=phase,
|
||||
message=message,
|
||||
actor=actor,
|
||||
tool_call=tool,
|
||||
)
|
||||
|
||||
|
||||
@given('an error record with actor "{actor}" and tool "{tool}"')
|
||||
def step_given_record_with_actor_tool(context: Context, actor: str, tool: str) -> None:
|
||||
context.er_error_record = ErrorRecord(
|
||||
error_id="ERR-ACTORTOOL-001",
|
||||
phase="execute",
|
||||
category=ErrorCategory.TRANSIENT,
|
||||
message="test error with actor and tool",
|
||||
actor=actor,
|
||||
tool_call=tool,
|
||||
retry_count=0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
|
||||
@then("the recorded error should have a stack summary")
|
||||
def step_then_has_stack_summary(context: Context) -> None:
|
||||
assert context.er_recorded_error is not None
|
||||
assert context.er_recorded_error.stack_summary is not None, (
|
||||
"Expected stack_summary to be set"
|
||||
)
|
||||
assert len(context.er_recorded_error.stack_summary) > 0
|
||||
|
||||
|
||||
@then("the service should not recommend retry for a clean plan")
|
||||
def step_then_no_retry_clean(context: Context) -> None:
|
||||
assert not context.er_service.should_retry(context.er_plan_id)
|
||||
|
||||
|
||||
@then("the service should not recommend escalation for a clean plan")
|
||||
def step_then_no_escalate_clean(context: Context) -> None:
|
||||
assert not context.er_service.should_escalate(context.er_plan_id)
|
||||
|
||||
|
||||
@then('the CLI recovery hints should contain "{text}"')
|
||||
def step_then_cli_hints_contain(context: Context, text: str) -> None:
|
||||
hints = context.er_service.format_recovery_hints_for_cli(context.er_plan_id)
|
||||
assert text in hints, f"Expected CLI hints to contain '{text}', got: {hints[:200]}"
|
||||
|
||||
|
||||
@then("the error history should not have retriable errors")
|
||||
def step_then_no_retriable(context: Context) -> None:
|
||||
history = context.er_service.get_error_history(context.er_plan_id)
|
||||
assert not history.has_retriable, "Expected no retriable errors"
|
||||
|
||||
|
||||
@then("the CLI recovery hints for a clean plan should be empty")
|
||||
def step_then_clean_plan_no_hints(context: Context) -> None:
|
||||
hints = context.er_service.format_recovery_hints_for_cli(context.er_plan_id)
|
||||
assert hints == "", f"Expected empty hints, got: {hints}"
|
||||
|
||||
|
||||
@given('an error record with stack summary "{summary}"')
|
||||
def step_given_record_with_stack(context: Context, summary: str) -> None:
|
||||
context.er_error_record = ErrorRecord(
|
||||
error_id="ERR-STACK-001",
|
||||
phase="execute",
|
||||
category=ErrorCategory.TRANSIENT,
|
||||
message="test error with stack",
|
||||
stack_summary=summary,
|
||||
retry_count=0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
|
||||
@given("a policy with threshold {threshold} and max retries {mr}")
|
||||
def step_given_policy(context: Context, threshold: str, mr: str) -> None:
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorRecoveryPolicy,
|
||||
)
|
||||
|
||||
context.er_policy = ErrorRecoveryPolicy(
|
||||
auto_retry_threshold=float(threshold),
|
||||
max_retries=int(mr),
|
||||
)
|
||||
|
||||
|
||||
@then("the policy should not recommend retry")
|
||||
def step_then_policy_no_retry(context: Context) -> None:
|
||||
assert context.er_error_record is not None
|
||||
assert not context.er_policy.should_retry(context.er_error_record), (
|
||||
"Expected should_retry=False"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# T1: Persistence failure path
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given("the lifecycle service update_error_details raises an error")
|
||||
def step_given_lifecycle_update_raises(context: Context) -> None:
|
||||
context.er_service._lifecycle.update_error_details.side_effect = RuntimeError(
|
||||
"DB connection lost"
|
||||
)
|
||||
|
||||
|
||||
@then("the error should still be recorded in memory")
|
||||
def step_then_error_in_memory(context: Context) -> None:
|
||||
history = context.er_service.get_error_history(context.er_plan_id)
|
||||
assert history is not None
|
||||
assert history.total_errors >= 1, "Expected at least 1 error in memory"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# T2: Branch coverage for policy and format_recovery_output
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@then("the policy should recommend escalation")
|
||||
def step_then_policy_escalation(context: Context) -> None:
|
||||
assert context.er_error_record is not None
|
||||
assert context.er_policy.should_escalate(context.er_error_record), (
|
||||
"Expected should_escalate=True"
|
||||
)
|
||||
|
||||
|
||||
@given(
|
||||
'a hinted error record with category "{cat}" and retry count {rc}'
|
||||
" and max retries {mr}"
|
||||
)
|
||||
def step_given_error_record_with_hints(
|
||||
context: Context,
|
||||
cat: str,
|
||||
rc: str,
|
||||
mr: str,
|
||||
) -> None:
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
RecoveryAction,
|
||||
RecoveryHint,
|
||||
)
|
||||
|
||||
hints = [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.RETRY,
|
||||
message="Retry the operation",
|
||||
cli_command="agents plan prompt TESTPLAN",
|
||||
priority=0,
|
||||
),
|
||||
]
|
||||
context.er_error_record = ErrorRecord(
|
||||
error_id="ERR-HINT-001",
|
||||
phase="execute",
|
||||
category=ErrorCategory(cat),
|
||||
message="test error with hints",
|
||||
retry_count=int(rc),
|
||||
max_retries=int(mr),
|
||||
recovery_hints=hints,
|
||||
)
|
||||
|
||||
|
||||
@then("the formatted recovery output should contain CLI command text")
|
||||
def step_then_format_contains_cli_cmd(context: Context) -> None:
|
||||
output = context.er_policy.format_recovery_output(context.er_error_record)
|
||||
assert "$ agents plan prompt" in output, (
|
||||
f"Expected CLI command in output, got: {output}"
|
||||
)
|
||||
|
||||
|
||||
@then('the formatted recovery output should contain "{text}"')
|
||||
def step_then_format_contains_text(context: Context, text: str) -> None:
|
||||
output = context.er_policy.format_recovery_output(context.er_error_record)
|
||||
assert text.lower() in output.lower(), f"Expected '{text}' in output, got: {output}"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# T4: Integration test with executor retry loop
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@given("a plan executor with error recovery service")
|
||||
def step_given_executor_with_recovery(context: Context) -> None:
|
||||
from cleveragents.application.services.plan_executor import PlanExecutor
|
||||
|
||||
# Build mock lifecycle service
|
||||
lifecycle = MagicMock()
|
||||
plan = _make_mock_plan(phase=PlanPhase.EXECUTE, state=ProcessingState.QUEUED)
|
||||
# Ensure real enum values (not MagicMock proxies)
|
||||
plan.phase = PlanPhase.EXECUTE
|
||||
plan.state = ProcessingState.QUEUED
|
||||
plan.decision_root_id = "01DECROOT00000000000001"
|
||||
plan.definition_of_done = "- Step one\n- Step two"
|
||||
lifecycle.get_plan.return_value = plan
|
||||
lifecycle.update_error_details = MagicMock()
|
||||
lifecycle._commit_plan = MagicMock()
|
||||
|
||||
# Build error recovery service wrapping the same lifecycle
|
||||
er_service = ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=0.0,
|
||||
max_retries=3,
|
||||
)
|
||||
|
||||
executor = PlanExecutor(
|
||||
lifecycle_service=lifecycle,
|
||||
error_recovery_service=er_service,
|
||||
)
|
||||
|
||||
context.er_executor = executor
|
||||
context.er_executor_lifecycle = lifecycle
|
||||
context.er_executor_recovery = er_service
|
||||
context.er_executor_plan_id = plan.identity.plan_id
|
||||
|
||||
|
||||
@given("the execute actor fails twice then succeeds")
|
||||
def step_given_actor_fails_twice(context: Context) -> None:
|
||||
call_count = {"n": 0}
|
||||
|
||||
def _side_effect(**kwargs):
|
||||
call_count["n"] += 1
|
||||
if call_count["n"] <= 2:
|
||||
raise RuntimeError("Connection timed out — transient failure")
|
||||
# Return a mock result on 3rd call
|
||||
result = MagicMock()
|
||||
result.changeset_id = "CS-001"
|
||||
result.sandbox_refs = []
|
||||
result.tool_calls_count = 1
|
||||
return result
|
||||
|
||||
context.er_executor._execute_actor.execute = MagicMock(side_effect=_side_effect)
|
||||
|
||||
|
||||
@when("I run the executor for the plan")
|
||||
def step_when_run_executor(context: Context) -> None:
|
||||
result = context.er_executor.run_execute(context.er_executor_plan_id)
|
||||
context.er_executor_result = result
|
||||
|
||||
|
||||
@then("the execute should complete after retries")
|
||||
def step_then_execute_completed(context: Context) -> None:
|
||||
assert context.er_executor_result is not None
|
||||
context.er_executor_lifecycle.complete_execute.assert_called_once()
|
||||
|
||||
|
||||
@then("the error recovery service should have recorded {n} errors")
|
||||
def step_then_recovery_recorded_n(context: Context, n: str) -> None:
|
||||
history = context.er_executor_recovery.get_error_history(
|
||||
context.er_executor_plan_id,
|
||||
)
|
||||
assert history.total_errors == int(n), (
|
||||
f"Expected {n} errors, got {history.total_errors}"
|
||||
)
|
||||
|
||||
|
||||
@given("a plan executor with error recovery service max retries {n}")
|
||||
def step_given_executor_max_retries(context: Context, n: str) -> None:
|
||||
from cleveragents.application.services.plan_executor import PlanExecutor
|
||||
|
||||
lifecycle = MagicMock()
|
||||
plan = _make_mock_plan(phase=PlanPhase.EXECUTE, state=ProcessingState.QUEUED)
|
||||
plan.phase = PlanPhase.EXECUTE
|
||||
plan.state = ProcessingState.QUEUED
|
||||
plan.decision_root_id = "01DECROOT00000000000001"
|
||||
plan.definition_of_done = "- Step one"
|
||||
lifecycle.get_plan.return_value = plan
|
||||
lifecycle.update_error_details = MagicMock()
|
||||
lifecycle._commit_plan = MagicMock()
|
||||
|
||||
er_service = ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=0.0,
|
||||
max_retries=int(n),
|
||||
)
|
||||
|
||||
executor = PlanExecutor(
|
||||
lifecycle_service=lifecycle,
|
||||
error_recovery_service=er_service,
|
||||
)
|
||||
|
||||
context.er_executor = executor
|
||||
context.er_executor_lifecycle = lifecycle
|
||||
context.er_executor_recovery = er_service
|
||||
context.er_executor_plan_id = plan.identity.plan_id
|
||||
|
||||
|
||||
@given('the execute actor always fails with "{msg}"')
|
||||
def step_given_actor_always_fails(context: Context, msg: str) -> None:
|
||||
context.er_executor._execute_actor.execute = MagicMock(
|
||||
side_effect=RuntimeError(f"Connection timed out — {msg}"),
|
||||
)
|
||||
|
||||
|
||||
@when("I run the executor expecting failure")
|
||||
def step_when_run_executor_failure(context: Context) -> None:
|
||||
try:
|
||||
context.er_executor.run_execute(context.er_executor_plan_id)
|
||||
context.er_executor_raised = False
|
||||
except RuntimeError:
|
||||
context.er_executor_raised = True
|
||||
|
||||
|
||||
@then("the lifecycle should record the plan as failed")
|
||||
def step_then_plan_failed(context: Context) -> None:
|
||||
assert context.er_executor_raised is True
|
||||
context.er_executor_lifecycle.fail_execute.assert_called_once()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# B5: Resource conflict classification
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@then('the classified error category should not be "{category}"')
|
||||
def step_then_category_not_eq(context: Context, category: str) -> None:
|
||||
assert context.er_classified_category is not None
|
||||
assert context.er_classified_category.value != category, (
|
||||
f"Expected category != '{category}', "
|
||||
f"got '{context.er_classified_category.value}'"
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# S1: Safe format-string substitution
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@when('I get recovery hints for category "{cat}" with plan_id "{pid}"')
|
||||
def step_when_hints_with_braces(context: Context, cat: str, pid: str) -> None:
|
||||
context.er_hints = get_recovery_hints(ErrorCategory(cat), pid)
|
||||
|
||||
|
||||
@then('the hints should contain "{text}" in CLI commands')
|
||||
def step_then_hints_contain_text(context: Context, text: str) -> None:
|
||||
cli_cmds = [h.cli_command for h in context.er_hints if h.cli_command]
|
||||
assert any(text in cmd for cmd in cli_cmds), (
|
||||
f"Expected '{text}' in CLI commands, got: {cli_cmds}"
|
||||
)
|
||||
@@ -6,6 +6,7 @@ Library OperatingSystem
|
||||
Library Process
|
||||
Library String
|
||||
Suite Setup Run Keywords Setup Test Environment AND Set Environment Variable CLEVERAGENTS_AUTO_APPLY_MIGRATIONS true
|
||||
Test Timeout 300 seconds
|
||||
|
||||
*** Variables ***
|
||||
# ${PYTHON} will be set in Setup Test Environment from common.resource
|
||||
@@ -19,7 +20,7 @@ Initialize New Project
|
||||
[Setup] Setup Test Directory
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME}
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT timeout=60s
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0 Init failed: ${result.stdout}
|
||||
Should Exist ${TEST_DIR}${/}.cleveragents
|
||||
@@ -37,13 +38,13 @@ Add Files To Context
|
||||
|
||||
# Add files to context
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context-load test.py main.py
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
# Verify files are in context
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context list
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Contain ${result.stdout} test.py
|
||||
Should Contain ${result.stdout} main.py
|
||||
@@ -55,13 +56,13 @@ Create Plan With Tell
|
||||
[Setup] Initialize Test Project
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents tell Add error handling
|
||||
... cwd=${TEST_DIR} timeout=60s
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
# Verify plan was created with correct name derived from prompt
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan current
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Contain ${result.stdout} add_error_handling
|
||||
|
||||
@@ -72,7 +73,7 @@ Build Plan
|
||||
[Setup] Initialize Test Project With Plan
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents build
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=60s
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
@@ -83,7 +84,7 @@ Apply Plan Changes
|
||||
[Setup] Initialize Test Project With Built Plan
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents apply --yes
|
||||
... cwd=${TEST_DIR} timeout=60s
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
@@ -94,13 +95,13 @@ Create New Empty Plan
|
||||
[Setup] Initialize Test Project
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan new feature-plan
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
# Verify new plan is current
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan current
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Contain ${result.stdout} feature-plan
|
||||
|
||||
@@ -111,7 +112,7 @@ Show Current Plan
|
||||
[Setup] Initialize Test Project
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan current
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} main
|
||||
@@ -123,7 +124,7 @@ List All Plans
|
||||
[Setup] Initialize Test Project With Multiple Plans
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan list
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} main
|
||||
@@ -138,13 +139,13 @@ Switch Between Plans
|
||||
|
||||
# Switch to feature-1
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan cd feature-1
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
# Verify switch
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan current
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Contain ${result.stdout} feature-1
|
||||
|
||||
@@ -166,7 +167,7 @@ List Context Files
|
||||
[Setup] Initialize Test Project With Context
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context list
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} test.py
|
||||
@@ -180,14 +181,14 @@ Show Context Content
|
||||
|
||||
# Show specific file content
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context show test.py
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} def hello
|
||||
|
||||
# Also test showing summary (without file argument)
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context show
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Context Summary
|
||||
@@ -201,13 +202,13 @@ Remove File From Context
|
||||
|
||||
# Remove test.py
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context rm test.py
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
# Verify removal
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context list
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
|
||||
Should Not Contain ${result.stdout} test.py
|
||||
Should Contain ${result.stdout} utils.py
|
||||
@@ -219,13 +220,13 @@ Clear All Context
|
||||
[Setup] Initialize Test Project With Context
|
||||
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context clear --yes
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT timeout=60s
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT timeout=120s
|
||||
|
||||
Should Be Equal As Integers ${result.rc} 0 Context clear failed: ${result.stdout}
|
||||
|
||||
# Verify context is empty
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context list
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT timeout=120s
|
||||
|
||||
Should Not Contain ${result.stdout} test.py
|
||||
Should Not Contain ${result.stdout} utils.py
|
||||
@@ -238,7 +239,7 @@ Command Error Handling
|
||||
|
||||
# Try to use command without project
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan tell test
|
||||
... cwd=${TEST_DIR} stderr=STDOUT
|
||||
... cwd=${TEST_DIR} stderr=STDOUT timeout=120s
|
||||
|
||||
Should Not Be Equal As Integers ${result.rc} 0
|
||||
|
||||
@@ -259,31 +260,31 @@ Initialize Test Project
|
||||
[Documentation] Initialize a test project
|
||||
Setup Test Directory
|
||||
${result}= Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME}
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
Initialize Test Project With Plan
|
||||
[Documentation] Initialize project and create a plan
|
||||
Initialize Test Project
|
||||
${result}= Run Process ${PYTHON} -m cleveragents tell Test instruction
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=60s
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=120s
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
Initialize Test Project With Built Plan
|
||||
[Documentation] Initialize project with a built plan
|
||||
Initialize Test Project With Plan
|
||||
${result}= Run Process ${PYTHON} -m cleveragents build
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=60s
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=120s
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
Initialize Test Project With Multiple Plans
|
||||
[Documentation] Initialize project with multiple plans
|
||||
Initialize Test Project
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan new feature-1
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
${result}= Run Process ${PYTHON} -m cleveragents plan new feature-2
|
||||
... cwd=${TEST_DIR}
|
||||
... cwd=${TEST_DIR} timeout=120s
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
|
||||
Initialize Test Project With Context
|
||||
@@ -292,5 +293,5 @@ Initialize Test Project With Context
|
||||
Create File ${TEST_DIR}${/}test.py def hello(): pass
|
||||
Create File ${TEST_DIR}${/}utils.py import os
|
||||
${result}= Run Process ${PYTHON} -m cleveragents context-load test.py utils.py
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT
|
||||
... cwd=${TEST_DIR} env:CLEVERAGENTS_AUTO_APPLY_MIGRATIONS=true stderr=STDOUT timeout=120s
|
||||
Should Be Equal As Integers ${result.rc} 0 context-load failed: ${result.stdout}
|
||||
|
||||
@@ -7,6 +7,7 @@ Library Collections
|
||||
Resource discovery_common.resource
|
||||
Suite Setup Run Keywords Clean Core CLI Temp Dir AND Setup Test Environment
|
||||
Suite Teardown Cleanup Test Environment
|
||||
Test Timeout 300 seconds
|
||||
|
||||
*** Variables ***
|
||||
${PYTHON} python
|
||||
@@ -16,7 +17,7 @@ ${PROJECT_NAME} test-project
|
||||
*** Test Cases ***
|
||||
Test CLI Help Shows All Commands
|
||||
[Documentation] Verify that CLI help displays all expected command groups
|
||||
${result} = Run Process ${PYTHON} -m cleveragents --help timeout=60s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents --help timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} AI-powered development assistant
|
||||
Should Contain ${result.stdout} project
|
||||
@@ -31,7 +32,7 @@ Test Project Initialization
|
||||
[Documentation] Test initializing a new CleverAgents project
|
||||
Create Directory ${TEST_DIR}/project1
|
||||
${result} = Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME}
|
||||
... cwd=${TEST_DIR}/project1 timeout=60s
|
||||
... cwd=${TEST_DIR}/project1 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Project '${PROJECT_NAME}' initialized successfully
|
||||
Directory Should Exist ${TEST_DIR}/project1/.cleveragents
|
||||
@@ -44,9 +45,9 @@ Test Project Initialization
|
||||
Test Project Cannot Initialize Twice
|
||||
[Documentation] Verify project cannot be initialized twice without force
|
||||
Create Directory ${TEST_DIR}/project2
|
||||
Run Process ${PYTHON} -m cleveragents init first-project cwd=${TEST_DIR}/project2 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init first-project cwd=${TEST_DIR}/project2 timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents init second-project
|
||||
... cwd=${TEST_DIR}/project2 timeout=60s
|
||||
... cwd=${TEST_DIR}/project2 timeout=120s
|
||||
Should Not Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stderr} Project already initialized
|
||||
Should Contain ${result.stderr} --force
|
||||
@@ -54,19 +55,19 @@ Test Project Cannot Initialize Twice
|
||||
Test Force Reinitialize Project
|
||||
[Documentation] Test force reinitialization of a project
|
||||
Create Directory ${TEST_DIR}/project3
|
||||
Run Process ${PYTHON} -m cleveragents init first-project cwd=${TEST_DIR}/project3 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init first-project cwd=${TEST_DIR}/project3 timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents init second-project --force
|
||||
... cwd=${TEST_DIR}/project3 timeout=60s
|
||||
... cwd=${TEST_DIR}/project3 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Project 'second-project' initialized successfully
|
||||
|
||||
Test Context Add Files
|
||||
[Documentation] Test adding files to context
|
||||
Create Directory ${TEST_DIR}/project4
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project4 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project4 timeout=120s
|
||||
Create File ${TEST_DIR}/project4/test.py print('hello world')
|
||||
${result} = Run Process ${PYTHON} -m cleveragents context add test.py
|
||||
... cwd=${TEST_DIR}/project4 timeout=60s
|
||||
... cwd=${TEST_DIR}/project4 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Added 1 file(s) to context
|
||||
Should Contain ${result.stdout} test.py
|
||||
@@ -74,13 +75,13 @@ Test Context Add Files
|
||||
Test Context List Files
|
||||
[Documentation] Test listing context files
|
||||
Create Directory ${TEST_DIR}/project5
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project5 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project5 timeout=120s
|
||||
Create File ${TEST_DIR}/project5/file1.py # File 1
|
||||
Create File ${TEST_DIR}/project5/file2.py # File 2
|
||||
Run Process ${PYTHON} -m cleveragents context add file1.py file2.py
|
||||
... cwd=${TEST_DIR}/project5 timeout=60s
|
||||
... cwd=${TEST_DIR}/project5 timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents context list
|
||||
... cwd=${TEST_DIR}/project5 timeout=60s
|
||||
... cwd=${TEST_DIR}/project5 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} file1.py
|
||||
Should Contain ${result.stdout} file2.py
|
||||
@@ -88,20 +89,20 @@ Test Context List Files
|
||||
Test Context Clear
|
||||
[Documentation] Test clearing context files
|
||||
Create Directory ${TEST_DIR}/project6
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project6 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project6 timeout=120s
|
||||
Create File ${TEST_DIR}/project6/test.py # Test file
|
||||
Run Process ${PYTHON} -m cleveragents context add test.py cwd=${TEST_DIR}/project6 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents context add test.py cwd=${TEST_DIR}/project6 timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents context clear --yes
|
||||
... cwd=${TEST_DIR}/project6 timeout=60s
|
||||
... cwd=${TEST_DIR}/project6 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Cleared all files from context
|
||||
|
||||
Test Plan Creation With Tell
|
||||
[Documentation] Test creating a plan using tell command
|
||||
Create Directory ${TEST_DIR}/project7
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project7 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project7 timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents tell Add error handling to main function
|
||||
... cwd=${TEST_DIR}/project7 timeout=60s
|
||||
... cwd=${TEST_DIR}/project7 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Plan created
|
||||
Should Contain ${result.stdout} error handling
|
||||
@@ -109,10 +110,10 @@ Test Plan Creation With Tell
|
||||
Test Plan Build
|
||||
[Documentation] Test building a plan
|
||||
Create Directory ${TEST_DIR}/project8
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project8 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents tell Create example code cwd=${TEST_DIR}/project8 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project8 timeout=120s
|
||||
Run Process ${PYTHON} -m cleveragents tell Create example code cwd=${TEST_DIR}/project8 timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents build cwd=${TEST_DIR}/project8
|
||||
... env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=60s
|
||||
... env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Plan built successfully
|
||||
Should Contain ${result.stdout} Generated
|
||||
@@ -121,11 +122,11 @@ Test Plan Build
|
||||
Test Plan Apply
|
||||
[Documentation] Test applying plan changes
|
||||
Create Directory ${TEST_DIR}/project9
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project9 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents tell Create example file cwd=${TEST_DIR}/project9 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project9 timeout=120s
|
||||
Run Process ${PYTHON} -m cleveragents tell Create example file cwd=${TEST_DIR}/project9 timeout=120s
|
||||
Run Process ${PYTHON} -m cleveragents build cwd=${TEST_DIR}/project9
|
||||
... env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=60s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents apply --yes cwd=${TEST_DIR}/project9 timeout=60s
|
||||
... env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents apply --yes cwd=${TEST_DIR}/project9 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Successfully applied
|
||||
File Should Exist ${TEST_DIR}/project9/example.py
|
||||
@@ -133,21 +134,21 @@ Test Plan Apply
|
||||
Test Plan List
|
||||
[Documentation] Test listing plans
|
||||
Create Directory ${TEST_DIR}/project10
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project10 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents tell First plan cwd=${TEST_DIR}/project10 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents plan new second-plan cwd=${TEST_DIR}/project10 timeout=60s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents plan list cwd=${TEST_DIR}/project10 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project10 timeout=120s
|
||||
Run Process ${PYTHON} -m cleveragents tell First plan cwd=${TEST_DIR}/project10 timeout=120s
|
||||
Run Process ${PYTHON} -m cleveragents plan new second-plan cwd=${TEST_DIR}/project10 timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents plan list cwd=${TEST_DIR}/project10 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Plans (3 total)
|
||||
|
||||
Test Shortcut Commands Work
|
||||
[Documentation] Test that shortcut commands work properly
|
||||
Create Directory ${TEST_DIR}/project11
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project11 timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init ${PROJECT_NAME} cwd=${TEST_DIR}/project11 timeout=120s
|
||||
Create File ${TEST_DIR}/project11/test.txt Test content
|
||||
# Test context-load shortcut
|
||||
${result} = Run Process ${PYTHON} -m cleveragents context-load test.txt
|
||||
... cwd=${TEST_DIR}/project11 timeout=60s
|
||||
... cwd=${TEST_DIR}/project11 timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Added 1 file(s) to context
|
||||
|
||||
@@ -156,30 +157,30 @@ Test End To End Workflow
|
||||
Create Directory ${TEST_DIR}/project12
|
||||
# Initialize project
|
||||
${init} = Run Process ${PYTHON} -m cleveragents init workflow-project
|
||||
... cwd=${TEST_DIR}/project12 timeout=60s
|
||||
... cwd=${TEST_DIR}/project12 timeout=120s
|
||||
Should Be Equal As Numbers ${init.rc} 0
|
||||
# Add context
|
||||
Create File ${TEST_DIR}/project12/input.py def main(): pass
|
||||
${add} = Run Process ${PYTHON} -m cleveragents context add input.py
|
||||
... cwd=${TEST_DIR}/project12 timeout=60s
|
||||
... cwd=${TEST_DIR}/project12 timeout=120s
|
||||
Should Be Equal As Numbers ${add.rc} 0
|
||||
# Create plan
|
||||
${tell} = Run Process ${PYTHON} -m cleveragents tell Add logging to main function
|
||||
... cwd=${TEST_DIR}/project12 timeout=60s
|
||||
... cwd=${TEST_DIR}/project12 timeout=120s
|
||||
Should Be Equal As Numbers ${tell.rc} 0
|
||||
# Build plan
|
||||
${build} = Run Process ${PYTHON} -m cleveragents build cwd=${TEST_DIR}/project12
|
||||
... env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=60s
|
||||
... env:CLEVERAGENTS_TESTING_USE_MOCK_AI=true timeout=120s
|
||||
Should Be Equal As Numbers ${build.rc} 0
|
||||
# Apply changes
|
||||
${apply} = Run Process ${PYTHON} -m cleveragents apply --yes cwd=${TEST_DIR}/project12 timeout=60s
|
||||
${apply} = Run Process ${PYTHON} -m cleveragents apply --yes cwd=${TEST_DIR}/project12 timeout=120s
|
||||
Should Be Equal As Numbers ${apply.rc} 0
|
||||
# Verify file created
|
||||
File Should Exist ${TEST_DIR}/project12/example.py
|
||||
|
||||
Test Command Error Handling
|
||||
[Documentation] Test error handling for invalid commands
|
||||
${result} = Run Process ${PYTHON} -m cleveragents invalid-command timeout=60s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents invalid-command timeout=120s
|
||||
Should Not Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stderr} Invalid command
|
||||
|
||||
@@ -187,16 +188,16 @@ Test Project Status Without Project
|
||||
[Documentation] Test project status command without initialized project
|
||||
Create Directory ${TEST_DIR}/no_project
|
||||
${result} = Run Process ${PYTHON} -m cleveragents project status
|
||||
... cwd=${TEST_DIR}/no_project timeout=60s
|
||||
... cwd=${TEST_DIR}/no_project timeout=120s
|
||||
Should Not Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stderr} No project found
|
||||
|
||||
Test Project Status With Project
|
||||
[Documentation] Test project status command with initialized project
|
||||
Create Directory ${TEST_DIR}/with_project
|
||||
Run Process ${PYTHON} -m cleveragents init status-test cwd=${TEST_DIR}/with_project timeout=60s
|
||||
Run Process ${PYTHON} -m cleveragents init status-test cwd=${TEST_DIR}/with_project timeout=120s
|
||||
${result} = Run Process ${PYTHON} -m cleveragents project status
|
||||
... cwd=${TEST_DIR}/with_project timeout=60s
|
||||
... cwd=${TEST_DIR}/with_project timeout=120s
|
||||
Should Be Equal As Numbers ${result.rc} 0
|
||||
Should Contain ${result.stdout} Project: status-test
|
||||
Should Contain ${result.stdout} Plans: 1
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
*** Settings ***
|
||||
Documentation Smoke tests for error recovery patterns,
|
||||
... including error classification, recording, retry policy,
|
||||
... output formatting, and model checks.
|
||||
Resource ${CURDIR}/common.resource
|
||||
Suite Setup Setup Test Environment
|
||||
Suite Teardown Cleanup Test Environment
|
||||
|
||||
*** Variables ***
|
||||
${HELPER} ${CURDIR}/helper_error_recovery.py
|
||||
|
||||
*** Test Cases ***
|
||||
Error Classification
|
||||
[Documentation] Verify error message classification into categories
|
||||
${result}= Run Process ${PYTHON} ${HELPER} classify cwd=${WORKSPACE}
|
||||
Log ${result.stdout}
|
||||
Log ${result.stderr}
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} classify-ok
|
||||
|
||||
Error Recording
|
||||
[Documentation] Verify error recording with metadata and recovery hints
|
||||
${result}= Run Process ${PYTHON} ${HELPER} record-error cwd=${WORKSPACE}
|
||||
Log ${result.stdout}
|
||||
Log ${result.stderr}
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} record-error-ok
|
||||
|
||||
Retry Policy Decisions
|
||||
[Documentation] Verify retry/escalation policy based on thresholds
|
||||
${result}= Run Process ${PYTHON} ${HELPER} retry-policy cwd=${WORKSPACE}
|
||||
Log ${result.stdout}
|
||||
Log ${result.stderr}
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} retry-policy-ok
|
||||
|
||||
Error Output Formatting
|
||||
[Documentation] Verify error output in plain and JSON formats
|
||||
${result}= Run Process ${PYTHON} ${HELPER} format-output cwd=${WORKSPACE}
|
||||
Log ${result.stdout}
|
||||
Log ${result.stderr}
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} format-output-ok
|
||||
|
||||
Error Recovery Model Checks
|
||||
[Documentation] Verify ErrorCategory, RecoveryAction, ErrorRecord, ErrorHistory models
|
||||
${result}= Run Process ${PYTHON} ${HELPER} model-checks cwd=${WORKSPACE}
|
||||
Log ${result.stdout}
|
||||
Log ${result.stderr}
|
||||
Should Be Equal As Integers ${result.rc} 0
|
||||
Should Contain ${result.stdout} model-checks-ok
|
||||
|
||||
Helper No Arguments
|
||||
[Documentation] Verify helper prints usage and exits 1 when called without args
|
||||
${result}= Run Process ${PYTHON} ${HELPER} cwd=${WORKSPACE}
|
||||
Log ${result.stdout}
|
||||
Should Be Equal As Integers ${result.rc} 1
|
||||
Should Contain ${result.stdout} Usage
|
||||
|
||||
Helper Unknown Command
|
||||
[Documentation] Verify helper exits 1 for an unknown sub-command
|
||||
${result}= Run Process ${PYTHON} ${HELPER} not-a-command cwd=${WORKSPACE}
|
||||
Log ${result.stdout}
|
||||
Should Be Equal As Integers ${result.rc} 1
|
||||
Should Contain ${result.stdout} Unknown command
|
||||
@@ -0,0 +1,213 @@
|
||||
"""Robot Framework helper for error recovery smoke tests.
|
||||
|
||||
Provides a CLI-style interface for Robot to invoke error recovery
|
||||
operations. Exit code 0 = success, 1 = failure.
|
||||
|
||||
Usage:
|
||||
python robot/helper_error_recovery.py classify
|
||||
python robot/helper_error_recovery.py record-error
|
||||
python robot/helper_error_recovery.py retry-policy
|
||||
python robot/helper_error_recovery.py format-output
|
||||
python robot/helper_error_recovery.py model-checks
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
# Ensure the src directory is on the import path.
|
||||
_SRC = str(Path(__file__).resolve().parents[1] / "src")
|
||||
if _SRC not in sys.path:
|
||||
sys.path.insert(0, _SRC)
|
||||
|
||||
from cleveragents.application.services.error_recovery_service import ( # noqa: E402
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
from cleveragents.domain.models.core.error_recovery import ( # noqa: E402
|
||||
ErrorCategory,
|
||||
ErrorHistory,
|
||||
ErrorRecord,
|
||||
RecoveryAction,
|
||||
RecoveryHint,
|
||||
classify_error,
|
||||
)
|
||||
from cleveragents.domain.models.core.plan import ( # noqa: E402
|
||||
PlanPhase,
|
||||
PlanTimestamps,
|
||||
ProcessingState,
|
||||
)
|
||||
|
||||
_PLAN_ID = "01ERRTEST000000000000001"
|
||||
|
||||
|
||||
def _make_plan() -> MagicMock:
|
||||
plan = MagicMock()
|
||||
plan.identity.plan_id = _PLAN_ID
|
||||
plan.phase = PlanPhase.EXECUTE
|
||||
plan.processing_state = ProcessingState.ERRORED
|
||||
plan.is_terminal = False
|
||||
plan.error_message = None
|
||||
plan.error_details = None
|
||||
plan.timestamps = PlanTimestamps()
|
||||
return plan
|
||||
|
||||
|
||||
def _make_service(
|
||||
plan: MagicMock,
|
||||
threshold: float = 0.0,
|
||||
max_retries: int = 3,
|
||||
) -> ErrorRecoveryService:
|
||||
lifecycle = MagicMock()
|
||||
lifecycle.get_plan.return_value = plan
|
||||
lifecycle._commit_plan = MagicMock()
|
||||
return ErrorRecoveryService(
|
||||
lifecycle_service=lifecycle,
|
||||
auto_retry_threshold=threshold,
|
||||
max_retries=max_retries,
|
||||
)
|
||||
|
||||
|
||||
def _cmd_classify() -> int:
|
||||
"""Test error classification."""
|
||||
assert classify_error("Connection timed out") == ErrorCategory.TRANSIENT
|
||||
assert classify_error("Schema validation failed") == ErrorCategory.VALIDATION
|
||||
assert classify_error("Missing configuration") == ErrorCategory.CONFIGURATION
|
||||
assert classify_error("Merge conflict in file") == ErrorCategory.MERGE_CONFLICT
|
||||
assert classify_error("Authentication failed") == ErrorCategory.AUTHENTICATION
|
||||
assert classify_error("Model not available") == ErrorCategory.PROVIDER
|
||||
assert classify_error("Unexpected internal error") == ErrorCategory.INTERNAL
|
||||
assert classify_error("Something novel") == ErrorCategory.UNKNOWN
|
||||
# Exception type classification
|
||||
assert classify_error("fail", "RateLimitError") == ErrorCategory.TRANSIENT
|
||||
print("classify-ok")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_record_error() -> int:
|
||||
"""Test error recording."""
|
||||
plan = _make_plan()
|
||||
svc = _make_service(plan)
|
||||
record = svc.record_error(
|
||||
plan_id=_PLAN_ID,
|
||||
phase="execute",
|
||||
message="Connection timed out after 30s",
|
||||
)
|
||||
assert record.category == ErrorCategory.TRANSIENT
|
||||
assert record.phase == "execute"
|
||||
assert len(record.recovery_hints) > 0
|
||||
|
||||
history = svc.get_error_history(_PLAN_ID)
|
||||
assert history.total_errors == 1
|
||||
|
||||
print("record-error-ok")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_retry_policy() -> int:
|
||||
"""Test retry policy decisions."""
|
||||
plan = _make_plan()
|
||||
# Auto-retry enabled
|
||||
svc = _make_service(plan, threshold=0.0)
|
||||
svc.record_error(_PLAN_ID, "execute", "Connection timed out")
|
||||
assert svc.should_retry(_PLAN_ID), "Expected should_retry=True"
|
||||
|
||||
# Auto-retry disabled (human threshold)
|
||||
svc2 = _make_service(plan, threshold=1.0)
|
||||
svc2.record_error(_PLAN_ID, "execute", "Connection timed out")
|
||||
assert svc2.should_escalate(_PLAN_ID), "Expected should_escalate=True"
|
||||
|
||||
print("retry-policy-ok")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_format_output() -> int:
|
||||
"""Test error output formatting."""
|
||||
plan = _make_plan()
|
||||
svc = _make_service(plan)
|
||||
svc.record_error(_PLAN_ID, "execute", "Provider timed out")
|
||||
|
||||
plain = svc.format_error_output(_PLAN_ID, fmt="plain")
|
||||
assert "transient" in plain
|
||||
assert "Provider timed out" in plain
|
||||
|
||||
json_out = svc.format_error_output(_PLAN_ID, fmt="json")
|
||||
assert "plan_id" in json_out
|
||||
|
||||
print("format-output-ok")
|
||||
return 0
|
||||
|
||||
|
||||
def _cmd_model_checks() -> int:
|
||||
"""Test model construction and enums."""
|
||||
# ErrorCategory values
|
||||
assert ErrorCategory.TRANSIENT == "transient"
|
||||
assert ErrorCategory.VALIDATION == "validation"
|
||||
assert ErrorCategory.UNKNOWN == "unknown"
|
||||
|
||||
# RecoveryAction values
|
||||
assert RecoveryAction.RETRY == "retry"
|
||||
assert RecoveryAction.CANCEL == "cancel"
|
||||
|
||||
# RecoveryHint construction
|
||||
hint = RecoveryHint(
|
||||
action=RecoveryAction.RETRY,
|
||||
message="Retry the operation",
|
||||
cli_command="agents plan prompt TEST",
|
||||
priority=0,
|
||||
)
|
||||
assert hint.action == RecoveryAction.RETRY
|
||||
|
||||
# ErrorRecord construction and properties
|
||||
record = ErrorRecord(
|
||||
error_id="ERR-001",
|
||||
phase="execute",
|
||||
category=ErrorCategory.TRANSIENT,
|
||||
message="test",
|
||||
retry_count=0,
|
||||
max_retries=3,
|
||||
)
|
||||
assert record.is_retriable
|
||||
assert not record.retries_exhausted
|
||||
|
||||
# ErrorHistory construction
|
||||
history = ErrorHistory(plan_id="TEST")
|
||||
history.add(record)
|
||||
assert history.total_errors == 1
|
||||
assert history.latest == record
|
||||
|
||||
# to_cli_dict
|
||||
cli_dict = record.to_cli_dict()
|
||||
assert "error_id" in cli_dict
|
||||
assert "category" in cli_dict
|
||||
|
||||
print("model-checks-ok")
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
if len(sys.argv) < 2:
|
||||
print(
|
||||
"Usage: helper_error_recovery.py "
|
||||
"<classify|record-error|retry-policy|format-output|model-checks>"
|
||||
)
|
||||
return 1
|
||||
|
||||
cmd = sys.argv[1]
|
||||
commands = {
|
||||
"classify": _cmd_classify,
|
||||
"record-error": _cmd_record_error,
|
||||
"retry-policy": _cmd_retry_policy,
|
||||
"format-output": _cmd_format_output,
|
||||
"model-checks": _cmd_model_checks,
|
||||
}
|
||||
func = commands.get(cmd)
|
||||
if func is None:
|
||||
print(f"Unknown command: {cmd}")
|
||||
return 1
|
||||
return func()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,346 @@
|
||||
"""Error recovery service for plan execution failures.
|
||||
|
||||
Provides the service layer for recording errors, generating recovery
|
||||
hints, managing retry state, and formatting error output for the CLI.
|
||||
|
||||
Integrates with:
|
||||
- ``PlanLifecycleService`` for plan lookups and state transitions.
|
||||
- ``ErrorRecoveryPolicy`` for retry/escalation decisions.
|
||||
- ``ErrorHistory`` for tracking error records across plan attempts.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import traceback
|
||||
from datetime import UTC, datetime
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import structlog
|
||||
from ulid import ULID
|
||||
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
ErrorHistory,
|
||||
ErrorRecord,
|
||||
ErrorRecoveryPolicy,
|
||||
classify_error,
|
||||
get_recovery_hints,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from cleveragents.application.services.plan_lifecycle_service import (
|
||||
PlanLifecycleService,
|
||||
)
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
|
||||
class ErrorRecoveryService:
|
||||
"""Service for error recording, recovery hints, and retry management.
|
||||
|
||||
Bridges ``PlanLifecycleService`` with structured error tracking and
|
||||
recovery hint generation.
|
||||
|
||||
Attributes:
|
||||
_lifecycle: The lifecycle service for plan lookups.
|
||||
_histories: In-memory store of error histories keyed by plan_id.
|
||||
_policy: The error recovery policy for retry decisions.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
lifecycle_service: PlanLifecycleService,
|
||||
auto_retry_threshold: float = 0.0,
|
||||
max_retries: int = 3,
|
||||
) -> None:
|
||||
"""Initialise the error recovery service.
|
||||
|
||||
Args:
|
||||
lifecycle_service: For plan lookups and state transitions.
|
||||
auto_retry_threshold: Automation profile threshold for
|
||||
auto-retry (0.0 = always auto, 1.0 = always human).
|
||||
max_retries: Maximum retry attempts per error.
|
||||
"""
|
||||
self._lifecycle = lifecycle_service
|
||||
self._histories: dict[str, ErrorHistory] = {}
|
||||
self._policy = ErrorRecoveryPolicy(
|
||||
auto_retry_threshold=auto_retry_threshold,
|
||||
max_retries=max_retries,
|
||||
)
|
||||
self._logger = logger.bind(service="error_recovery")
|
||||
|
||||
# -- Record errors -------------------------------------------------------
|
||||
|
||||
def record_error(
|
||||
self,
|
||||
plan_id: str,
|
||||
phase: str,
|
||||
message: str,
|
||||
*,
|
||||
exception: BaseException | None = None,
|
||||
exception_type: str | None = None,
|
||||
actor: str | None = None,
|
||||
tool_call: str | None = None,
|
||||
category: ErrorCategory | None = None,
|
||||
) -> ErrorRecord:
|
||||
"""Record an error event for a plan.
|
||||
|
||||
Classifies the error, generates recovery hints, and appends
|
||||
the record to the plan's error history.
|
||||
|
||||
Also updates the plan's ``error_details`` field with structured
|
||||
metadata via the lifecycle service.
|
||||
|
||||
Args:
|
||||
plan_id: The plan ULID.
|
||||
phase: Phase when error occurred (strategize/execute/apply).
|
||||
message: Error message.
|
||||
exception: Optional exception object (for stack trace).
|
||||
exception_type: Optional exception class name (for
|
||||
classification if exception is not provided).
|
||||
actor: Optional actor name involved.
|
||||
tool_call: Optional tool call identifier.
|
||||
category: Optional explicit category override. If not
|
||||
provided, the error is classified automatically.
|
||||
|
||||
Returns:
|
||||
The created ``ErrorRecord``.
|
||||
"""
|
||||
# Classify
|
||||
exc_type_name = exception_type
|
||||
if exc_type_name is None and exception is not None:
|
||||
exc_type_name = type(exception).__name__
|
||||
resolved_category = category or classify_error(message, exc_type_name)
|
||||
|
||||
# Stack summary
|
||||
stack_summary: str | None = None
|
||||
if exception is not None:
|
||||
frames = traceback.format_exception(
|
||||
type(exception), exception, exception.__traceback__
|
||||
)
|
||||
# Keep last 3 frames
|
||||
stack_summary = "".join(frames[-3:]).strip()
|
||||
|
||||
# Get retry count from history
|
||||
history = self._get_or_create_history(plan_id)
|
||||
# Count retries for same category
|
||||
same_category_retries = sum(
|
||||
1 for r in history.records if r.category == resolved_category
|
||||
)
|
||||
|
||||
# Generate recovery hints
|
||||
hints = get_recovery_hints(resolved_category, plan_id)
|
||||
|
||||
# Create record
|
||||
record = ErrorRecord(
|
||||
error_id=str(ULID()),
|
||||
phase=phase,
|
||||
category=resolved_category,
|
||||
message=message,
|
||||
actor=actor,
|
||||
tool_call=tool_call,
|
||||
stack_summary=stack_summary,
|
||||
recovery_hints=hints,
|
||||
retry_count=same_category_retries,
|
||||
max_retries=self._policy.max_retries,
|
||||
timestamp=datetime.now(tz=UTC),
|
||||
)
|
||||
|
||||
history.add(record)
|
||||
|
||||
self._logger.info(
|
||||
"Error recorded",
|
||||
plan_id=plan_id,
|
||||
error_id=record.error_id,
|
||||
phase=phase,
|
||||
category=resolved_category.value,
|
||||
retry_count=record.retry_count,
|
||||
is_retriable=record.is_retriable,
|
||||
)
|
||||
|
||||
# Persist structured metadata to plan
|
||||
self._persist_error_metadata(plan_id, record)
|
||||
|
||||
return record
|
||||
|
||||
# -- Query error history -------------------------------------------------
|
||||
|
||||
def get_error_history(self, plan_id: str) -> ErrorHistory:
|
||||
"""Return the error history for a plan.
|
||||
|
||||
Args:
|
||||
plan_id: The plan ULID.
|
||||
|
||||
Returns:
|
||||
The ``ErrorHistory`` for the plan (empty if no errors).
|
||||
"""
|
||||
return self._get_or_create_history(plan_id)
|
||||
|
||||
def get_latest_error(self, plan_id: str) -> ErrorRecord | None:
|
||||
"""Return the most recent error for a plan.
|
||||
|
||||
Args:
|
||||
plan_id: The plan ULID.
|
||||
|
||||
Returns:
|
||||
The latest ``ErrorRecord``, or ``None`` if no errors.
|
||||
"""
|
||||
history = self._get_or_create_history(plan_id)
|
||||
return history.latest
|
||||
|
||||
# -- Recovery decisions ---------------------------------------------------
|
||||
|
||||
def should_retry(self, plan_id: str) -> bool:
|
||||
"""Check whether the latest error for a plan should be retried.
|
||||
|
||||
Args:
|
||||
plan_id: The plan ULID.
|
||||
|
||||
Returns:
|
||||
``True`` if automatic retry is appropriate.
|
||||
"""
|
||||
latest = self.get_latest_error(plan_id)
|
||||
if latest is None:
|
||||
return False
|
||||
return self._policy.should_retry(latest)
|
||||
|
||||
def should_escalate(self, plan_id: str) -> bool:
|
||||
"""Check whether the latest error requires human intervention.
|
||||
|
||||
Args:
|
||||
plan_id: The plan ULID.
|
||||
|
||||
Returns:
|
||||
``True`` if human intervention is required.
|
||||
"""
|
||||
latest = self.get_latest_error(plan_id)
|
||||
if latest is None:
|
||||
return False
|
||||
return self._policy.should_escalate(latest)
|
||||
|
||||
# -- Formatting for CLI ---------------------------------------------------
|
||||
|
||||
def format_error_output(
|
||||
self,
|
||||
plan_id: str,
|
||||
fmt: str = "rich",
|
||||
) -> str:
|
||||
"""Format the error history for CLI display.
|
||||
|
||||
Args:
|
||||
plan_id: The plan ULID.
|
||||
fmt: Output format (``rich``, ``plain``, ``json``).
|
||||
|
||||
Returns:
|
||||
Formatted error output string.
|
||||
"""
|
||||
import json as json_mod
|
||||
|
||||
history = self.get_error_history(plan_id)
|
||||
|
||||
if fmt == "json":
|
||||
return json_mod.dumps(history.to_cli_dict(), indent=2, default=str)
|
||||
|
||||
if history.total_errors == 0:
|
||||
return f"No errors recorded for plan {plan_id}."
|
||||
|
||||
lines: list[str] = []
|
||||
lines.append(f"Error History for Plan {plan_id}")
|
||||
lines.append(
|
||||
f"Total errors: {history.total_errors} | "
|
||||
f"Total retries: {history.total_retries}"
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
for i, record in enumerate(history.records, 1):
|
||||
lines.append(f"--- Error #{i} ---")
|
||||
lines.append(self._policy.format_recovery_output(record))
|
||||
lines.append("")
|
||||
|
||||
return "\n".join(lines).rstrip()
|
||||
|
||||
def format_recovery_hints_for_cli(
|
||||
self,
|
||||
plan_id: str,
|
||||
) -> str:
|
||||
"""Format recovery hints for the latest error as CLI output.
|
||||
|
||||
This is the output appended to error messages in the CLI
|
||||
to guide users on what to do next.
|
||||
|
||||
Args:
|
||||
plan_id: The plan ULID.
|
||||
|
||||
Returns:
|
||||
Formatted recovery hints string, or empty string if
|
||||
no errors.
|
||||
"""
|
||||
latest = self.get_latest_error(plan_id)
|
||||
if latest is None:
|
||||
return ""
|
||||
|
||||
return self._policy.format_recovery_output(latest)
|
||||
|
||||
# -- Internal helpers ----------------------------------------------------
|
||||
|
||||
def _get_or_create_history(self, plan_id: str) -> ErrorHistory:
|
||||
"""Get or create an error history for a plan."""
|
||||
if plan_id not in self._histories:
|
||||
self._histories[plan_id] = ErrorHistory(plan_id=plan_id)
|
||||
return self._histories[plan_id]
|
||||
|
||||
def _persist_error_metadata(
|
||||
self,
|
||||
plan_id: str,
|
||||
record: ErrorRecord,
|
||||
) -> None:
|
||||
"""Store structured error metadata in the plan's error_details.
|
||||
|
||||
Updates the plan via the lifecycle service to persist the
|
||||
error classification, retry count, and latest recovery hint.
|
||||
"""
|
||||
try:
|
||||
details: dict[str, str] = {
|
||||
"error_category": record.category.value,
|
||||
"error_phase": record.phase,
|
||||
"retry_count": str(record.retry_count),
|
||||
"max_retries": str(record.max_retries),
|
||||
"is_retriable": str(record.is_retriable).lower(),
|
||||
}
|
||||
|
||||
if record.actor:
|
||||
details["error_actor"] = record.actor
|
||||
if record.tool_call:
|
||||
details["error_tool_call"] = record.tool_call
|
||||
if record.stack_summary:
|
||||
# Truncate to fit in dict[str, str] value
|
||||
details["stack_summary"] = record.stack_summary[:500]
|
||||
|
||||
if record.recovery_hints:
|
||||
top_hint = sorted(record.recovery_hints, key=lambda h: h.priority)[0]
|
||||
details["recovery_action"] = top_hint.action.value
|
||||
details["recovery_hint"] = top_hint.message
|
||||
if top_hint.cli_command:
|
||||
details["recovery_command"] = top_hint.cli_command
|
||||
|
||||
self._lifecycle.update_error_details(plan_id, details)
|
||||
|
||||
self._logger.debug(
|
||||
"Error metadata persisted to plan",
|
||||
plan_id=plan_id,
|
||||
error_category=record.category.value,
|
||||
)
|
||||
except (
|
||||
OSError,
|
||||
ValueError,
|
||||
KeyError,
|
||||
AttributeError,
|
||||
RuntimeError,
|
||||
):
|
||||
# Best-effort: don't crash the error-recording flow if
|
||||
# persistence fails (e.g. DB down, plan already deleted).
|
||||
self._logger.error(
|
||||
"Failed to persist error metadata to plan",
|
||||
plan_id=plan_id,
|
||||
exc_info=True,
|
||||
)
|
||||
@@ -10,8 +10,8 @@ from __future__ import annotations
|
||||
|
||||
import traceback
|
||||
from collections.abc import Callable
|
||||
from datetime import datetime
|
||||
from typing import Any
|
||||
from datetime import UTC, datetime
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import structlog
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
@@ -32,6 +32,11 @@ from cleveragents.domain.models.core.plan import (
|
||||
from cleveragents.tool.builtins.changeset import ChangeSet, ChangeSetCapture
|
||||
from cleveragents.tool.runner import ToolRunner
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from cleveragents.application.services.error_recovery_service import (
|
||||
ErrorRecoveryService,
|
||||
)
|
||||
|
||||
logger = structlog.get_logger(__name__)
|
||||
|
||||
StreamCallback = Callable[[str, dict[str, Any]], None]
|
||||
@@ -241,13 +246,25 @@ class PlanExecutor:
|
||||
tool_runner: ToolRunner | None = None,
|
||||
sandbox_root: str | None = None,
|
||||
execution_context: PlanExecutionContext | None = None,
|
||||
error_recovery_service: ErrorRecoveryService | None = None,
|
||||
) -> None:
|
||||
"""Initialize the plan executor.
|
||||
|
||||
Args:
|
||||
lifecycle_service: The ``PlanLifecycleService`` instance.
|
||||
tool_runner: Optional ``ToolRunner`` for execute phase.
|
||||
sandbox_root: Optional sandbox root directory.
|
||||
execution_context: Optional execution context for runtime mode.
|
||||
error_recovery_service: Optional error recovery service for
|
||||
recording errors and managing retry logic.
|
||||
"""
|
||||
if lifecycle_service is None:
|
||||
raise ValidationError("lifecycle_service must not be None")
|
||||
self._lifecycle = lifecycle_service
|
||||
self._tool_runner = tool_runner
|
||||
self._sandbox_root = sandbox_root
|
||||
self._execution_context = execution_context
|
||||
self._error_recovery = error_recovery_service
|
||||
self._strategize_actor = StrategizeStubActor()
|
||||
self._execute_actor = ExecuteStubActor()
|
||||
self._logger = logger.bind(service="plan_executor")
|
||||
@@ -295,7 +312,11 @@ class PlanExecutor:
|
||||
)
|
||||
plan = self._lifecycle.get_plan(plan_id)
|
||||
plan.decision_root_id = result.decision_root_id
|
||||
plan.timestamps.updated_at = datetime.now()
|
||||
plan.timestamps.updated_at = datetime.now(tz=UTC)
|
||||
|
||||
# Store decisions and invariant records in error_details as
|
||||
# structured metadata (Plan model uses error_details for
|
||||
# arbitrary metadata storage until a dedicated field lands)
|
||||
plan.error_details = {
|
||||
"strategy_decisions": str(len(result.decisions)),
|
||||
"invariant_records": str(len(result.invariant_records)),
|
||||
@@ -312,6 +333,16 @@ class PlanExecutor:
|
||||
)
|
||||
return result
|
||||
except Exception as exc:
|
||||
# Record via error recovery service if available
|
||||
if self._error_recovery is not None:
|
||||
self._error_recovery.record_error(
|
||||
plan_id=plan_id,
|
||||
phase="strategize",
|
||||
message=str(exc),
|
||||
exception=exc,
|
||||
exception_type=type(exc).__name__,
|
||||
)
|
||||
|
||||
error_msg = f"{type(exc).__name__}: {exc}"
|
||||
plan = self._lifecycle.get_plan(plan_id)
|
||||
plan.error_details = {
|
||||
@@ -396,7 +427,7 @@ class PlanExecutor:
|
||||
"execution_duration_ms": str(result.execution_duration_ms),
|
||||
"mode": "runtime",
|
||||
}
|
||||
plan.timestamps.updated_at = datetime.now()
|
||||
plan.timestamps.updated_at = datetime.now(tz=UTC)
|
||||
self._lifecycle._commit_plan(plan)
|
||||
self._lifecycle.complete_execute(plan_id)
|
||||
self._logger.info(
|
||||
@@ -423,45 +454,83 @@ class PlanExecutor:
|
||||
plan_id: str,
|
||||
stream_callback: StreamCallback | None = None,
|
||||
) -> ExecuteResult:
|
||||
"""Execute using the legacy ExecuteStubActor."""
|
||||
"""Execute using the legacy ExecuteStubActor with optional retry."""
|
||||
plan = self._guard_execute(plan_id)
|
||||
decisions = self._build_decisions(plan)
|
||||
|
||||
self._lifecycle.start_execute(plan_id)
|
||||
try:
|
||||
result = self._execute_actor.execute(
|
||||
plan_id=plan_id,
|
||||
decisions=decisions,
|
||||
tool_runner=self._tool_runner,
|
||||
sandbox_root=self._sandbox_root,
|
||||
stream_callback=stream_callback,
|
||||
)
|
||||
plan = self._lifecycle.get_plan(plan_id)
|
||||
plan.changeset_id = result.changeset_id
|
||||
plan.sandbox_refs = result.sandbox_refs
|
||||
plan.error_details = {
|
||||
"tool_calls_count": str(result.tool_calls_count),
|
||||
"sandbox_refs_count": str(len(result.sandbox_refs)),
|
||||
"mode": "stub",
|
||||
}
|
||||
plan.timestamps.updated_at = datetime.now()
|
||||
self._lifecycle._commit_plan(plan)
|
||||
self._lifecycle.complete_execute(plan_id)
|
||||
self._logger.info(
|
||||
"Execute completed (stub)",
|
||||
plan_id=plan_id,
|
||||
changeset_id=result.changeset_id,
|
||||
tool_calls=result.tool_calls_count,
|
||||
)
|
||||
return result
|
||||
except Exception as exc:
|
||||
error_msg = f"{type(exc).__name__}: {exc}"
|
||||
plan = self._lifecycle.get_plan(plan_id)
|
||||
plan.error_details = {
|
||||
"exception_type": type(exc).__name__,
|
||||
"traceback": traceback.format_exc(),
|
||||
"mode": "stub",
|
||||
}
|
||||
self._lifecycle._commit_plan(plan)
|
||||
self._lifecycle.fail_execute(plan_id, error_msg)
|
||||
raise
|
||||
|
||||
# Determine max attempts: 1 (no recovery) or policy max_retries + 1.
|
||||
max_attempts = (
|
||||
self._error_recovery._policy.max_retries + 1
|
||||
if self._error_recovery is not None
|
||||
else 1
|
||||
)
|
||||
last_exc: Exception | None = None
|
||||
|
||||
for attempt in range(max_attempts):
|
||||
try:
|
||||
result = self._execute_actor.execute(
|
||||
plan_id=plan_id,
|
||||
decisions=decisions,
|
||||
tool_runner=self._tool_runner,
|
||||
sandbox_root=self._sandbox_root,
|
||||
stream_callback=stream_callback,
|
||||
)
|
||||
plan = self._lifecycle.get_plan(plan_id)
|
||||
plan.changeset_id = result.changeset_id
|
||||
plan.sandbox_refs = result.sandbox_refs
|
||||
plan.error_details = {
|
||||
"tool_calls_count": str(result.tool_calls_count),
|
||||
"sandbox_refs_count": str(len(result.sandbox_refs)),
|
||||
"mode": "stub",
|
||||
}
|
||||
plan.timestamps.updated_at = datetime.now(tz=UTC)
|
||||
self._lifecycle._commit_plan(plan)
|
||||
self._lifecycle.complete_execute(plan_id)
|
||||
self._logger.info(
|
||||
"Execute completed (stub)",
|
||||
plan_id=plan_id,
|
||||
changeset_id=result.changeset_id,
|
||||
tool_calls=result.tool_calls_count,
|
||||
)
|
||||
return result
|
||||
|
||||
except Exception as exc:
|
||||
last_exc = exc
|
||||
|
||||
# Record error via recovery service if available
|
||||
if self._error_recovery is not None:
|
||||
self._error_recovery.record_error(
|
||||
plan_id=plan_id,
|
||||
phase="execute",
|
||||
message=str(exc),
|
||||
exception=exc,
|
||||
exception_type=type(exc).__name__,
|
||||
)
|
||||
# Check if retry is available
|
||||
if self._error_recovery.should_retry(plan_id):
|
||||
self._logger.info(
|
||||
"Retrying execute phase after recoverable error",
|
||||
plan_id=plan_id,
|
||||
attempt=attempt + 1,
|
||||
max_attempts=max_attempts,
|
||||
error=str(exc),
|
||||
)
|
||||
continue
|
||||
|
||||
# No recovery service or no more retries — fail
|
||||
break
|
||||
|
||||
# All attempts exhausted — persist error and fail
|
||||
assert last_exc is not None
|
||||
error_msg = f"{type(last_exc).__name__}: {last_exc}"
|
||||
plan = self._lifecycle.get_plan(plan_id)
|
||||
plan.error_details = {
|
||||
"exception_type": type(last_exc).__name__,
|
||||
"traceback": traceback.format_exc(),
|
||||
"mode": "stub",
|
||||
}
|
||||
self._lifecycle._commit_plan(plan)
|
||||
self._lifecycle.fail_execute(plan_id, error_msg)
|
||||
raise last_exc
|
||||
|
||||
@@ -217,6 +217,29 @@ class PlanLifecycleService:
|
||||
with self.unit_of_work.transaction() as ctx:
|
||||
self._persist_plan_update(plan, ctx)
|
||||
|
||||
def update_error_details(
|
||||
self,
|
||||
plan_id: str,
|
||||
error_details: dict[str, str],
|
||||
) -> None:
|
||||
"""Merge *error_details* into the plan and persist.
|
||||
|
||||
This is the public API for external services (e.g.
|
||||
``ErrorRecoveryService``) to attach structured error metadata
|
||||
to a plan without reaching into private persistence methods.
|
||||
|
||||
Args:
|
||||
plan_id: The plan to update.
|
||||
error_details: Key/value pairs to merge into the plan's
|
||||
``error_details`` dict.
|
||||
"""
|
||||
plan = self.get_plan(plan_id)
|
||||
existing = dict(plan.error_details) if plan.error_details else {}
|
||||
existing.update(error_details)
|
||||
plan.error_details = existing
|
||||
plan.timestamps.updated_at = datetime.now()
|
||||
self._commit_plan(plan)
|
||||
|
||||
def _generate_ulid(self) -> str:
|
||||
"""Generate a new ULID string."""
|
||||
return str(ULID())
|
||||
|
||||
@@ -15,6 +15,7 @@ plan lifecycle.
|
||||
| ``agents plan cancel`` | Cancel a non-terminal plan |
|
||||
| ``agents plan diff`` | Show ChangeSet as unified diff |
|
||||
| ``agents plan artifacts`` | Show ChangeSet ID, sandbox refs, summary|
|
||||
| ``agents plan errors`` | Show error decisions + recovery hints |
|
||||
|
||||
## Deprecated Legacy Commands
|
||||
|
||||
@@ -1676,6 +1677,129 @@ def plan_status(
|
||||
raise typer.Abort() from e
|
||||
|
||||
|
||||
@app.command("errors")
|
||||
def plan_errors(
|
||||
plan_id: Annotated[
|
||||
str,
|
||||
typer.Argument(help="Plan ID to show errors for"),
|
||||
],
|
||||
fmt: Annotated[
|
||||
str,
|
||||
typer.Option(
|
||||
"--format",
|
||||
"-f",
|
||||
help=_FORMAT_HELP,
|
||||
),
|
||||
] = "rich",
|
||||
) -> None:
|
||||
"""Show error decisions with recovery hints and retry history.
|
||||
|
||||
Displays structured error metadata from the plan, including error
|
||||
category, phase, retry count, and recovery suggestions with CLI
|
||||
commands.
|
||||
"""
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
get_recovery_hints,
|
||||
)
|
||||
|
||||
try:
|
||||
service = _get_lifecycle_service()
|
||||
plan = service.get_plan(plan_id)
|
||||
|
||||
details = plan.error_details or {}
|
||||
has_error_recovery = "error_category" in details
|
||||
|
||||
# Build recovery hints from stored category
|
||||
hints: list[dict[str, Any]] = []
|
||||
if has_error_recovery:
|
||||
with suppress(ValueError):
|
||||
cat = ErrorCategory(details["error_category"])
|
||||
for h in get_recovery_hints(cat, plan_id):
|
||||
entry: dict[str, Any] = {
|
||||
"action": h.action.value,
|
||||
"message": h.message,
|
||||
}
|
||||
if h.cli_command:
|
||||
entry["cli_command"] = h.cli_command
|
||||
hints.append(entry)
|
||||
|
||||
if fmt != OutputFormat.RICH.value:
|
||||
data: dict[str, Any] = {
|
||||
"plan_id": plan_id,
|
||||
"phase": plan.phase.value,
|
||||
"state": plan.state.value if plan.state else "unknown",
|
||||
}
|
||||
if plan.error_message:
|
||||
data["error_message"] = plan.error_message
|
||||
if has_error_recovery:
|
||||
data["error_category"] = details.get("error_category")
|
||||
data["error_phase"] = details.get("error_phase")
|
||||
data["retry_count"] = details.get("retry_count")
|
||||
data["max_retries"] = details.get("max_retries")
|
||||
data["is_retriable"] = details.get("is_retriable")
|
||||
if details.get("error_actor"):
|
||||
data["error_actor"] = details["error_actor"]
|
||||
if details.get("error_tool_call"):
|
||||
data["error_tool_call"] = details["error_tool_call"]
|
||||
if details.get("stack_summary"):
|
||||
data["stack_summary"] = details["stack_summary"]
|
||||
if hints:
|
||||
data["recovery_hints"] = hints
|
||||
console.print(format_output(data, fmt))
|
||||
return
|
||||
|
||||
# Rich output
|
||||
header = (
|
||||
f"[bold]Plan:[/bold] {plan_id}\n"
|
||||
f"[bold]Phase:[/bold] {plan.phase.value}\n"
|
||||
f"[bold]State:[/bold] "
|
||||
f"{plan.state.value if plan.state else 'unknown'}"
|
||||
)
|
||||
|
||||
if plan.error_message:
|
||||
header += f"\n[bold red]Error:[/bold red] {plan.error_message}"
|
||||
|
||||
if not has_error_recovery and not plan.error_message:
|
||||
header += "\n\n[green]No error recovery records for this plan.[/green]"
|
||||
console.print(Panel(header, title="Plan Errors", expand=False))
|
||||
return
|
||||
|
||||
if has_error_recovery:
|
||||
header += (
|
||||
f"\n\n[bold]Error Category:[/bold] "
|
||||
f"{details.get('error_category', 'unknown')}\n"
|
||||
f"[bold]Error Phase:[/bold] "
|
||||
f"{details.get('error_phase', 'unknown')}\n"
|
||||
f"[bold]Retry Count:[/bold] "
|
||||
f"{details.get('retry_count', '0')}"
|
||||
f"/{details.get('max_retries', '3')}\n"
|
||||
f"[bold]Retriable:[/bold] "
|
||||
f"{details.get('is_retriable', 'false')}"
|
||||
)
|
||||
if details.get("error_actor"):
|
||||
header += f"\n[bold]Actor:[/bold] {details['error_actor']}"
|
||||
if details.get("error_tool_call"):
|
||||
header += f"\n[bold]Tool Call:[/bold] {details['error_tool_call']}"
|
||||
|
||||
if hints:
|
||||
header += "\n\n[bold]Recovery Suggestions:[/bold]"
|
||||
for hint in hints:
|
||||
header += f"\n → {hint['message']}"
|
||||
if hint.get("cli_command"):
|
||||
header += f"\n [dim]$ {hint['cli_command']}[/dim]"
|
||||
|
||||
if details.get("stack_summary"):
|
||||
summary = details["stack_summary"][:500]
|
||||
header += f"\n\n[bold]Stack Summary:[/bold]\n[dim]{summary}[/dim]"
|
||||
|
||||
console.print(Panel(header, title="Plan Errors", expand=False))
|
||||
|
||||
except CleverAgentsError as e:
|
||||
console.print(f"[red]Error:[/red] {e.message}")
|
||||
raise typer.Abort() from e
|
||||
|
||||
|
||||
@app.command("lifecycle-list")
|
||||
def lifecycle_list_plans(
|
||||
regex: Annotated[
|
||||
|
||||
@@ -69,6 +69,19 @@ from cleveragents.domain.models.core.diff_review import (
|
||||
ReviewArtifact,
|
||||
)
|
||||
|
||||
|
||||
# Error recovery models
|
||||
from cleveragents.domain.models.core.error_recovery import (
|
||||
ErrorCategory,
|
||||
ErrorHistory,
|
||||
ErrorRecord,
|
||||
ErrorRecoveryPolicy,
|
||||
RecoveryAction,
|
||||
RecoveryHint,
|
||||
classify_error,
|
||||
get_recovery_hints,
|
||||
)
|
||||
|
||||
# Invariant domain models
|
||||
from cleveragents.domain.models.core.invariant import (
|
||||
Invariant,
|
||||
@@ -226,6 +239,10 @@ __all__ = [
|
||||
"DoDResult",
|
||||
"DoDStatus",
|
||||
"DoDSummary",
|
||||
"ErrorCategory",
|
||||
"ErrorHistory",
|
||||
"ErrorRecord",
|
||||
"ErrorRecoveryPolicy",
|
||||
"GuardResult",
|
||||
"InMemoryChangeSetStore",
|
||||
"InMemoryInvocationTracker",
|
||||
@@ -265,6 +282,8 @@ __all__ = [
|
||||
"ProjectLink",
|
||||
"ProjectSettings",
|
||||
"ProjectStats",
|
||||
"RecoveryAction",
|
||||
"RecoveryHint",
|
||||
"ResolvedToolEntry",
|
||||
"Resource",
|
||||
"ResourceAccessMode",
|
||||
@@ -307,7 +326,9 @@ __all__ = [
|
||||
"Validation",
|
||||
"ValidationMode",
|
||||
"can_transition",
|
||||
"classify_error",
|
||||
"get_builtin_profile",
|
||||
"get_recovery_hints",
|
||||
"merge_invariants",
|
||||
"normalize_change_path",
|
||||
"parse_dod_criteria",
|
||||
|
||||
@@ -0,0 +1,640 @@
|
||||
"""Error recovery domain models.
|
||||
|
||||
Provides structured error classification, recovery hints, and retry
|
||||
policy logic for plan execution failures.
|
||||
|
||||
Core types:
|
||||
|
||||
- **ErrorCategory**: Classifies errors as transient, validation,
|
||||
configuration, etc.
|
||||
- **RecoveryAction**: Suggested recovery actions (retry, revert, abort).
|
||||
- **RecoveryHint**: A structured suggestion for recovering from an error,
|
||||
including a human-readable message and a CLI command to execute.
|
||||
- **ErrorRecord**: A single recorded error with metadata (phase, category,
|
||||
message, stack summary, recovery hints, timestamp).
|
||||
- **ErrorHistory**: An ordered collection of ErrorRecords for a plan.
|
||||
- **ErrorRecoveryPolicy**: Determines whether retry is appropriate given
|
||||
the error category and automation profile thresholds.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from datetime import UTC, datetime
|
||||
from enum import StrEnum
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error classification
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class ErrorCategory(StrEnum):
|
||||
"""Classification of plan execution errors.
|
||||
|
||||
Used to determine whether an error is retriable, what recovery
|
||||
hints to suggest, and how to display the error in CLI output.
|
||||
"""
|
||||
|
||||
TRANSIENT = "transient"
|
||||
"""Temporary failure (network, rate-limit, timeout)."""
|
||||
|
||||
VALIDATION = "validation"
|
||||
"""Validation failure during Execute or Apply phase."""
|
||||
|
||||
CONFIGURATION = "configuration"
|
||||
"""Bad config, missing required fields, invalid arguments."""
|
||||
|
||||
RESOURCE = "resource"
|
||||
"""Missing or conflicting resource (file, sandbox, etc.)."""
|
||||
|
||||
MERGE_CONFLICT = "merge_conflict"
|
||||
"""Sandbox merge failed with conflicts."""
|
||||
|
||||
AUTHENTICATION = "authentication"
|
||||
"""Auth/authz failure accessing provider or resource."""
|
||||
|
||||
PROVIDER = "provider"
|
||||
"""LLM provider error (model unavailable, token limit, etc.)."""
|
||||
|
||||
INTERNAL = "internal"
|
||||
"""Unexpected internal error (bug, assertion, etc.)."""
|
||||
|
||||
UNKNOWN = "unknown"
|
||||
"""Error could not be classified."""
|
||||
|
||||
|
||||
class RecoveryAction(StrEnum):
|
||||
"""Suggested recovery action for a plan error.
|
||||
|
||||
Each action maps to a concrete CLI command or user workflow.
|
||||
"""
|
||||
|
||||
RETRY = "retry"
|
||||
"""Retry the failed phase (transient / provider errors)."""
|
||||
|
||||
REVERT_STRATEGIZE = "revert_strategize"
|
||||
"""Revert to Strategize phase and re-plan."""
|
||||
|
||||
CORRECT = "correct"
|
||||
"""Provide corrections and re-run validation."""
|
||||
|
||||
PROMPT = "prompt"
|
||||
"""Resume with additional user guidance."""
|
||||
|
||||
CANCEL = "cancel"
|
||||
"""Cancel the plan (non-recoverable error)."""
|
||||
|
||||
MANUAL = "manual"
|
||||
"""Requires manual investigation (internal errors)."""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Recovery hint
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class RecoveryHint(BaseModel):
|
||||
"""A structured suggestion for recovering from an error.
|
||||
|
||||
Each hint includes a human-readable message explaining what to do
|
||||
and an optional CLI command that can be copy-pasted.
|
||||
"""
|
||||
|
||||
action: RecoveryAction = Field(..., description="The suggested recovery action")
|
||||
message: str = Field(..., description="Human-readable recovery suggestion")
|
||||
cli_command: str | None = Field(
|
||||
default=None,
|
||||
description="CLI command to execute for recovery (copy-pasteable)",
|
||||
)
|
||||
priority: int = Field(
|
||||
default=0,
|
||||
ge=0,
|
||||
description="Priority of this hint (0 = highest priority)",
|
||||
)
|
||||
|
||||
model_config = ConfigDict(
|
||||
str_strip_whitespace=True,
|
||||
validate_assignment=True,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error record
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class ErrorRecord(BaseModel):
|
||||
"""A single recorded error event for a plan.
|
||||
|
||||
Captures structured metadata about what went wrong, in which phase,
|
||||
what actor/tool was involved, and what recovery options are available.
|
||||
"""
|
||||
|
||||
error_id: str = Field(..., description="Unique identifier for this error event")
|
||||
phase: str = Field(
|
||||
..., description="Plan phase when error occurred (strategize/execute/apply)"
|
||||
)
|
||||
category: ErrorCategory = Field(..., description="Error classification")
|
||||
message: str = Field(..., description="Error message")
|
||||
actor: str | None = Field(
|
||||
default=None, description="Actor name involved in the error"
|
||||
)
|
||||
tool_call: str | None = Field(
|
||||
default=None, description="Tool call that triggered the error"
|
||||
)
|
||||
stack_summary: str | None = Field(
|
||||
default=None,
|
||||
description="Abbreviated stack trace (last 3 frames)",
|
||||
)
|
||||
recovery_hints: list[RecoveryHint] = Field(
|
||||
default_factory=list,
|
||||
description="Ordered recovery suggestions",
|
||||
)
|
||||
retry_count: int = Field(
|
||||
default=0,
|
||||
ge=0,
|
||||
description="Number of retry attempts already made for this error",
|
||||
)
|
||||
max_retries: int = Field(
|
||||
default=3,
|
||||
ge=0,
|
||||
description="Maximum retry attempts allowed",
|
||||
)
|
||||
timestamp: datetime = Field(
|
||||
default_factory=lambda: datetime.now(tz=UTC),
|
||||
description="When the error was recorded (UTC)",
|
||||
)
|
||||
|
||||
model_config = ConfigDict(
|
||||
str_strip_whitespace=True,
|
||||
validate_assignment=True,
|
||||
)
|
||||
|
||||
@property
|
||||
def is_retriable(self) -> bool:
|
||||
"""Whether this error can be retried.
|
||||
|
||||
An error is retriable if:
|
||||
- Its category is in the retriable set
|
||||
- The retry count has not exceeded max_retries
|
||||
"""
|
||||
retriable_categories = {
|
||||
ErrorCategory.TRANSIENT,
|
||||
ErrorCategory.VALIDATION,
|
||||
ErrorCategory.MERGE_CONFLICT,
|
||||
ErrorCategory.PROVIDER,
|
||||
}
|
||||
return (
|
||||
self.category in retriable_categories
|
||||
and self.retry_count < self.max_retries
|
||||
)
|
||||
|
||||
@property
|
||||
def retries_exhausted(self) -> bool:
|
||||
"""Whether all retry attempts have been used."""
|
||||
return self.retry_count >= self.max_retries
|
||||
|
||||
def to_cli_dict(self) -> dict[str, Any]:
|
||||
"""Convert to a dictionary suitable for CLI display."""
|
||||
result: dict[str, Any] = {
|
||||
"error_id": self.error_id,
|
||||
"phase": self.phase,
|
||||
"category": self.category.value,
|
||||
"message": self.message,
|
||||
"retry_count": self.retry_count,
|
||||
"max_retries": self.max_retries,
|
||||
"is_retriable": self.is_retriable,
|
||||
"timestamp": self.timestamp.isoformat(),
|
||||
}
|
||||
if self.actor:
|
||||
result["actor"] = self.actor
|
||||
if self.tool_call:
|
||||
result["tool_call"] = self.tool_call
|
||||
if self.stack_summary:
|
||||
result["stack_summary"] = self.stack_summary
|
||||
if self.recovery_hints:
|
||||
result["recovery_hints"] = [
|
||||
{
|
||||
"action": h.action.value,
|
||||
"message": h.message,
|
||||
"cli_command": h.cli_command,
|
||||
}
|
||||
for h in sorted(self.recovery_hints, key=lambda h: h.priority)
|
||||
]
|
||||
return result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error history
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class ErrorHistory(BaseModel):
|
||||
"""Ordered collection of error records for a plan.
|
||||
|
||||
Provides convenience methods for querying error state and
|
||||
formatting for CLI output.
|
||||
"""
|
||||
|
||||
plan_id: str = Field(..., description="Plan ULID")
|
||||
records: list[ErrorRecord] = Field(
|
||||
default_factory=list,
|
||||
description="Error records in chronological order",
|
||||
)
|
||||
|
||||
model_config = ConfigDict(
|
||||
str_strip_whitespace=True,
|
||||
validate_assignment=True,
|
||||
)
|
||||
|
||||
def add(self, record: ErrorRecord) -> None:
|
||||
"""Append an error record."""
|
||||
self.records.append(record)
|
||||
|
||||
@property
|
||||
def latest(self) -> ErrorRecord | None:
|
||||
"""Return the most recent error, or None."""
|
||||
return self.records[-1] if self.records else None
|
||||
|
||||
@property
|
||||
def total_errors(self) -> int:
|
||||
"""Total number of recorded errors."""
|
||||
return len(self.records)
|
||||
|
||||
@property
|
||||
def total_retries(self) -> int:
|
||||
"""Sum of all retry counts across records."""
|
||||
return sum(r.retry_count for r in self.records)
|
||||
|
||||
@property
|
||||
def has_retriable(self) -> bool:
|
||||
"""Whether any error in the history is still retriable."""
|
||||
return any(r.is_retriable for r in self.records)
|
||||
|
||||
def errors_by_category(self) -> dict[str, int]:
|
||||
"""Count errors by category."""
|
||||
counts: dict[str, int] = {}
|
||||
for record in self.records:
|
||||
key = record.category.value
|
||||
counts[key] = counts.get(key, 0) + 1
|
||||
return counts
|
||||
|
||||
def to_cli_dict(self) -> dict[str, Any]:
|
||||
"""Convert to a dictionary suitable for CLI display."""
|
||||
return {
|
||||
"plan_id": self.plan_id,
|
||||
"total_errors": self.total_errors,
|
||||
"total_retries": self.total_retries,
|
||||
"has_retriable": self.has_retriable,
|
||||
"errors_by_category": self.errors_by_category(),
|
||||
"records": [r.to_cli_dict() for r in self.records],
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Recovery hint generation
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Maps error categories to default recovery hints
|
||||
_DEFAULT_HINTS: dict[ErrorCategory, list[RecoveryHint]] = {
|
||||
ErrorCategory.TRANSIENT: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.RETRY,
|
||||
message="Retry the operation — the failure was transient.",
|
||||
cli_command="agents plan prompt {plan_id}",
|
||||
priority=0,
|
||||
),
|
||||
],
|
||||
ErrorCategory.VALIDATION: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.CORRECT,
|
||||
message="Fix validation issues and re-run.",
|
||||
cli_command="agents plan correct {plan_id}",
|
||||
priority=0,
|
||||
),
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.PROMPT,
|
||||
message="Provide additional guidance to help resolve the issue.",
|
||||
cli_command="agents plan prompt {plan_id}",
|
||||
priority=1,
|
||||
),
|
||||
],
|
||||
ErrorCategory.CONFIGURATION: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.CANCEL,
|
||||
message="Fix the configuration and re-create the plan.",
|
||||
priority=0,
|
||||
),
|
||||
],
|
||||
ErrorCategory.RESOURCE: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.REVERT_STRATEGIZE,
|
||||
message=("Revert to Strategize to re-plan with updated resources."),
|
||||
cli_command=("agents plan revert {plan_id} --to-phase strategize"),
|
||||
priority=0,
|
||||
),
|
||||
],
|
||||
ErrorCategory.MERGE_CONFLICT: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.REVERT_STRATEGIZE,
|
||||
message="Resolve merge conflicts and re-run from Strategize.",
|
||||
cli_command=("agents plan revert {plan_id} --to-phase strategize"),
|
||||
priority=0,
|
||||
),
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.RETRY,
|
||||
message="Retry apply after resolving conflicts manually.",
|
||||
cli_command="agents plan prompt {plan_id}",
|
||||
priority=1,
|
||||
),
|
||||
],
|
||||
ErrorCategory.AUTHENTICATION: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.CANCEL,
|
||||
message=(
|
||||
"Check authentication credentials and retry. "
|
||||
"This error is not auto-retriable."
|
||||
),
|
||||
priority=0,
|
||||
),
|
||||
],
|
||||
ErrorCategory.PROVIDER: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.RETRY,
|
||||
message="Retry — the LLM provider may be temporarily unavailable.",
|
||||
cli_command="agents plan prompt {plan_id}",
|
||||
priority=0,
|
||||
),
|
||||
],
|
||||
ErrorCategory.INTERNAL: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.MANUAL,
|
||||
message=(
|
||||
"An internal error occurred. Check logs and report "
|
||||
"the issue if it persists."
|
||||
),
|
||||
priority=0,
|
||||
),
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.CANCEL,
|
||||
message="Cancel the plan and start fresh if the error persists.",
|
||||
cli_command="agents plan cancel {plan_id}",
|
||||
priority=1,
|
||||
),
|
||||
],
|
||||
ErrorCategory.UNKNOWN: [
|
||||
RecoveryHint(
|
||||
action=RecoveryAction.MANUAL,
|
||||
message="Unknown error — investigate logs for details.",
|
||||
priority=0,
|
||||
),
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def get_recovery_hints(
|
||||
category: ErrorCategory,
|
||||
plan_id: str,
|
||||
) -> list[RecoveryHint]:
|
||||
"""Return recovery hints for an error category.
|
||||
|
||||
Substitutes ``{plan_id}`` placeholders in CLI commands with
|
||||
the actual plan ID.
|
||||
|
||||
Args:
|
||||
category: The error classification.
|
||||
plan_id: The plan ULID to substitute into CLI commands.
|
||||
|
||||
Returns:
|
||||
List of ``RecoveryHint`` objects ordered by priority.
|
||||
"""
|
||||
templates = _DEFAULT_HINTS.get(category, _DEFAULT_HINTS[ErrorCategory.UNKNOWN])
|
||||
hints: list[RecoveryHint] = []
|
||||
for tmpl in templates:
|
||||
cli_cmd = (
|
||||
tmpl.cli_command.replace("{plan_id}", plan_id) if tmpl.cli_command else None
|
||||
)
|
||||
hints.append(
|
||||
RecoveryHint(
|
||||
action=tmpl.action,
|
||||
message=tmpl.message,
|
||||
cli_command=cli_cmd,
|
||||
priority=tmpl.priority,
|
||||
)
|
||||
)
|
||||
return hints
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error classification helper
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Compiled regex patterns for classifying error messages.
|
||||
# Ordered list of (category, pattern) — checked in order, first match wins.
|
||||
# More specific categories (authentication, merge_conflict) must appear
|
||||
# before broader ones (validation, resource) to avoid false matches.
|
||||
_CATEGORY_PATTERNS: list[tuple[ErrorCategory, re.Pattern[str]]] = [
|
||||
(
|
||||
ErrorCategory.AUTHENTICATION,
|
||||
re.compile(
|
||||
r"authentication|unauthorized|forbidden|permission denied",
|
||||
),
|
||||
),
|
||||
(
|
||||
ErrorCategory.MERGE_CONFLICT,
|
||||
re.compile(r"merge conflict|merge failed"),
|
||||
),
|
||||
(
|
||||
ErrorCategory.TRANSIENT,
|
||||
re.compile(
|
||||
r"timeout|timed out|temporary|connection reset"
|
||||
r"|connection refused|rate limit|throttl",
|
||||
),
|
||||
),
|
||||
(
|
||||
ErrorCategory.PROVIDER,
|
||||
re.compile(
|
||||
r"model not available|token limit|context length|provider error",
|
||||
),
|
||||
),
|
||||
(
|
||||
ErrorCategory.CONFIGURATION,
|
||||
re.compile(
|
||||
r"configuration|config|missing setting|not configured",
|
||||
),
|
||||
),
|
||||
(
|
||||
ErrorCategory.VALIDATION,
|
||||
re.compile(
|
||||
r"validation|invalid|constraint|does not match"
|
||||
r"|required field|schema",
|
||||
),
|
||||
),
|
||||
(
|
||||
ErrorCategory.RESOURCE,
|
||||
re.compile(r"not found|missing resource|no such file"),
|
||||
),
|
||||
(
|
||||
ErrorCategory.INTERNAL,
|
||||
re.compile(r"internal error|assertion|unexpected|bug"),
|
||||
),
|
||||
]
|
||||
|
||||
# Exception-type → category mapping (checked before keyword patterns).
|
||||
_EXCEPTION_TYPE_MAP: dict[str, ErrorCategory] = {
|
||||
"ratelimiterror": ErrorCategory.TRANSIENT,
|
||||
"networkerror": ErrorCategory.TRANSIENT,
|
||||
"timeouterror": ErrorCategory.TRANSIENT,
|
||||
"validationerror": ErrorCategory.VALIDATION,
|
||||
"configurationerror": ErrorCategory.CONFIGURATION,
|
||||
"missingconfigurationerror": ErrorCategory.CONFIGURATION,
|
||||
"resourcenotfounderror": ErrorCategory.RESOURCE,
|
||||
"resourceconflicterror": ErrorCategory.RESOURCE,
|
||||
"authenticationerror": ErrorCategory.AUTHENTICATION,
|
||||
"authorizationerror": ErrorCategory.AUTHENTICATION,
|
||||
"providererror": ErrorCategory.PROVIDER,
|
||||
"modelnotavailableerror": ErrorCategory.PROVIDER,
|
||||
"tokenlimitexceedederror": ErrorCategory.PROVIDER,
|
||||
}
|
||||
|
||||
|
||||
def classify_error(
|
||||
message: str,
|
||||
exception_type: str | None = None,
|
||||
) -> ErrorCategory:
|
||||
"""Classify an error message into an ErrorCategory.
|
||||
|
||||
Uses a compiled regex per category for keyword matching on the
|
||||
message text. If ``exception_type`` is provided, it takes
|
||||
precedence.
|
||||
|
||||
Args:
|
||||
message: The error message to classify.
|
||||
exception_type: Optional exception class name.
|
||||
|
||||
Returns:
|
||||
The best-matching ``ErrorCategory``.
|
||||
"""
|
||||
lower_exc = (exception_type or "").lower()
|
||||
|
||||
# Check exception type first (more specific)
|
||||
if lower_exc in _EXCEPTION_TYPE_MAP:
|
||||
return _EXCEPTION_TYPE_MAP[lower_exc]
|
||||
|
||||
# Single compiled-regex search per category (first match wins)
|
||||
lower_msg = message.lower()
|
||||
for category, pattern in _CATEGORY_PATTERNS:
|
||||
if pattern.search(lower_msg):
|
||||
return category
|
||||
|
||||
return ErrorCategory.UNKNOWN
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Error recovery policy
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
class ErrorRecoveryPolicy:
|
||||
"""Determines whether retry is appropriate for a given error.
|
||||
|
||||
Uses the automation profile thresholds to decide between automatic
|
||||
retry (threshold 0.0) and human intervention (threshold 1.0).
|
||||
|
||||
Attributes:
|
||||
auto_retry_threshold: Threshold from automation profile
|
||||
(0.0 = always auto-retry, 1.0 = always ask human).
|
||||
max_retries: Maximum retry attempts.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
auto_retry_threshold: float = 0.0,
|
||||
max_retries: int = 3,
|
||||
) -> None:
|
||||
self.auto_retry_threshold = auto_retry_threshold
|
||||
self.max_retries = max_retries
|
||||
|
||||
def should_retry(self, record: ErrorRecord) -> bool:
|
||||
"""Determine whether the error should be retried.
|
||||
|
||||
Returns ``True`` if:
|
||||
- The error is in a retriable category
|
||||
- Retries have not been exhausted
|
||||
- The automation profile allows auto-retry (threshold < 1.0)
|
||||
|
||||
Args:
|
||||
record: The error record to evaluate.
|
||||
|
||||
Returns:
|
||||
``True`` if retry should proceed automatically.
|
||||
"""
|
||||
if not record.is_retriable:
|
||||
return False
|
||||
if record.retries_exhausted:
|
||||
return False
|
||||
# Threshold < 1.0 means auto-retry is enabled
|
||||
return self.auto_retry_threshold < 1.0
|
||||
|
||||
def should_escalate(self, record: ErrorRecord) -> bool:
|
||||
"""Determine whether the error requires human intervention.
|
||||
|
||||
Returns ``True`` if:
|
||||
- The error is not retriable, OR
|
||||
- Retries are exhausted, OR
|
||||
- The automation profile requires human decision
|
||||
|
||||
Args:
|
||||
record: The error record to evaluate.
|
||||
|
||||
Returns:
|
||||
``True`` if the error should be escalated to a human.
|
||||
"""
|
||||
if not record.is_retriable:
|
||||
return True
|
||||
if record.retries_exhausted:
|
||||
return True
|
||||
return self.auto_retry_threshold >= 1.0
|
||||
|
||||
def format_recovery_output(
|
||||
self,
|
||||
record: ErrorRecord,
|
||||
) -> str:
|
||||
"""Format a human-readable recovery output for CLI display.
|
||||
|
||||
Includes error details, recovery hints with CLI commands,
|
||||
and retry status.
|
||||
|
||||
Args:
|
||||
record: The error record to format.
|
||||
|
||||
Returns:
|
||||
Formatted string for CLI output.
|
||||
"""
|
||||
lines: list[str] = []
|
||||
lines.append(f"Error [{record.category.value}]: {record.message}")
|
||||
lines.append(f" Phase: {record.phase}")
|
||||
if record.actor:
|
||||
lines.append(f" Actor: {record.actor}")
|
||||
if record.tool_call:
|
||||
lines.append(f" Tool: {record.tool_call}")
|
||||
lines.append(f" Retry: {record.retry_count}/{record.max_retries}")
|
||||
|
||||
if record.recovery_hints:
|
||||
lines.append(" Recovery suggestions:")
|
||||
for hint in sorted(record.recovery_hints, key=lambda h: h.priority):
|
||||
lines.append(f" → {hint.message}")
|
||||
if hint.cli_command:
|
||||
lines.append(f" $ {hint.cli_command}")
|
||||
|
||||
if record.retries_exhausted:
|
||||
lines.append(" ⚠ All retry attempts exhausted.")
|
||||
elif self.should_retry(record):
|
||||
lines.append(" ✓ Automatic retry available.")
|
||||
else:
|
||||
lines.append(" ⚠ Human intervention required.")
|
||||
|
||||
return "\n".join(lines)
|
||||
Reference in New Issue
Block a user