Files
cleveragents-core/features/steps/estimation_actor_steps.py
brent.edwards eb46f0ff54
CI / push-validation (push) Successful in 40s
CI / helm (push) Successful in 44s
CI / build (push) Successful in 1m14s
CI / lint (push) Successful in 1m18s
CI / quality (push) Successful in 1m36s
CI / e2e_tests (push) Successful in 1m19s
CI / typecheck (push) Successful in 2m22s
CI / security (push) Successful in 2m23s
CI / benchmark-regression (push) Failing after 41s
CI / integration_tests (push) Successful in 4m46s
CI / unit_tests (push) Failing after 20m13s
CI / benchmark-publish (push) Failing after 28m28s
CI / coverage (push) Has been cancelled
CI / docker (push) Has been cancelled
CI / status-check (push) Has been cancelled
fix(plan): add tier hydration and improve architecture review output (#10938)
## Summary

This PR fixes issue #10878 where architecture reviews were truncated because the regex pattern for parsing file output would stop at the first ``` encountered in the Markdown report.

## Changes

- Change file delimiters from ``` to >>>>>>>/<<<<<<< to avoid Markdown conflicts
- Add tier hydration before strategize phase in plan_executor.py
- Increase max_tokens to 16384 in llm_actors.py for longer outputs
- Increase context_max_tokens_hot from 16000 to 32000 in settings.py
- Fix get_hot_view → get_hot_fragments in strategy_actor.py and plan_executor.py
- Add opencode to skip directories in context_tier_hydrator.py
- Change sandbox output location to plan-output/ directory in plan.py
- Add get_context_summary stub method to acms_service.py

## Testing

Run architecture review action and verify the output report is complete with all sections.

Reviewed-on: #10938
2026-05-19 12:43:34 +00:00

394 lines
15 KiB
Python

"""Step definitions for Estimation Actor feature tests."""
from __future__ import annotations
from behave import given, then, when
from behave.runner import Context
from pydantic import ValidationError as PydanticValidationError
from cleveragents.application.services.plan_executor import EstimationStubActor
from cleveragents.application.services.plan_lifecycle_service import (
PlanLifecycleService,
)
from cleveragents.config.settings import Settings
from cleveragents.core.exceptions import ValidationError
from cleveragents.domain.models.acms.tiers import ActorRole
from cleveragents.domain.models.core.estimation import EstimationResult
from cleveragents.domain.models.core.plan import PlanPhase, ProcessingState
# ---------------------------------------------------------------------------
# EstimationResult domain model steps
# ---------------------------------------------------------------------------
@given("I create an estimation result with all fields")
def step_create_estimation_all_fields(context: Context) -> None:
"""Create an EstimationResult with all fields populated."""
context.estimation_result = EstimationResult(
estimated_cost_usd=0.0512,
estimated_tokens=12000,
estimated_steps=5,
estimated_child_plans=2,
estimated_time_seconds=30.5,
risk_level="medium",
risk_factors=("API rate limits", "Large codebase"),
summary="Estimated 5 steps, ~$0.05 cost",
raw_output="raw estimation output text",
)
@then("the estimation result should have all expected values")
def step_check_estimation_all_values(context: Context) -> None:
"""Verify all fields on the estimation result."""
est: EstimationResult = context.estimation_result
assert est.estimated_cost_usd == 0.0512
assert est.estimated_tokens == 12000
assert est.estimated_steps == 5
assert est.estimated_child_plans == 2
assert est.estimated_time_seconds == 30.5
assert est.risk_level == "medium"
assert est.risk_factors == ("API rate limits", "Large codebase")
assert est.summary == "Estimated 5 steps, ~$0.05 cost"
assert est.raw_output == "raw estimation output text"
@given("I create an estimation result with defaults")
def step_create_estimation_defaults(context: Context) -> None:
"""Create an EstimationResult with only default values."""
context.estimation_result = EstimationResult()
@then("the estimation result summary should be empty")
def step_check_estimation_summary_empty(context: Context) -> None:
"""Verify the summary field is an empty string."""
assert context.estimation_result.summary == ""
@then("the estimation result risk_factors should be empty")
def step_check_estimation_risk_factors_empty(context: Context) -> None:
"""Verify risk_factors is an empty tuple."""
assert context.estimation_result.risk_factors == ()
@then("all optional estimation fields should be None")
def step_check_estimation_optional_none(context: Context) -> None:
"""Verify all optional fields are None."""
est: EstimationResult = context.estimation_result
assert est.estimated_cost_usd is None
assert est.estimated_tokens is None
assert est.estimated_steps is None
assert est.estimated_child_plans is None
assert est.estimated_time_seconds is None
assert est.risk_level is None
@when("I call as_display_dict on the estimation result")
def step_call_as_display_dict(context: Context) -> None:
"""Call as_display_dict on the estimation result."""
est: EstimationResult = context.estimation_result
context.display_dict = est.as_display_dict()
@then("the display dict should include estimated_child_plans")
def step_check_display_dict_child_plans(context: Context) -> None:
"""Verify display dict includes estimated_child_plans."""
assert "estimated_child_plans" in context.display_dict
assert context.display_dict["estimated_child_plans"] == 2
@then("the display dict should include estimated_time_seconds")
def step_check_display_dict_time_seconds(context: Context) -> None:
"""Verify display dict includes estimated_time_seconds."""
assert "estimated_time_seconds" in context.display_dict
assert context.display_dict["estimated_time_seconds"] == 30.5
@then("the display dict should include risk_factors")
def step_check_display_dict_risk_factors(context: Context) -> None:
"""Verify display dict includes risk_factors as a list."""
assert "risk_factors" in context.display_dict
assert context.display_dict["risk_factors"] == [
"API rate limits",
"Large codebase",
]
@when("I try to create an estimation result with 101 risk factors")
def step_create_estimation_too_many_risk_factors(context: Context) -> None:
"""Try to create EstimationResult with >100 risk factors."""
factors = tuple(f"risk-{i}" for i in range(101))
try:
EstimationResult(risk_factors=factors)
context.error = None
except PydanticValidationError as exc:
context.error = exc
@then("a validation error should be raised for risk_factors limit")
def step_check_risk_factors_limit_error(context: Context) -> None:
"""Verify a validation error was raised for too many risk factors."""
assert context.error is not None
assert "100" in str(context.error)
@when("I try to modify the estimation result summary")
def step_try_modify_estimation(context: Context) -> None:
"""Try to mutate the frozen EstimationResult."""
try:
context.estimation_result.summary = "modified" # type: ignore[misc]
context.error = None
except PydanticValidationError as exc:
context.error = exc
@then("a validation error should be raised for frozen model")
def step_check_frozen_error(context: Context) -> None:
"""Verify a validation error was raised due to frozen model."""
assert context.error is not None
assert (
"frozen" in str(context.error).lower()
or "immutable" in str(context.error).lower()
)
@when("I serialize the estimation result to dict")
def step_serialize_estimation(context: Context) -> None:
"""Serialize EstimationResult to dict."""
context.estimation_dict = context.estimation_result.model_dump()
@when("I deserialize it back to an EstimationResult")
def step_deserialize_estimation(context: Context) -> None:
"""Deserialize dict back to EstimationResult."""
context.deserialized_estimation = EstimationResult(**context.estimation_dict)
@then("the deserialized result should match the original")
def step_check_deserialized_matches(context: Context) -> None:
"""Verify deserialized object matches the original."""
original: EstimationResult = context.estimation_result
deserialized: EstimationResult = context.deserialized_estimation
assert original == deserialized
# ---------------------------------------------------------------------------
# Plan model + estimation_result field steps
# ---------------------------------------------------------------------------
@given("I have a plan lifecycle service for estimation tests")
def step_create_lifecycle_for_estimation(context: Context) -> None:
"""Create a plan lifecycle service."""
settings = Settings()
context.est_lifecycle = PlanLifecycleService(settings=settings)
context.error = None
@given("I have an available action for estimation tests")
def step_create_action_for_estimation(context: Context) -> None:
"""Create an action with estimation actor."""
context.est_action = context.est_lifecycle.create_action(
name="local/est-test",
description="Test action for estimation",
definition_of_done="All tests pass",
strategy_actor="openai/gpt-4",
execution_actor="openai/gpt-4",
estimation_actor="local/estimator",
)
@when("I use the action to create a plan for estimation tests")
def step_use_action_for_estimation(context: Context) -> None:
"""Use the action to create a plan."""
context.est_plan = context.est_lifecycle.use_action(
action_name="local/est-test",
)
@when("I set an estimation result on the plan")
def step_set_estimation_on_plan(context: Context) -> None:
"""Set an estimation result directly on the plan."""
context.est_plan.estimation_result = EstimationResult(
estimated_cost_usd=1.23,
estimated_tokens=5000,
estimated_steps=3,
risk_level="low",
summary="Low risk, ~$1.23 estimated cost",
)
@then("the plan estimation_result should be set")
def step_check_plan_estimation_set(context: Context) -> None:
"""Verify the plan has an estimation result."""
assert context.est_plan.estimation_result is not None
assert context.est_plan.estimation_result.summary != ""
@then("the plan cli dict should include estimation data")
def step_check_plan_cli_dict_estimation(context: Context) -> None:
"""Verify as_cli_dict includes estimation data."""
cli_dict = context.est_plan.as_cli_dict()
assert "estimation" in cli_dict
est_data = cli_dict["estimation"]
assert est_data["estimated_cost_usd"] == 1.23
assert est_data["risk_level"] == "low"
# ---------------------------------------------------------------------------
# Plans without estimation actor
# ---------------------------------------------------------------------------
@when("I create an action without estimation actor for estimation tests")
def step_create_action_no_estimation(context: Context) -> None:
"""Create an action without estimation actor."""
context.est_action_no_est = context.est_lifecycle.create_action(
name="local/no-est-test",
description="Test action without estimation",
definition_of_done="All tests pass",
strategy_actor="openai/gpt-4",
execution_actor="openai/gpt-4",
)
@when("I use that action to create an estimation test plan")
def step_use_no_est_action(context: Context) -> None:
"""Use the action without estimation actor."""
context.est_plan = context.est_lifecycle.use_action(
action_name="local/no-est-test",
)
@then("the plan estimation_actor should be None")
def step_check_plan_no_estimation_actor(context: Context) -> None:
"""Verify plan has no estimation actor."""
assert context.est_plan.estimation_actor is None
@then("the plan estimation_result should be None")
def step_check_plan_no_estimation_result(context: Context) -> None:
"""Verify plan has no estimation result."""
assert context.est_plan.estimation_result is None
# ---------------------------------------------------------------------------
# EstimationStubActor steps
# ---------------------------------------------------------------------------
@when("I run the estimation stub actor with a valid plan id")
def step_run_stub_valid(context: Context) -> None:
"""Run the stub actor with a valid plan ID."""
stub = EstimationStubActor()
context.stub_result = stub.estimate("01ARZ3NDEKTSV4RRFFQ69G5FAV")
@then('the stub result summary should contain "stub actor"')
def step_check_stub_summary(context: Context) -> None:
"""Verify the stub result summary mentions stub."""
assert "stub actor" in context.stub_result.summary.lower()
@when("I run the estimation stub actor with an empty plan id")
def step_run_stub_empty_id(context: Context) -> None:
"""Run the stub actor with an empty plan ID."""
stub = EstimationStubActor()
try:
stub.estimate("")
context.error = None
except ValidationError as exc:
context.error = exc
@then("a validation error should be raised for estimation")
def step_check_estimation_validation_error(context: Context) -> None:
"""Verify a validation error was raised."""
assert context.error is not None
# ---------------------------------------------------------------------------
# ActorRole enum steps
# ---------------------------------------------------------------------------
@then("the ActorRole enum should have an ESTIMATOR value")
def step_check_actor_role_estimator(context: Context) -> None:
"""Verify ActorRole has ESTIMATOR."""
assert hasattr(ActorRole, "ESTIMATOR")
@then('the ESTIMATOR value should be "estimator"')
def step_check_estimator_value(context: Context) -> None:
"""Verify ESTIMATOR value."""
assert ActorRole.ESTIMATOR.value == "estimator"
# ---------------------------------------------------------------------------
# Estimation invocation in lifecycle steps
# ---------------------------------------------------------------------------
@given("I have an action with estimation actor set")
def step_create_action_with_estimation(context: Context) -> None:
"""Create an action with estimation actor configured."""
context.est_action = context.est_lifecycle.create_action(
name="local/est-lifecycle-test",
description="Test action with estimation",
definition_of_done="All tests pass",
strategy_actor="openai/gpt-4",
execution_actor="openai/gpt-4",
estimation_actor="local/estimator",
)
@given("I have an action without estimation actor")
def step_create_action_without_estimation(context: Context) -> None:
"""Create an action without estimation actor."""
context.est_action = context.est_lifecycle.create_action(
name="local/no-est-lifecycle-test",
description="Test action without estimation",
definition_of_done="All tests pass",
strategy_actor="openai/gpt-4",
execution_actor="openai/gpt-4",
)
@when("I use the estimation action and complete strategize")
def step_use_estimation_action_and_complete_strategize(context: Context) -> None:
"""Use the action and drive through strategize phase."""
action_name = str(context.est_action.namespaced_name)
context.est_plan = context.est_lifecycle.use_action(action_name=action_name)
plan_id = context.est_plan.identity.plan_id
# Drive through strategize: start -> processing -> complete
context.est_lifecycle.start_strategize(plan_id)
# Manually set processing state to PROCESSING then COMPLETE
plan = context.est_lifecycle.get_plan(plan_id)
plan.processing_state = ProcessingState.PROCESSING
context.est_lifecycle.commit_plan(plan)
context.est_lifecycle.complete_strategize(plan_id)
@when("I transition the estimation plan to execute phase")
def step_transition_estimation_plan_to_execute(context: Context) -> None:
"""Transition plan to execute phase (triggers estimation)."""
plan_id = context.est_plan.identity.plan_id
plan = context.est_lifecycle.get_plan(plan_id)
# If auto_progress already advanced to Execute, just refresh
if plan.phase == PlanPhase.EXECUTE:
context.est_plan = plan
return
# Otherwise manually transition
context.est_plan = context.est_lifecycle.execute_plan(plan_id)
@when("I transition the estimation plan to execute phase without estimation")
def step_transition_estimation_plan_to_execute_no_estimation(context: Context) -> None:
"""Transition plan to execute phase (no estimation actor)."""
plan_id = context.est_plan.identity.plan_id
plan = context.est_lifecycle.get_plan(plan_id)
if plan.phase == PlanPhase.EXECUTE:
context.est_plan = plan
return
context.est_plan = context.est_lifecycle.execute_plan(plan_id)