Files
cleveragents-core/benchmarks/execution_throughput_bench.py
T
brent.edwards e77cf8bb1c
CI / lint (pull_request) Successful in 21s
CI / quality (pull_request) Successful in 47s
CI / typecheck (pull_request) Successful in 50s
CI / security (pull_request) Successful in 1m6s
CI / benchmark-publish (pull_request) Has been skipped
CI / build (pull_request) Successful in 22s
CI / unit_tests (pull_request) Successful in 3m44s
CI / integration_tests (pull_request) Successful in 3m50s
CI / docker (pull_request) Successful in 1m0s
CI / e2e_tests (pull_request) Successful in 6m26s
CI / coverage (pull_request) Successful in 8m0s
CI / benchmark-regression (pull_request) Failing after 59m59s
feat(perf): large project scaling tests
Add ASV benchmark suites for large project scaling: indexing at
10K/50K/100K files with walk_and_index throughput and incremental
refresh metrics, context assembly at 1K/5K/10K fragments, and
execution throughput for sequential/concurrent plans. Update scale
fixture thresholds for 50K/100K profiles. Add Behave and Robot
tests for baseline validation. Document scaling baselines.

Also fixes pre-existing test failures and integrates recent master
changes discovered during full nox suite verification.

ISSUES CLOSED: #859
2026-03-19 22:25:29 +00:00

124 lines
3.9 KiB
Python

"""ASV benchmarks for plan execution throughput at scale.
Measures sequential and concurrent plan execution overhead at varying
plan counts (10, 50, 100). Uses the lightweight in-process executor
path (no database, no LLM) to isolate execution-dispatch cost.
"""
from __future__ import annotations
import importlib
import sys
from pathlib import Path
from typing import ClassVar
from unittest.mock import MagicMock
# Ensure the local *source* tree is importable even when ASV has an
# older build of the package installed.
_SRC = str(Path(__file__).resolve().parents[1] / "src")
if _SRC not in sys.path:
sys.path.insert(0, _SRC)
import cleveragents # noqa: E402
importlib.reload(cleveragents)
from ulid import ULID # noqa: E402
from cleveragents.application.services.plan_execution_context import ( # noqa: E402
PlanExecutionContext,
RuntimeExecuteActor,
)
from cleveragents.application.services.plan_executor import ( # noqa: E402
PlanExecutor,
StrategyDecision,
)
from cleveragents.domain.models.core.change import ( # noqa: E402
InMemoryChangeSetStore,
)
from cleveragents.tool.registry import ToolRegistry # noqa: E402
from cleveragents.tool.runner import ToolRunner # noqa: E402
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
_RESOURCE_ID = "01HGZ6FE0AQDYTR4BXVQZ6EB00"
def _make_runner() -> ToolRunner:
return ToolRunner(registry=ToolRegistry())
def _make_decisions(count: int) -> list[StrategyDecision]:
"""Build a linear chain of *count* decisions."""
root_id = str(ULID())
return [
StrategyDecision(
decision_id=root_id if i == 0 else str(ULID()),
step_text=f"Step {i + 1}",
sequence=i,
parent_id=root_id if i > 0 else None,
)
for i in range(count)
]
def _execute_single_plan(runner: ToolRunner) -> None:
"""Execute one plan with 3 decisions (fire-and-forget)."""
plan_id = str(ULID())
ctx = PlanExecutionContext(
plan_id=plan_id,
changeset_store=InMemoryChangeSetStore(),
)
actor = RuntimeExecuteActor(tool_runner=runner, execution_context=ctx)
actor.execute(decisions=_make_decisions(3))
# ---------------------------------------------------------------------------
# Parameterized execution throughput suite
# ---------------------------------------------------------------------------
class ExecutionThroughputSuite:
"""Benchmark plan execution throughput at varying plan counts."""
params: ClassVar[list[int]] = [10, 50, 100]
param_names: ClassVar[list[str]] = ["plan_count"]
timeout = 300
_runner: ToolRunner
def setup(self, plan_count: int) -> None:
"""Prepare a shared tool runner."""
self._runner = _make_runner()
def time_sequential_plans(self, plan_count: int) -> None:
"""Execute *plan_count* plans sequentially."""
for _ in range(plan_count):
_execute_single_plan(self._runner)
def time_executor_construction(self, plan_count: int) -> None:
"""Construct *plan_count* PlanExecutor instances."""
lifecycle = MagicMock()
for _ in range(plan_count):
ctx = PlanExecutionContext(
plan_id=str(ULID()),
changeset_store=InMemoryChangeSetStore(),
)
PlanExecutor(
lifecycle_service=lifecycle,
tool_runner=self._runner,
execution_context=ctx,
)
def time_decision_tree_scaling(self, plan_count: int) -> None:
"""Execute one plan with *plan_count* decisions."""
plan_id = str(ULID())
ctx = PlanExecutionContext(
plan_id=plan_id,
changeset_store=InMemoryChangeSetStore(),
)
actor = RuntimeExecuteActor(tool_runner=self._runner, execution_context=ctx)
actor.execute(decisions=_make_decisions(plan_count))