Files
cleveragents-core/benchmarks/bench_metrics_collection.py
T
freemo c65e8a5285
CI / benchmark-publish (pull_request) Has been skipped
CI / build (pull_request) Successful in 16s
CI / lint (pull_request) Successful in 17s
CI / quality (pull_request) Successful in 29s
CI / typecheck (pull_request) Successful in 59s
CI / security (pull_request) Successful in 59s
CI / unit_tests (pull_request) Successful in 3m9s
CI / integration_tests (pull_request) Successful in 3m34s
CI / e2e_tests (pull_request) Successful in 3m54s
CI / docker (pull_request) Successful in 55s
CI / coverage (pull_request) Successful in 6m1s
CI / build (push) Successful in 15s
CI / lint (push) Successful in 16s
CI / quality (push) Successful in 27s
CI / typecheck (push) Successful in 38s
CI / benchmark-regression (push) Has been skipped
CI / security (push) Successful in 48s
CI / unit_tests (push) Successful in 3m14s
CI / integration_tests (push) Successful in 3m30s
CI / docker (push) Successful in 56s
CI / e2e_tests (push) Successful in 4m43s
CI / coverage (push) Successful in 6m1s
CI / benchmark-publish (push) Successful in 19m50s
CI / benchmark-regression (pull_request) Successful in 37m17s
feat(resource): add cloud infrastructure resources
Implement cloud resource types (aws, gcp, azure) with credential
fields, region/tenant metadata, and stubbed sandbox strategies.
Credential resolution uses environment variables and profile names
with no secrets logged.

Key changes:
- Add CloudResourceHandler with aws/gcp/azure type definitions
- Add credential resolution from env vars and profile names
- Add stubbed sandbox strategies (validate config, raise NotImplementedError)
- Register cloud types in bootstrap_builtin_types
- Credential masking via existing redaction patterns
- Add Behave BDD tests, Robot integration tests, ASV benchmarks

ISSUES CLOSED: #343
2026-03-17 13:14:37 -04:00

100 lines
3.2 KiB
Python

"""ASV benchmarks for metrics collection framework overhead.
Measures the time to create metric entries via histogram/counter/gauge
factory methods, exercise all 14 convenience methods, and emit metrics
through the MetricsEmitter.
"""
from __future__ import annotations
import importlib
import sys
from pathlib import Path
_SRC = str(Path(__file__).resolve().parents[1] / "src")
if _SRC not in sys.path:
sys.path.insert(0, _SRC)
import cleveragents # noqa: E402
importlib.reload(cleveragents)
from cleveragents.domain.models.observability.metrics import ( # noqa: E402
MetricCollector,
OperationalMetricKey,
)
from cleveragents.infrastructure.observability.metrics_emitter import ( # noqa: E402
MetricsEmitter,
)
VALID_PLAN_ID = "01HX0000000000MMMMMMMMMMMM"
class MetricCollectorSuite:
"""Benchmark MetricCollector typed factory methods."""
def time_histogram(self) -> None:
"""Create a histogram metric entry."""
MetricCollector.histogram(
OperationalMetricKey.PLAN_DURATION_MS,
1500.0,
VALID_PLAN_ID,
)
def time_counter(self) -> None:
"""Create a counter metric entry."""
MetricCollector.counter(
OperationalMetricKey.LLM_CALL_COUNT,
5.0,
VALID_PLAN_ID,
)
def time_gauge(self) -> None:
"""Create a gauge metric entry."""
MetricCollector.gauge(
OperationalMetricKey.PLAN_DECISION_COUNT,
3.0,
VALID_PLAN_ID,
)
def time_all_14_convenience_methods(self) -> None:
"""Exercise all 14 convenience methods."""
MetricCollector.plan_duration(VALID_PLAN_ID, 1000.0)
MetricCollector.plan_cost(VALID_PLAN_ID, 0.05)
MetricCollector.plan_decision_count(VALID_PLAN_ID, 5.0)
MetricCollector.subplan_count(VALID_PLAN_ID, 2.0)
MetricCollector.actor_invocation_count(VALID_PLAN_ID, 1.0)
MetricCollector.actor_latency(VALID_PLAN_ID, 250.0)
MetricCollector.tool_invocation_count(VALID_PLAN_ID, 10.0)
MetricCollector.tool_error_rate(VALID_PLAN_ID, 0.01)
MetricCollector.context_build_time(VALID_PLAN_ID, 50.0)
MetricCollector.context_token_count(VALID_PLAN_ID, 8000.0)
MetricCollector.llm_call_count(VALID_PLAN_ID, 3.0)
MetricCollector.llm_total_tokens(VALID_PLAN_ID, 5000.0)
MetricCollector.llm_total_cost(VALID_PLAN_ID, 0.5)
MetricCollector.llm_avg_latency(VALID_PLAN_ID, 200.0)
class MetricsEmitterSuite:
"""Benchmark MetricsEmitter operations."""
def setup(self) -> None:
self._emitter = MetricsEmitter(enabled=True)
self._entries = [
MetricCollector.plan_duration(VALID_PLAN_ID, float(i * 100))
for i in range(100)
]
def time_emit_single(self) -> None:
"""Emit a single metric entry."""
self._emitter.emit(self._entries[0])
def time_emit_batch_100(self) -> None:
"""Emit 100 metric entries in a batch."""
self._emitter.emit_batch(self._entries)
def time_emit_disabled_noop(self) -> None:
"""Verify disabled emitter is fast no-op."""
disabled = MetricsEmitter(enabled=False)
disabled.emit_batch(self._entries)