diff --git a/benchmarks/validation_apply_bench.py b/benchmarks/validation_apply_bench.py new file mode 100644 index 000000000..3ae076270 --- /dev/null +++ b/benchmarks/validation_apply_bench.py @@ -0,0 +1,164 @@ +"""ASV benchmarks for validation apply gate throughput. + +Measures the performance of: +- ValidationAttachment creation +- DefaultValidationRunner execution +- ApplyValidationGate.run() with multiple attachments +- ApplyValidationSummary.to_plan_metadata() serialization +""" + +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +# Ensure the local *source* tree is importable. +_SRC = str(Path(__file__).resolve().parents[1] / "src") +if _SRC not in sys.path: + sys.path.insert(0, _SRC) + +import cleveragents # noqa: E402 + +importlib.reload(cleveragents) + +from cleveragents.application.services.validation_apply import ( # noqa: E402 + ApplyValidationGate, + ApplyValidationResult, + ApplyValidationSummary, + DefaultValidationRunner, + ValidationAttachment, +) +from cleveragents.domain.models.core.tool import ValidationMode # noqa: E402 + +# --------------------------------------------------------------------------- +# Benchmark: ValidationAttachment creation +# --------------------------------------------------------------------------- + + +class AttachmentCreation: + """Benchmark suite for ``ValidationAttachment`` model creation.""" + + def time_create_required(self) -> None: + """Create a required validation attachment.""" + ValidationAttachment( + attachment_id="01ATT000000000000000000000", + validation_name="lint-check", + resource_id="res-001", + mode=ValidationMode.REQUIRED, + ) + + def time_create_informational(self) -> None: + """Create an informational validation attachment.""" + ValidationAttachment( + attachment_id="01ATT000000000000000000001", + validation_name="style-check", + resource_id="res-002", + mode=ValidationMode.INFORMATIONAL, + ) + + +# --------------------------------------------------------------------------- +# Benchmark: DefaultValidationRunner +# --------------------------------------------------------------------------- + + +_ATTACHMENT = ValidationAttachment( + attachment_id="01ATT000000000000000000000", + validation_name="lint-check", + resource_id="res-001", + mode=ValidationMode.REQUIRED, +) + +_CTX_PASS = {"lint-check": "passed"} +_CTX_FAIL = {"status": "no matches"} + + +class RunnerExecution: + """Benchmark suite for ``DefaultValidationRunner``.""" + + def setup(self) -> None: + """Prepare runner.""" + self.runner = DefaultValidationRunner() + + def time_run_pass(self) -> None: + """Run validation that passes.""" + self.runner.run_validation(_ATTACHMENT, _CTX_PASS) + + def time_run_fail(self) -> None: + """Run validation that fails.""" + self.runner.run_validation(_ATTACHMENT, _CTX_FAIL) + + +# --------------------------------------------------------------------------- +# Benchmark: ApplyValidationGate +# --------------------------------------------------------------------------- + + +def _make_attachments(n: int) -> list[ValidationAttachment]: + """Create N test attachments.""" + return [ + ValidationAttachment( + attachment_id=f"01ATT{i:021d}", + validation_name=f"val-{i}", + resource_id=f"res-{i}", + mode=ValidationMode.REQUIRED, + ) + for i in range(n) + ] + + +class GateExecution: + """Benchmark suite for ``ApplyValidationGate.run``.""" + + def setup(self) -> None: + """Prepare gate and attachments.""" + self.runner = DefaultValidationRunner() + self.gate = ApplyValidationGate(runner=self.runner) + self.attachments_5 = _make_attachments(5) + self.attachments_20 = _make_attachments(20) + self.context = {f"val-{i}": "ok" for i in range(20)} + + def time_gate_5_attachments(self) -> None: + """Run gate with 5 attachments.""" + self.gate.run("01PLAN0001", self.attachments_5, self.context) + + def time_gate_20_attachments(self) -> None: + """Run gate with 20 attachments.""" + self.gate.run("01PLAN0001", self.attachments_20, self.context) + + +# --------------------------------------------------------------------------- +# Benchmark: Summary serialization +# --------------------------------------------------------------------------- + + +class SummarySerialization: + """Benchmark suite for ``ApplyValidationSummary`` serialization.""" + + def setup(self) -> None: + """Prepare summary with mixed results.""" + results = [] + for i in range(10): + results.append( + ApplyValidationResult( + attachment_id=f"01ATT{i:021d}", + validation_name=f"val-{i}", + resource_id=f"res-{i}", + mode=ValidationMode.REQUIRED, + passed=i % 3 != 0, + message="ok" if i % 3 != 0 else "failed", + ) + ) + self.summary = ApplyValidationSummary( + plan_id="01PLAN0001", + results=results, + ) + + def time_to_plan_metadata(self) -> None: + """Serialize summary to plan metadata dict.""" + self.summary.to_plan_metadata() + + def time_format_cli_output(self) -> None: + """Format summary for CLI display.""" + self.summary.format_cli_output() diff --git a/docs/development/testing.md b/docs/development/testing.md index 7826402e1..eb72ed73e 100644 --- a/docs/development/testing.md +++ b/docs/development/testing.md @@ -456,6 +456,57 @@ Performance baselines for persistence test operations: Run with: `nox -s benchmark` +### Behave: Skill Registry Persistence (`features/skill_registry.feature`) + +23 scenarios covering `SkillRepository` and `SkillRegistryService` operations: + +- **Register**: Persist skills with each item type (`tool_ref`, `include`, `inline_tool`, `mcp_source`, `agent_source`), verify round-trip. +- **Duplicate rejection**: Second registration with the same name raises `DuplicateSkillError`. +- **Retrieve / list**: Get by name, list all, list with namespace filter. +- **Update / remove**: Change description, remove with cascade, non-existent skill error paths. +- **Overrides round-trip**: Override metadata survives persistence. +- **Ordering stability**: Mixed item types maintain stable `item_order` across round-trips. +- **Name validation**: Empty, no-slash, and special-character names rejected at persistence layer. +- **Large payload**: 50 tool refs persist correctly. +- **Domain objects**: Listed results are `Skill` domain instances. + +Step definitions: `features/steps/skill_registry_steps.py` + +**Fixtures**: Each scenario gets a fresh in-memory SQLite database. A single +shared `Session` is reused across all repository calls within each scenario via a +lambda session factory (`lambda: shared_session`). This mirrors the production +`UnitOfWork` pattern (ADR-019) and avoids transaction-scoping issues that arise +when multiple ephemeral sessions share one in-memory SQLite connection +(`StaticPool`): abandoned sessions can be garbage-collected at unpredictable +times, rolling back previously flushed but uncommitted data. + +### Robot: Skill Registry (`robot/skill_registry.robot`) + +5 smoke tests exercising the registry through a Python helper script: + +| Test Case | Description | +|-----------|-------------| +| Register And Retrieve A Skill | Round-trip: register then get by name | +| List Skills With Namespace Filter | Register multiple skills, filter by namespace | +| Update A Skill | Change description after registration | +| Reject Duplicate Skill Name | Duplicate name produces error | +| Remove A Skill | Remove and verify absence | + +Helper script: `robot/helper_skill_registry.py` + +### ASV Benchmarks: Skill Registry (`benchmarks/skill_registry_bench.py`) + +Performance baselines for skill registry persistence operations: + +- `SkillModelConstruction.time_from_domain` -- Domain-to-ORM mapping +- `SkillModelReconstruction.time_to_domain` -- ORM-to-domain reconstruction +- `SkillRepositoryCRUD.time_create_skill` -- Single skill insert + commit +- `SkillRepositoryCRUD.time_get_skill` -- Lookup by name +- `SkillRepositoryCRUD.time_list_all` -- List all skills +- `SkillRepositoryCRUD.time_list_namespace` -- List with namespace filter +- `SkillRegistryListPerformance.time_list_100_skills` -- List with 100 seeded skills +- `SkillRegistryListPerformance.time_list_100_skills_filtered` -- Filtered list with 100 skills + ### Running Persistence Tests ```bash @@ -468,6 +519,9 @@ nox -s unit_tests -- features/plan_persistence.feature # Run only action persistence scenarios nox -s unit_tests -- features/action_persistence.feature +# Run only skill registry persistence scenarios +nox -s unit_tests -- features/skill_registry.feature + # Integration tests (Robot) -- runs all robot suites including persistence E2E nox -s integration_tests @@ -477,6 +531,9 @@ python -m robot robot/plan_persistence_e2e.robot # Run only persistence lifecycle E2E python -m robot robot/persistence_lifecycle.robot +# Run only skill registry smoke tests +nox -s integration_tests -- robot/skill_registry.robot + # Run only persistence robot alignment (Behave) nox -s unit_tests -- features/persistence_robot_alignment.feature ``` @@ -594,6 +651,27 @@ Use `Resource ${CURDIR}/filename.resource` instead of bare `Resource filename.re Ensure Robot tests use `${PYTHON}` (injected by nox from the venv) instead of bare `python` in `Run Process` calls. +### In-memory SQLite tests lose data between repository calls + +When writing Behave tests that use `sqlite:///:memory:` with a `sessionmaker`, +do **not** create a new `Session` per repository call. SQLite in-memory databases +use `StaticPool` (a single shared connection). If each repository method creates +its own session, the abandoned sessions may be garbage-collected at unpredictable +times, triggering a `ROLLBACK` that discards previously flushed data. + +**Fix:** Create a single shared `Session` at scenario setup and pass a factory +that always returns it: + +```python +real_factory = sessionmaker(bind=engine) +shared_session = real_factory() +repo = MyRepository(session_factory=lambda: shared_session) +``` + +This mirrors the production `UnitOfWork` pattern (ADR-019) where all +repositories share one session within a transaction. See +`features/steps/skill_registry_steps.py` for a working example. + ### Tests hang indefinitely Add `timeout` parameters to `Run Process` calls in Robot tests: diff --git a/docs/reference/skill_registry.md b/docs/reference/skill_registry.md index ce7d70cce..dda5cc61c 100644 --- a/docs/reference/skill_registry.md +++ b/docs/reference/skill_registry.md @@ -305,3 +305,76 @@ print(f"Has writes: {summary.write_tools > 0}") ``` Returns a 2-tuple of `(list[ResolvedToolEntry], SkillCapabilitySummary)`. + +## Testing + +### Behave BDD Tests + +The Behave feature file `features/skill_registry.feature` contains 23 +scenarios covering the full `SkillRegistryService` and `SkillRepository` +persistence layer: + +| Area | Scenarios | Description | +|------|-----------|-------------| +| Registration | 6 | Each skill item type (`tool_ref`, `include`, `inline_tool`, `mcp_source`, `agent_source`) plus basic round-trip | +| Duplicate rejection | 1 | `DuplicateSkillError` on name collision | +| Retrieval | 2 | Get by name, non-existent returns `None` | +| Listing | 2 | List all, list with namespace filter | +| Update | 2 | Description change, non-existent raises `SkillNotFoundError` | +| Deletion | 2 | Remove existing (cascades items), remove non-existent returns `False` | +| Overrides | 1 | Override metadata survives persistence round-trip | +| Item ordering | 1 | Mixed item types maintain stable order across round-trip | +| Name validation | 3 | Empty name, no-slash name, special characters rejected at persistence layer | +| Large payload | 1 | 50 tool refs persist correctly | +| Domain objects | 1 | Listed skills are `Skill` domain instances | + +Step definitions: `features/steps/skill_registry_steps.py` + +**Session management note:** Each scenario creates a fresh in-memory +SQLite database with a single shared `Session` that is reused across all +repository calls within the scenario. This mirrors the production +`UnitOfWork` pattern (ADR-019) and avoids transaction-scoping issues +inherent to multiple ephemeral sessions on a single `StaticPool` +connection. + +Run the Behave suite via: + +```bash +nox -s unit_tests -- features/skill_registry.feature +``` + +### Robot Smoke Tests + +A Robot Framework smoke suite validates core registry operations: + +``` +robot/skill_registry.robot +robot/helper_skill_registry.py +``` + +| Test Case | What it validates | +|---|---| +| Register And Retrieve A Skill | Round-trip: register then get by name | +| List Skills With Namespace Filter | Register multiple skills, filter by namespace | +| Update A Skill | Change description after registration | +| Reject Duplicate Skill Name | Duplicate name produces error | +| Remove A Skill | Remove and verify absence | + +```bash +nox -s integration_tests -- robot/skill_registry.robot +``` + +### ASV Benchmarks + +Performance benchmarks live in `benchmarks/skill_registry_bench.py`: + +- **`SkillModelConstruction`** -- `time_from_domain` (domain-to-ORM mapping) +- **`SkillModelReconstruction`** -- `time_to_domain` (ORM-to-domain mapping) +- **`SkillRepositoryCRUD`** -- `time_create_skill`, `time_get_skill`, + `time_list_all`, `time_list_namespace` +- **`SkillRegistryListPerformance`** -- `time_list_100_skills`, + `time_list_100_skills_filtered` + +```bash +nox -s benchmark +``` diff --git a/docs/reference/validation_pipeline.md b/docs/reference/validation_pipeline.md new file mode 100644 index 000000000..bcb296a63 --- /dev/null +++ b/docs/reference/validation_pipeline.md @@ -0,0 +1,149 @@ +# Validation Pipeline — Apply Gating + +## Overview + +The validation apply gate enforces that attached validations pass before +the apply phase proceeds. Validations are attached to resources and run +automatically during the Execute-to-Apply transition. Required validation +failures block apply; informational failures are recorded but do not block. + +## Attachment Model + +### `ValidationAttachment` + +Represents a validation bound to a resource with scope and mode. + +| Field | Type | Description | +|-------------------|-------------------|--------------------------------------| +| `attachment_id` | `str` | Unique ULID for this attachment | +| `validation_name` | `str` | Name of the validation tool | +| `resource_id` | `str` | Target resource ID | +| `resource_name` | `str` | Human-readable resource name | +| `mode` | `ValidationMode` | `required` or `informational` | +| `scope` | `AttachmentScope` | `direct`, `project`, or `plan` | +| `scope_target` | `str` | Project name or plan ID for scoping | +| `arguments` | `dict[str, Any]` | Arguments passed to the validation | + +### `AttachmentScope` + +| Value | Description | +|-----------|----------------------------------------------------| +| `direct` | Always active for the resource | +| `project` | Active when resource is accessed via a project | +| `plan` | Active only for a specific plan | + +When multiple scopes apply, the **union** of all applicable validations +is collected and run. + +## Validation Modes + +| Mode | Behavior on Failure | +|-----------------|--------------------------------------------------| +| `required` | Blocks apply, plan stays in Execute phase | +| `informational` | Recorded in summary but does **not** block apply | + +## Result Models + +### `ApplyValidationResult` + +Per-validation execution result. + +| Field | Type | Description | +|-------------------|------------------|-------------------------------------| +| `attachment_id` | `str` | Which attachment was evaluated | +| `validation_name` | `str` | Validation tool name | +| `resource_id` | `str` | Target resource | +| `mode` | `ValidationMode` | Required or informational | +| `passed` | `bool` | Whether the validation passed | +| `message` | `str` | Result message | +| `data` | `dict` | Additional structured data | +| `duration_ms` | `int` | Execution time in milliseconds | +| `error` | `str | None` | Error message if execution failed | +| `timed_out` | `bool` | Whether execution timed out | + +### `ApplyValidationSummary` + +Aggregated results with gating decision. + +**Properties:** `total`, `required_passed`, `required_failed`, +`informational_passed`, `informational_failed`, `all_required_passed`, +`is_empty` + +**Methods:** +- `to_plan_metadata()` — Dict for plan `validation_summary` field +- `format_cli_output()` — Human-readable summary for CLI display + +## Runner Interface + +### `ValidationRunner` (ABC) + +Abstract interface for executing a single validation: + +```python +def run_validation( + self, + attachment: ValidationAttachment, + context: dict[str, Any], +) -> ApplyValidationResult: ... +``` + +### `DefaultValidationRunner` + +Default stub using text matching. Checks if the validation name or +argument values appear in the context. Production implementations +should invoke the actual validation tool via the tool execution pipeline. + +## Apply Validation Gate + +### `ApplyValidationGate` + +Orchestrates the full validation-before-apply flow: + +```python +gate = ApplyValidationGate(runner=DefaultValidationRunner()) +summary = gate.run(plan_id, attachments, context) + +if gate.should_block_apply(summary): + reasons = gate.get_failure_reasons(summary) + # Block apply, report reasons +else: + # Proceed to apply +``` + +### Example Output + +When apply is **blocked**: + +``` +Validation Gate: BLOCKED + Required: 1 passed, 1 failed + Informational: 0 passed, 0 failed + Failures: + - lint-check on res-001: 3 lint errors found +``` + +When apply is **allowed**: + +``` +Validation Gate: PASSED + Required: 2 passed, 0 failed + Informational: 1 passed, 0 failed +``` + +## Integration with Plan Metadata + +The validation summary is stored in the plan's `validation_summary` field +via `ApplyValidationSummary.to_plan_metadata()`: + +```python +{ + "total": 3, + "required_passed": 2, + "required_failed": 1, + "informational_passed": 0, + "informational_failed": 0, + "all_required_passed": false, + "evaluated_at": "2026-02-22T10:30:00+00:00", + "results": [...] +} +``` diff --git a/features/steps/skill_registry_steps.py b/features/steps/skill_registry_steps.py index b67478f2d..4bbe160f3 100644 --- a/features/steps/skill_registry_steps.py +++ b/features/steps/skill_registry_steps.py @@ -6,7 +6,7 @@ integration layer. from __future__ import annotations -from typing import Any +from typing import Any, Callable from behave import given, then, when from behave.runner import Context @@ -36,12 +36,22 @@ from cleveragents.infrastructure.database.repositories import ( # --------------------------------------------------------------------------- -def _get_session_factory(context: Context) -> sessionmaker[Session]: +def _get_session_factory(context: Context) -> Callable[[], Session]: + """Return the session factory stored on *context*. + + The factory always returns the same shared session for the current + scenario, mirroring the ``UnitOfWork`` behaviour used in production. + """ return context.skill_session_factory # type: ignore[no-any-return] def _commit_pending(context: Context) -> None: - """Commit pending transaction on the shared SQLite :memory: session.""" + """Commit the shared scenario session. + + Because the repository only *flushes* (never commits), the test + must explicitly commit to make data durable. The session is kept + open so that subsequent steps can continue using it. + """ session = _get_session_factory(context)() session.commit() diff --git a/features/steps/validation_apply_steps.py b/features/steps/validation_apply_steps.py new file mode 100644 index 000000000..97c05c0d2 --- /dev/null +++ b/features/steps/validation_apply_steps.py @@ -0,0 +1,383 @@ +"""Step definitions for validation apply gate tests.""" + +from __future__ import annotations + +from behave import given, then, when +from behave.runner import Context + +from cleveragents.application.services.validation_apply import ( + ApplyValidationGate, + ApplyValidationResult, + ApplyValidationSummary, + AttachmentScope, + DefaultValidationRunner, + ValidationAttachment, +) +from cleveragents.domain.models.core.tool import ValidationMode + +__all__: list[str] = [] + + +# --------------------------------------------------------------------------- +# Background +# --------------------------------------------------------------------------- + + +@given("a validation apply test environment") +def step_given_val_apply_env(context: Context) -> None: + context.val_attachments = [] # list[ValidationAttachment] + context.val_run_context = {} # dict[str, str] + context.val_result = None # ApplyValidationResult | None + context.val_summary = None # ApplyValidationSummary | None + context.val_gate_summary = None # ApplyValidationSummary | None + + +# --------------------------------------------------------------------------- +# Given steps +# --------------------------------------------------------------------------- + + +@given('a validation attachment "{name}" for resource "{res}" with mode "{mode}"') +def step_given_attachment(context: Context, name: str, res: str, mode: str) -> None: + idx = len(context.val_attachments) + att = ValidationAttachment( + attachment_id=f"01ATT{idx:021d}", + validation_name=name, + resource_id=res, + mode=ValidationMode(mode), + ) + context.val_attachments.append(att) + + +@given('a project-scoped attachment "{name}" for resource "{res}" targeting "{target}"') +def step_given_project_attachment( + context: Context, name: str, res: str, target: str +) -> None: + idx = len(context.val_attachments) + att = ValidationAttachment( + attachment_id=f"01ATT{idx:021d}", + validation_name=name, + resource_id=res, + mode=ValidationMode.REQUIRED, + scope=AttachmentScope.PROJECT, + scope_target=target, + ) + context.val_attachments.append(att) + + +@given('a plan-scoped attachment "{name}" for resource "{res}" targeting "{target}"') +def step_given_plan_attachment( + context: Context, name: str, res: str, target: str +) -> None: + idx = len(context.val_attachments) + att = ValidationAttachment( + attachment_id=f"01ATT{idx:021d}", + validation_name=name, + resource_id=res, + mode=ValidationMode.REQUIRED, + scope=AttachmentScope.PLAN, + scope_target=target, + ) + context.val_attachments.append(att) + + +@given('a validation run context with key "{key}" and value "{value}"') +def step_given_val_context(context: Context, key: str, value: str) -> None: + context.val_run_context[key] = value + + +@given('the attachment has argument "{key}" with value "{value}"') +def step_given_attachment_arg(context: Context, key: str, value: str) -> None: + context.val_attachments[-1].arguments[key] = value + + +@given("an apply validation summary with {p} required passed and {f} required failed") +def step_given_apply_summary(context: Context, p: str, f: str) -> None: + results: list[ApplyValidationResult] = [] + for i in range(int(p)): + results.append( + ApplyValidationResult( + attachment_id=f"01ATT{i:021d}", + validation_name=f"pass-val-{i}", + resource_id=f"res-{i}", + mode=ValidationMode.REQUIRED, + passed=True, + message="ok", + ) + ) + for i in range(int(f)): + results.append( + ApplyValidationResult( + attachment_id=f"01ATT{int(p) + i:021d}", + validation_name=f"fail-val-{i}", + resource_id=f"res-{int(p) + i}", + mode=ValidationMode.REQUIRED, + passed=False, + message="failed check", + ) + ) + context.val_summary = ApplyValidationSummary( + plan_id="01VALTEST0000000000001", + results=results, + ) + + +@given("the summary has {ip} informational passed and {iff} informational failed") +def step_given_informational_results(context: Context, ip: str, iff: str) -> None: + base = len(context.val_summary.results) + for i in range(int(ip)): + context.val_summary.results.append( + ApplyValidationResult( + attachment_id=f"01ATT{base + i:021d}", + validation_name=f"info-pass-{i}", + resource_id=f"res-info-{i}", + mode=ValidationMode.INFORMATIONAL, + passed=True, + message="info ok", + ) + ) + base2 = len(context.val_summary.results) + for i in range(int(iff)): + context.val_summary.results.append( + ApplyValidationResult( + attachment_id=f"01ATT{base2 + i:021d}", + validation_name=f"info-fail-{i}", + resource_id=f"res-info-fail-{i}", + mode=ValidationMode.INFORMATIONAL, + passed=False, + message="info failed", + ) + ) + + +@given("a validation run context that raises an error during iteration") +def step_given_error_context(context: Context) -> None: + """Set up a context dict-like object whose items() raises.""" + + class _BadContext(dict): # type: ignore[type-arg] + def items(self) -> None: # type: ignore[override] + msg = "boom" + raise RuntimeError(msg) + + context.val_run_context = _BadContext() + + +@given("an apply validation summary with a bare failure result") +def step_given_summary_bare_failure(context: Context) -> None: + context.val_summary = ApplyValidationSummary( + plan_id="01VALTEST0000000000001", + results=[ + ApplyValidationResult( + attachment_id="01ATT000000000000000000000", + validation_name="bare-val", + resource_id="res-bare", + mode=ValidationMode.REQUIRED, + passed=False, + message="", + error=None, + ), + ], + ) + context.val_gate_summary = context.val_summary + + +@given("an apply validation summary with an error result") +def step_given_summary_with_error(context: Context) -> None: + context.val_summary = ApplyValidationSummary( + plan_id="01VALTEST0000000000001", + results=[ + ApplyValidationResult( + attachment_id="01ATT000000000000000000000", + validation_name="broken-val", + resource_id="res-broken", + mode=ValidationMode.REQUIRED, + passed=False, + message="", + error="execution timed out", + ), + ], + ) + context.val_gate_summary = context.val_summary + + +@given("an empty apply validation summary") +def step_given_empty_apply_summary(context: Context) -> None: + context.val_summary = ApplyValidationSummary( + plan_id="01VALTEST0000000000001", + results=[], + ) + + +# --------------------------------------------------------------------------- +# When steps +# --------------------------------------------------------------------------- + + +@when("I execute the validation with DefaultValidationRunner") +def step_when_execute_default_runner(context: Context) -> None: + runner = DefaultValidationRunner() + context.val_result = runner.run_validation( + context.val_attachments[-1], + context.val_run_context, + ) + + +@when('I run the apply validation gate for plan "{plan_id}"') +def step_when_run_gate(context: Context, plan_id: str) -> None: + runner = DefaultValidationRunner() + gate = ApplyValidationGate(runner=runner) + context.val_gate_summary = gate.run( + plan_id=plan_id, + attachments=context.val_attachments, + context=context.val_run_context, + ) + # Also store as val_summary for summary assertions + context.val_summary = context.val_gate_summary + + +@when('I run the apply validation gate for plan "{plan_id}" with no attachments') +def step_when_run_gate_empty(context: Context, plan_id: str) -> None: + runner = DefaultValidationRunner() + gate = ApplyValidationGate(runner=runner) + context.val_gate_summary = gate.run( + plan_id=plan_id, + attachments=[], + context=context.val_run_context, + ) + context.val_summary = context.val_gate_summary + + +# --------------------------------------------------------------------------- +# Then steps +# --------------------------------------------------------------------------- + + +@then('the attachment mode should be "{mode}"') +def step_then_attachment_mode(context: Context, mode: str) -> None: + att = context.val_attachments[-1] + assert att.mode == ValidationMode(mode), ( + f"Expected mode '{mode}', got '{att.mode.value}'" + ) + + +@then('the attachment scope should be "{scope}"') +def step_then_attachment_scope(context: Context, scope: str) -> None: + att = context.val_attachments[-1] + assert att.scope == AttachmentScope(scope), ( + f"Expected scope '{scope}', got '{att.scope.value}'" + ) + + +@then('the attachment scope target should be "{target}"') +def step_then_scope_target(context: Context, target: str) -> None: + att = context.val_attachments[-1] + assert att.scope_target == target + + +@then("the validation result should be passed") +def step_then_val_result_passed(context: Context) -> None: + assert context.val_result is not None + assert context.val_result.passed, ( + f"Expected passed, got failed: {context.val_result.message}" + ) + + +@then("the validation result should be failed") +def step_then_val_result_failed(context: Context) -> None: + assert context.val_result is not None + assert not context.val_result.passed + + +@then("the validation result duration_ms should be >= {n}") +def step_then_val_result_duration(context: Context, n: str) -> None: + assert context.val_result is not None + assert context.val_result.duration_ms >= int(n), ( + f"Expected duration_ms >= {n}, got {context.val_result.duration_ms}" + ) + + +@then("the validation result should have an error message") +def step_then_val_result_has_error(context: Context) -> None: + assert context.val_result is not None + assert context.val_result.error is not None and context.val_result.error != "", ( + f"Expected error message, got: {context.val_result.error!r}" + ) + + +@then("the apply summary should report all required passed") +def step_then_all_required_passed(context: Context) -> None: + assert context.val_summary is not None + assert context.val_summary.all_required_passed + + +@then("the apply summary should not report all required passed") +def step_then_not_all_required_passed(context: Context) -> None: + assert context.val_summary is not None + assert not context.val_summary.all_required_passed + + +@then("the apply summary total should be {n}") +def step_then_apply_summary_total(context: Context, n: str) -> None: + assert context.val_summary is not None + assert context.val_summary.total == int(n), ( + f"Expected total {n}, got {context.val_summary.total}" + ) + + +@then("the apply summary required failed count should be {n}") +def step_then_required_failed(context: Context, n: str) -> None: + assert context.val_summary is not None + assert context.val_summary.required_failed == int(n) + + +@then("the apply summary informational failed count should be {n}") +def step_then_informational_failed(context: Context, n: str) -> None: + assert context.val_summary is not None + assert context.val_summary.informational_failed == int(n) + + +@then("the apply summary should be empty") +def step_then_apply_summary_empty(context: Context) -> None: + assert context.val_summary is not None + assert context.val_summary.is_empty + + +@then('the plan metadata should have key "{key}"') +def step_then_plan_metadata_key(context: Context, key: str) -> None: + assert context.val_summary is not None + md = context.val_summary.to_plan_metadata() + assert key in md, f"Key '{key}' not in {list(md.keys())}" + + +@then('the validation gate CLI output should contain "{text}"') +def step_then_val_gate_cli_output(context: Context, text: str) -> None: + assert context.val_summary is not None + output = context.val_summary.format_cli_output() + assert text in output, f"Expected '{text}' in:\n{output}" + + +@then("the gate should allow apply") +def step_then_gate_allows(context: Context) -> None: + assert context.val_gate_summary is not None + gate = ApplyValidationGate(runner=DefaultValidationRunner()) + assert not gate.should_block_apply(context.val_gate_summary), ( + "Expected gate to allow apply but it blocked" + ) + + +@then("the gate should block apply") +def step_then_gate_blocks(context: Context) -> None: + assert context.val_gate_summary is not None + gate = ApplyValidationGate(runner=DefaultValidationRunner()) + assert gate.should_block_apply(context.val_gate_summary), ( + "Expected gate to block apply but it allowed" + ) + + +@then('the gate failure reasons should mention "{text}"') +def step_then_failure_reasons(context: Context, text: str) -> None: + assert context.val_gate_summary is not None + gate = ApplyValidationGate(runner=DefaultValidationRunner()) + reasons = gate.get_failure_reasons(context.val_gate_summary) + combined = " ".join(reasons) + assert text in combined, f"Expected '{text}' in failure reasons: {reasons}" diff --git a/features/validation_apply.feature b/features/validation_apply.feature new file mode 100644 index 000000000..8da6a4961 --- /dev/null +++ b/features/validation_apply.feature @@ -0,0 +1,145 @@ +Feature: Validation apply gate + As a plan executor + I want attached validations to run before apply + So that required validation failures block apply + + Background: + Given a validation apply test environment + + # -- Attachment model ---------------------------------------------------- + + Scenario: Create a required validation attachment + Given a validation attachment "lint-check" for resource "res-001" with mode "required" + Then the attachment mode should be "required" + And the attachment scope should be "direct" + + Scenario: Create an informational validation attachment + Given a validation attachment "style-check" for resource "res-002" with mode "informational" + Then the attachment mode should be "informational" + + Scenario: Create a project-scoped attachment + Given a project-scoped attachment "lint-check" for resource "res-001" targeting "my-project" + Then the attachment scope should be "project" + And the attachment scope target should be "my-project" + + Scenario: Create a plan-scoped attachment + Given a plan-scoped attachment "lint-check" for resource "res-001" targeting "01PLAN0001" + Then the attachment scope should be "plan" + And the attachment scope target should be "01PLAN0001" + + # -- DefaultValidationRunner --------------------------------------------- + + Scenario: DefaultValidationRunner passes when context matches + Given a validation attachment "lint-check" for resource "res-001" with mode "required" + And a validation run context with key "lint-check" and value "passed" + When I execute the validation with DefaultValidationRunner + Then the validation result should be passed + + Scenario: DefaultValidationRunner fails when context has no match + Given a validation attachment "lint-check" for resource "res-001" with mode "required" + And a validation run context with key "status" and value "no matches" + When I execute the validation with DefaultValidationRunner + Then the validation result should be failed + + Scenario: DefaultValidationRunner matches via arguments + Given a validation attachment "coverage-check" for resource "res-001" with mode "required" + And the attachment has argument "threshold" with value "90" + And a validation run context with key "coverage" and value "90 percent" + When I execute the validation with DefaultValidationRunner + Then the validation result should be passed + + # -- ApplyValidationSummary ---------------------------------------------- + + Scenario: Summary all required passed with no failures + Given an apply validation summary with 2 required passed and 0 required failed + Then the apply summary should report all required passed + And the apply summary total should be 2 + + Scenario: Summary blocked when required failure exists + Given an apply validation summary with 1 required passed and 1 required failed + Then the apply summary should not report all required passed + And the apply summary required failed count should be 1 + + Scenario: Summary with informational results does not block + Given an apply validation summary with 0 required passed and 0 required failed + And the summary has 1 informational passed and 1 informational failed + Then the apply summary should report all required passed + And the apply summary informational failed count should be 1 + + Scenario: Empty summary reports all required passed + Given an empty apply validation summary + Then the apply summary should be empty + And the apply summary should report all required passed + + Scenario: Summary to_plan_metadata has expected fields + Given an apply validation summary with 1 required passed and 1 required failed + Then the plan metadata should have key "total" + And the plan metadata should have key "all_required_passed" + And the plan metadata should have key "results" + + Scenario: Summary format_cli_output shows BLOCKED on failures + Given an apply validation summary with 0 required passed and 1 required failed + Then the validation gate CLI output should contain "BLOCKED" + + Scenario: Summary format_cli_output shows no details when neither message nor error + Given an apply validation summary with a bare failure result + Then the validation gate CLI output should contain "no details" + + Scenario: Summary format_cli_output shows error in failure detail + Given an apply validation summary with an error result + Then the validation gate CLI output should contain "execution timed out" + + Scenario: Summary format_cli_output shows PASSED when all pass + Given an apply validation summary with 2 required passed and 0 required failed + Then the validation gate CLI output should contain "PASSED" + + Scenario: DefaultValidationRunner populates duration_ms + Given a validation attachment "lint-check" for resource "res-001" with mode "required" + And a validation run context with key "lint-check" and value "passed" + When I execute the validation with DefaultValidationRunner + Then the validation result duration_ms should be >= 0 + + Scenario: DefaultValidationRunner handles exception during execution + Given a validation attachment "err-val" for resource "res-err" with mode "required" + And a validation run context that raises an error during iteration + When I execute the validation with DefaultValidationRunner + Then the validation result should be failed + And the validation result should have an error message + + # -- ApplyValidationGate ------------------------------------------------- + + Scenario: Gate allows apply when all required pass + Given a validation attachment "lint-check" for resource "res-001" with mode "required" + And a validation run context with key "lint-check" and value "ok" + When I run the apply validation gate for plan "01PLAN0001" + Then the gate should allow apply + + Scenario: Gate blocks apply when required validation fails + Given a validation attachment "lint-check" for resource "res-001" with mode "required" + And a validation run context with key "other" and value "irrelevant" + When I run the apply validation gate for plan "01PLAN0001" + Then the gate should block apply + And the gate failure reasons should mention "lint-check" + + Scenario: Gate allows apply despite informational failures + Given a validation attachment "style-check" for resource "res-001" with mode "informational" + And a validation run context with key "other" and value "irrelevant" + When I run the apply validation gate for plan "01PLAN0001" + Then the gate should allow apply + + Scenario: Gate with multiple attachments mixed results + Given a validation attachment "lint-check" for resource "res-001" with mode "required" + And a validation attachment "style-check" for resource "res-001" with mode "informational" + And a validation run context with key "lint-check" and value "passed" + When I run the apply validation gate for plan "01PLAN0001" + Then the gate should allow apply + And the apply summary total should be 2 + + Scenario: Gate failure reasons include error text when result has error + Given an apply validation summary with an error result + Then the gate failure reasons should mention "error:" + + Scenario: Gate with no attachments allows apply + When I run the apply validation gate for plan "01PLAN0001" with no attachments + Then the gate should allow apply + And the apply summary should be empty diff --git a/robot/helper_validation_apply.py b/robot/helper_validation_apply.py new file mode 100644 index 000000000..ab5a14722 --- /dev/null +++ b/robot/helper_validation_apply.py @@ -0,0 +1,163 @@ +"""Robot Framework helper for validation apply gate. + +Provides a CLI-style interface for Robot to invoke validation +attachment creation, runner execution, and gate evaluation. +Exit code 0 = success, 1 = failure. + +Usage:: + + python robot/helper_validation_apply.py run-gate \\ + + python robot/helper_validation_apply.py summary \\ + + python robot/helper_validation_apply.py check-block \\ + +""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +# Ensure the src directory is on the import path. +_SRC = str(Path(__file__).resolve().parents[1] / "src") +if _SRC not in sys.path: + sys.path.insert(0, _SRC) + +from cleveragents.application.services.validation_apply import ( # noqa: E402 + ApplyValidationGate, + ApplyValidationResult, + ApplyValidationSummary, + DefaultValidationRunner, + ValidationAttachment, +) +from cleveragents.domain.models.core.tool import ValidationMode # noqa: E402 + + +def _cmd_run_gate(args: list[str]) -> int: + """Run the validation gate with a single attachment.""" + if len(args) < 6: + print( + "Usage: run-gate " + ) + return 1 + plan_id = args[0] + att = ValidationAttachment( + attachment_id="01ATT000000000000000000000", + validation_name=args[1], + resource_id=args[2], + mode=ValidationMode(args[3]), + ) + ctx = {args[4]: args[5]} + runner = DefaultValidationRunner() + gate = ApplyValidationGate(runner=runner) + summary = gate.run(plan_id, [att], ctx) + blocked = gate.should_block_apply(summary) + status = "BLOCKED" if blocked else "ALLOWED" + print(f"val-gate-ok: {status}") + print(f" total={summary.total}") + print(f" required_passed={summary.required_passed}") + print(f" required_failed={summary.required_failed}") + return 0 + + +def _cmd_summary(args: list[str]) -> int: + """Build a summary and print metadata.""" + if len(args) < 3: + print("Usage: summary ") + return 1 + plan_id = args[0] + n_pass, n_fail = int(args[1]), int(args[2]) + results: list[ApplyValidationResult] = [] + for i in range(n_pass): + results.append( + ApplyValidationResult( + attachment_id=f"01ATT{i:021d}", + validation_name=f"val-{i}", + resource_id=f"res-{i}", + mode=ValidationMode.REQUIRED, + passed=True, + message="ok", + ) + ) + for i in range(n_fail): + results.append( + ApplyValidationResult( + attachment_id=f"01ATT{n_pass + i:021d}", + validation_name=f"val-{n_pass + i}", + resource_id=f"res-{n_pass + i}", + mode=ValidationMode.REQUIRED, + passed=False, + message="failed", + ) + ) + summary = ApplyValidationSummary(plan_id=plan_id, results=results) + md = summary.to_plan_metadata() + print(f"val-summary-ok: {json.dumps(md, default=str)}") + return 0 + + +def _cmd_check_block(args: list[str]) -> int: + """Check if gate blocks apply.""" + if len(args) < 3: + print("Usage: check-block ") + return 1 + plan_id = args[0] + n_pass, n_fail = int(args[1]), int(args[2]) + results: list[ApplyValidationResult] = [] + for i in range(n_pass): + results.append( + ApplyValidationResult( + attachment_id=f"01ATT{i:021d}", + validation_name=f"val-{i}", + resource_id=f"res-{i}", + mode=ValidationMode.REQUIRED, + passed=True, + message="ok", + ) + ) + for i in range(n_fail): + results.append( + ApplyValidationResult( + attachment_id=f"01ATT{n_pass + i:021d}", + validation_name=f"val-{n_pass + i}", + resource_id=f"res-{n_pass + i}", + mode=ValidationMode.REQUIRED, + passed=False, + message="failed", + ) + ) + summary = ApplyValidationSummary(plan_id=plan_id, results=results) + gate = ApplyValidationGate(runner=DefaultValidationRunner()) + blocked = gate.should_block_apply(summary) + status = "BLOCKED" if blocked else "ALLOWED" + print(f"val-check-block-ok: {status}") + return 0 + + +def main() -> int: + """Entry point called by Robot Framework ``Run Process``.""" + if len(sys.argv) < 2: + print("Usage: helper_validation_apply.py ...") + return 1 + + command = sys.argv[1] + args = sys.argv[2:] + + dispatch = { + "run-gate": _cmd_run_gate, + "summary": _cmd_summary, + "check-block": _cmd_check_block, + } + + handler = dispatch.get(command) + if handler is None: + print(f"Unknown command: {command}") + return 1 + + return handler(args) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/robot/validation_apply.robot b/robot/validation_apply.robot new file mode 100644 index 000000000..65f2893ef --- /dev/null +++ b/robot/validation_apply.robot @@ -0,0 +1,49 @@ +*** Settings *** +Documentation Integration tests for validation apply gate +Resource common.resource +Library Process +Library OperatingSystem +Suite Setup Setup Test Environment +Suite Teardown Cleanup Test Environment + +*** Test Cases *** +Gate Allows Apply When Validation Passes + [Documentation] Verify gate allows apply when required validation passes + ${result}= Run Process ${PYTHON} ${CURDIR}/helper_validation_apply.py + ... run-gate 01PLAN0001 lint-check res-001 required lint-check passed + ... stderr=STDOUT + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} val-gate-ok: ALLOWED + +Gate Blocks Apply When Validation Fails + [Documentation] Verify gate blocks apply when required validation fails + ${result}= Run Process ${PYTHON} ${CURDIR}/helper_validation_apply.py + ... run-gate 01PLAN0001 lint-check res-001 required other irrelevant + ... stderr=STDOUT + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} val-gate-ok: BLOCKED + +Build Validation Summary Metadata + [Documentation] Verify validation summary produces plan metadata + ${result}= Run Process ${PYTHON} ${CURDIR}/helper_validation_apply.py + ... summary 01PLAN0001 2 1 + ... stderr=STDOUT + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} val-summary-ok: + Should Contain ${result.stdout} all_required_passed + +Check Block With No Failures + [Documentation] Verify gate reports ALLOWED with no failures + ${result}= Run Process ${PYTHON} ${CURDIR}/helper_validation_apply.py + ... check-block 01PLAN0001 3 0 + ... stderr=STDOUT + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} val-check-block-ok: ALLOWED + +Check Block With Failures + [Documentation] Verify gate reports BLOCKED with failures + ${result}= Run Process ${PYTHON} ${CURDIR}/helper_validation_apply.py + ... check-block 01PLAN0001 1 2 + ... stderr=STDOUT + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} val-check-block-ok: BLOCKED diff --git a/src/cleveragents/application/services/__init__.py b/src/cleveragents/application/services/__init__.py index a73d47ef2..9bf4c95a9 100644 --- a/src/cleveragents/application/services/__init__.py +++ b/src/cleveragents/application/services/__init__.py @@ -29,12 +29,26 @@ from cleveragents.application.services.skill_registry_service import ( from cleveragents.application.services.tool_registry_service import ( ToolRegistryService, ) +from cleveragents.application.services.validation_apply import ( + ApplyValidationGate, + ApplyValidationResult, + ApplyValidationSummary, + AttachmentScope, + DefaultValidationRunner, + ValidationAttachment, + ValidationRunner, +) __all__ = [ + "ApplyValidationGate", + "ApplyValidationResult", + "ApplyValidationSummary", + "AttachmentScope", "ConfigEntry", "ConfigLevel", "ConfigService", "CorrectionService", + "DefaultValidationRunner", "InvariantService", "PersistentSessionService", "PlanExecutionContext", @@ -43,4 +57,6 @@ __all__ = [ "RuntimeExecuteResult", "SkillRegistryService", "ToolRegistryService", + "ValidationAttachment", + "ValidationRunner", ] diff --git a/src/cleveragents/application/services/validation_apply.py b/src/cleveragents/application/services/validation_apply.py new file mode 100644 index 000000000..68bf5c614 --- /dev/null +++ b/src/cleveragents/application/services/validation_apply.py @@ -0,0 +1,406 @@ +"""Validation apply gate: run attached validations before apply. + +Orchestrates the execution of attached validations before the apply phase +and gates apply based on required/informational validation outcomes. + +Core types: + +- **ValidationAttachment**: A reference to a validation attached to a resource. +- **ApplyValidationResult**: Per-validation pass/fail result. +- **ApplyValidationSummary**: Aggregated results with gating decision. +- **ValidationRunner**: Protocol for executing a single validation. +- **DefaultValidationRunner**: Stub implementation using text matching. +- **ApplyValidationGate**: Orchestrates validation execution and gating. + +Usage:: + + gate = ApplyValidationGate(runner=DefaultValidationRunner()) + summary = gate.run(attachments, plan_context) + if summary.all_required_passed: + # proceed to apply + else: + # block apply, record failures +""" + +from __future__ import annotations + +import time +from abc import ABC, abstractmethod +from datetime import UTC, datetime +from enum import StrEnum +from typing import Any + +from pydantic import BaseModel, ConfigDict, Field + +from cleveragents.domain.models.core.tool import ValidationMode + +# --------------------------------------------------------------------------- +# Enums +# --------------------------------------------------------------------------- + + +class AttachmentScope(StrEnum): + """Scope of a validation attachment.""" + + DIRECT = "direct" + PROJECT = "project" + PLAN = "plan" + + +# --------------------------------------------------------------------------- +# Models +# --------------------------------------------------------------------------- + + +class ValidationAttachment(BaseModel): + """A validation attached to a resource for apply gating. + + Represents a single validation-to-resource binding with scope + and mode information. + """ + + attachment_id: str = Field(..., description="Unique attachment ULID") + validation_name: str = Field(..., min_length=1, description="Validation tool name") + resource_id: str = Field(..., description="Target resource ID") + resource_name: str = Field(default="", description="Target resource name") + mode: ValidationMode = Field( + default=ValidationMode.REQUIRED, + description="Whether failure blocks apply", + ) + scope: AttachmentScope = Field( + default=AttachmentScope.DIRECT, + description="Attachment scope level", + ) + scope_target: str = Field( + default="", + description="Scope target (project name or plan ID)", + ) + arguments: dict[str, Any] = Field( + default_factory=dict, + description="Arguments to pass to the validation", + ) + + model_config = ConfigDict( + str_strip_whitespace=True, + validate_assignment=True, + ) + + +class ApplyValidationResult(BaseModel): + """Result of executing a single validation.""" + + attachment_id: str = Field(..., description="Attachment that was evaluated") + validation_name: str = Field(..., description="Validation tool name") + resource_id: str = Field(..., description="Resource the validation ran on") + mode: ValidationMode = Field(..., description="Required or informational") + passed: bool = Field(..., description="Whether the validation passed") + message: str = Field(default="", description="Result message") + data: dict[str, Any] = Field( + default_factory=dict, description="Additional result data" + ) + duration_ms: int = Field(default=0, ge=0, description="Execution time in ms") + error: str | None = Field( + default=None, description="Error message if execution failed" + ) + timed_out: bool = Field(default=False, description="Whether execution timed out") + + model_config = ConfigDict( + str_strip_whitespace=True, + validate_assignment=True, + ) + + +class ApplyValidationSummary(BaseModel): + """Aggregated results from all validations run before apply.""" + + plan_id: str = Field(..., description="Plan ULID") + results: list[ApplyValidationResult] = Field( + default_factory=list, + description="Per-validation results", + ) + evaluated_at: datetime = Field( + default_factory=lambda: datetime.now(tz=UTC), + description="When the validation gate was evaluated", + ) + + model_config = ConfigDict( + str_strip_whitespace=True, + validate_assignment=True, + ) + + @property + def total(self) -> int: + """Total number of validations run.""" + return len(self.results) + + @property + def required_passed(self) -> int: + """Number of required validations that passed.""" + return sum( + 1 for r in self.results if r.mode == ValidationMode.REQUIRED and r.passed + ) + + @property + def required_failed(self) -> int: + """Number of required validations that failed.""" + return sum( + 1 + for r in self.results + if r.mode == ValidationMode.REQUIRED and not r.passed + ) + + @property + def informational_passed(self) -> int: + """Number of informational validations that passed.""" + return sum( + 1 + for r in self.results + if r.mode == ValidationMode.INFORMATIONAL and r.passed + ) + + @property + def informational_failed(self) -> int: + """Number of informational validations that failed.""" + return sum( + 1 + for r in self.results + if r.mode == ValidationMode.INFORMATIONAL and not r.passed + ) + + @property + def all_required_passed(self) -> bool: + """Whether all required validations passed.""" + return self.required_failed == 0 + + @property + def is_empty(self) -> bool: + """Whether no validations were run.""" + return self.total == 0 + + def to_plan_metadata(self) -> dict[str, Any]: + """Convert to a dict for plan validation_summary field. + + Compatible with the plan model's ``validation_summary`` field. + """ + return { + "total": self.total, + "required_passed": self.required_passed, + "required_failed": self.required_failed, + "informational_passed": self.informational_passed, + "informational_failed": self.informational_failed, + "all_required_passed": self.all_required_passed, + "evaluated_at": self.evaluated_at.isoformat(), + "results": [ + { + "attachment_id": r.attachment_id, + "validation_name": r.validation_name, + "resource_id": r.resource_id, + "mode": r.mode.value, + "passed": r.passed, + "message": r.message, + "duration_ms": r.duration_ms, + "error": r.error, + "timed_out": r.timed_out, + } + for r in self.results + ], + } + + def format_cli_output(self) -> str: + """Format a human-readable summary for CLI display.""" + lines: list[str] = [] + status = "PASSED" if self.all_required_passed else "BLOCKED" + lines.append(f"Validation Gate: {status}") + lines.append( + f" Required: {self.required_passed} passed, {self.required_failed} failed" + ) + lines.append( + f" Informational: {self.informational_passed} passed, " + f"{self.informational_failed} failed" + ) + if self.required_failed > 0: + lines.append(" Failures:") + for r in self.results: + if r.mode == ValidationMode.REQUIRED and not r.passed: + msg = r.message or r.error or "no details" + lines.append(f" - {r.validation_name} on {r.resource_id}: {msg}") + return "\n".join(lines) + + +# --------------------------------------------------------------------------- +# Runner interface +# --------------------------------------------------------------------------- + + +class ValidationRunner(ABC): + """Interface for executing a single validation. + + Implementations handle the actual invocation of a validation tool + (e.g., subprocess, MCP call, in-process function). + """ + + @abstractmethod + def run_validation( + self, + attachment: ValidationAttachment, + context: dict[str, Any], + ) -> ApplyValidationResult: + """Execute a single validation. + + Args: + attachment: The validation attachment to execute. + context: Plan context (changeset summary, metadata, etc.). + + Returns: + The validation result. + """ + ... + + +class DefaultValidationRunner(ValidationRunner): + """Default validation runner using text matching. + + For each attachment, checks if the validation name appears in the + context values. This is a simple default for development and testing. + Production implementations should use the full tool execution pipeline. + """ + + def run_validation( + self, + attachment: ValidationAttachment, + context: dict[str, Any], + ) -> ApplyValidationResult: + """Execute a validation via text matching against context. + + Checks if the validation name (or any argument value) appears + in the context. This is a placeholder — real implementations + invoke the actual validation tool. + """ + start = time.monotonic() + try: + context_text = " ".join(f"{k}={v}" for k, v in context.items()).lower() + val_name = attachment.validation_name.lower() + + # Simple heuristic: check if validation name is referenced in context + passed = val_name in context_text + if not passed: + # Check arguments + for val in attachment.arguments.values(): + if str(val).lower() in context_text: + passed = True + break + + elapsed = int((time.monotonic() - start) * 1000) + return ApplyValidationResult( + attachment_id=attachment.attachment_id, + validation_name=attachment.validation_name, + resource_id=attachment.resource_id, + mode=attachment.mode, + passed=passed, + message="matched" if passed else "not matched", + duration_ms=elapsed, + ) + except Exception as exc: + elapsed = int((time.monotonic() - start) * 1000) + return ApplyValidationResult( + attachment_id=attachment.attachment_id, + validation_name=attachment.validation_name, + resource_id=attachment.resource_id, + mode=attachment.mode, + passed=False, + message="", + error=str(exc), + duration_ms=elapsed, + ) + + +# --------------------------------------------------------------------------- +# Apply validation gate +# --------------------------------------------------------------------------- + + +class ApplyValidationGate: + """Orchestrates validation execution and apply gating. + + Collects applicable validation attachments, runs them via the + provided runner, and produces an ``ApplyValidationSummary`` + that determines whether apply should proceed. + """ + + def __init__(self, runner: ValidationRunner) -> None: + """Initialise the gate with a validation runner. + + Args: + runner: The ``ValidationRunner`` to execute each validation. + """ + self._runner = runner + + def run( + self, + plan_id: str, + attachments: list[ValidationAttachment], + context: dict[str, Any], + ) -> ApplyValidationSummary: + """Run all attached validations and return gating result. + + Executes each validation attachment against the provided context + and returns an aggregated summary. + + Args: + plan_id: The plan ULID. + attachments: List of validation attachments to evaluate. + context: Plan context for evaluation. + + Returns: + An ``ApplyValidationSummary`` with per-validation results + and the overall gating decision. + """ + results: list[ApplyValidationResult] = [] + for attachment in attachments: + result = self._runner.run_validation(attachment, context) + results.append(result) + + return ApplyValidationSummary( + plan_id=plan_id, + results=results, + ) + + def should_block_apply( + self, + summary: ApplyValidationSummary, + ) -> bool: + """Determine if apply should be blocked. + + Args: + summary: The validation summary to check. + + Returns: + ``True`` if apply should be blocked (required failures exist). + """ + return not summary.all_required_passed + + def get_failure_reasons( + self, + summary: ApplyValidationSummary, + ) -> list[str]: + """Extract human-readable failure reasons from a summary. + + Args: + summary: The validation summary. + + Returns: + List of failure reason strings for required failures. + """ + reasons: list[str] = [] + for r in summary.results: + if r.mode == ValidationMode.REQUIRED and not r.passed: + if r.error: + reasons.append( + f"{r.validation_name} on {r.resource_id}: error: {r.error}" + ) + else: + reasons.append( + f"{r.validation_name} on {r.resource_id}: " + f"{r.message or 'failed'}" + ) + return reasons