"""Diff-aware classifier for ``local_ci_gate.sh`` quality-gate output. Cross-references the failing BDD scenarios reported by the gate runner against the PR's changed-files list, splitting failures into two buckets: - ``related_failures`` — failures in feature files whose paths match (or are colocated with) files the PR touches. Almost always real regressions caused by the PR. - ``unrelated_failures`` — failures in feature files the PR did not touch. Strongly correlated with environmental flakiness or CI-side intermittency — these are the failures Tier 0 historically bailed out on (e.g., PR #30 Run 1, 2026-05-12, where six CLI scenarios failed unrelated to the LangGraph fix and gpt-5-mini gave up rather than re-running them). Used by the dispatcher's pre-flight pre-fetch path (Heavy flaky pre-flight, plan §"#1 Heavy") to verify a failure is persistent before surfacing it to the worker. Also useful as a standalone helper the implementer worker can call mid-session. This module is pure-Python and has no I/O of its own — callers provide the gate output text and the list of changed files. Easy to unit-test. """ from __future__ import annotations import re from typing import Iterable # Pattern for failing-BDD-scenario lines in ``behave`` output: # # features/path/to/file.feature:42 Scenario: did the thing # # behave reports failures as "Failing scenarios:" header followed by # indented ``path:line`` lines. The trailing scenario name is # optional and may contain anything; we only capture the file path # and line number which are the load-bearing identifiers. _BEHAVE_FAILURE_RE = re.compile( r"^\s*(?Pfeatures/[^\s:]+\.feature):(?P\d+)", re.MULTILINE, ) # When a gate reports "FAILED" via the local_ci_gate.sh wrapper the # header line looks like ``CI / unit_tests* FAIL ...``. Used as a # coarse status indicator for gates the parser doesn't deep-parse. _GATE_STATUS_RE = re.compile( r"^(?Plint|typecheck|unit_tests|integration_tests|" r"e2e_tests|coverage)\s*:\s*(?PPASS|FAIL|SKIP)", re.MULTILINE | re.IGNORECASE, ) def parse_failing_scenarios(gate_output: str) -> list[dict[str, str]]: """Extract failing scenario paths + line numbers from gate output. Returns a list of ``{"path": str, "line": str}`` dicts in the order they appear in the output. Duplicates are de-duplicated by ``(path, line)`` because a behave summary tends to list each failure both inline and in the "Failing scenarios:" footer. """ seen: set[tuple[str, str]] = set() out: list[dict[str, str]] = [] for m in _BEHAVE_FAILURE_RE.finditer(gate_output or ""): key = (m.group("path"), m.group("line")) if key in seen: continue seen.add(key) out.append({"path": m.group("path"), "line": m.group("line")}) return out def parse_gate_statuses(gate_output: str) -> dict[str, str]: """Map gate name → status (PASS/FAIL/SKIP) parsed from output. Coarse-grained — used by the dispatcher to surface a high-level gate roll-up in the worker prompt without exposing the full raw output. The agent reads this dict and can decide whether to drill into specific failures. """ out: dict[str, str] = {} for m in _GATE_STATUS_RE.finditer(gate_output or ""): out[m.group("gate").lower()] = m.group("status").upper() return out def _normalise_path(p: str) -> str: """Strip leading ``./`` and trailing whitespace from a path.""" return (p or "").strip().removeprefix("./") def _feature_likely_related( feature_path: str, changed_files: Iterable[str] ) -> bool: """Heuristic: a feature is likely related to a PR's changes if EITHER the feature file itself was changed OR the feature's name-stem appears in a changed source-file path. Examples: - feature ``features/auto_debug_cli_coverage.feature`` is related to a PR touching ``src/cleveragents/auto_debug/cli.py`` (stem ``auto_debug`` appears in both). - feature ``features/plan_cli_commands_r2.feature`` is unrelated to a PR touching ``langgraph/graph.py`` (no overlap). This is intentionally simple — false positives (calling something related when it isn't) are cheap (we just don't auto-suppress that failure). False negatives (calling something unrelated when it IS) are also cheap because the worker still sees the failure list and can decide. """ feature_path_norm = _normalise_path(feature_path) changed_norm = {_normalise_path(c) for c in changed_files} # Direct hit — feature file itself changed. if feature_path_norm in changed_norm: return True # Extract the stem of the feature name (strip extension + path). # ``features/auto_debug_cli_coverage.feature`` → ``auto_debug_cli_coverage``. stem = feature_path_norm.split("/")[-1] if stem.endswith(".feature"): stem = stem[: -len(".feature")] if not stem: return False # Split the stem into prefix tokens — the most-specific token # first. We compare just the first 2-3 tokens because feature # filenames are often verbose ("plan_cli_commands_r2_boost") and # matching the WHOLE stem against a path would miss legitimate # matches like ``plan_cli/commands.py``. tokens = stem.split("_") # Take the first 2 tokens as the "salient prefix" — the bit # that's most likely to map to a source-tree path component. prefix_candidates = [] if tokens: prefix_candidates.append(tokens[0]) if len(tokens) >= 2: prefix_candidates.append("_".join(tokens[:2])) for prefix in prefix_candidates: if not prefix: continue for changed in changed_norm: # Match against PATH SEGMENTS — avoid matching substrings # of unrelated names (``plan`` matches ``plans/`` but # not ``misplanned``). segments = changed.replace(".", "/").split("/") if prefix in segments: return True return False def classify_failures( gate_output: str, changed_files: Iterable[str] ) -> dict[str, object]: """Top-level classifier: parse + relate. Returns a structured dict the dispatcher embeds in the worker prompt: .. code-block:: python { "gate_statuses": {"lint": "PASS", "unit_tests": "FAIL", ...}, "failures_total": 6, "failures_related_to_diff": 0, "failures_unrelated_to_diff": 6, "related": [{"path": ..., "line": ...}, ...], "unrelated": [{"path": ..., "line": ...}, ...], } The agent's instruction (in implementation-worker / task-implementor prompts): if ``failures_related_to_diff == 0`` and the gate pre-flight already retried persistent failures, the run is PROBABLY flaky — focus on compliance gaps / fill-the-blanks rather than diagnosing the unrelated failures. """ statuses = parse_gate_statuses(gate_output) failures = parse_failing_scenarios(gate_output) changed = list(changed_files) related: list[dict[str, str]] = [] unrelated: list[dict[str, str]] = [] for f in failures: if _feature_likely_related(f["path"], changed): related.append(f) else: unrelated.append(f) return { "gate_statuses": statuses, "failures_total": len(failures), "failures_related_to_diff": len(related), "failures_unrelated_to_diff": len(unrelated), "related": related, "unrelated": unrelated, } def render_prompt_stanza(classification: dict[str, object]) -> str: """Render the classification result as a markdown stanza for the worker prompt. Kept here (next to the classifier) so the rendered shape stays lockstep with the data shape. Returns the empty string when there are zero failures to surface — the dispatcher embeds the stanza unconditionally when the classification ran, but the agent reads "empty stanza = all gates passed in pre-flight" without a special-case. """ total = classification.get("failures_total", 0) related = classification.get("related") or [] unrelated = classification.get("unrelated") or [] statuses = classification.get("gate_statuses") or {} if not total and not statuses: return "" lines = ["## Pre-flight quality gates (verified persistent)\n"] if statuses: lines.append("Gate summary:") for gate, status in sorted(statuses.items()): lines.append(f"- `{gate}`: **{status}**") lines.append("") if total == 0: lines.append( "_No failures persisted after the pre-flight retry — " "you do NOT need to re-run `local_ci_gate.sh --fast` " "yourself unless you change files._" ) return "\n".join(lines) lines.append( f"Failures total: **{total}** " f"(related to your diff: {len(related)}, " f"unrelated: {len(unrelated)})." ) lines.append("") if related: lines.append("**Related to your diff** — investigate these:") for f in related: lines.append(f"- `{f['path']}:{f['line']}`") lines.append("") if unrelated: lines.append( "**Unrelated to your diff** — these were retried by " "the dispatcher's pre-flight and persisted, but they " "exercise code your PR did not touch. Likely " "environmental / pre-existing. Do NOT bail on the " "cycle for these — focus on the related failures " "above (if any) and on compliance gaps (next section). " "If you have time, you may re-run these locally to " "confirm reproducibility before exiting." ) for f in unrelated: lines.append(f"- `{f['path']}:{f['line']}`") return "\n".join(lines)