diff --git a/CHANGELOG.md b/CHANGELOG.md index 5e1879cb8..836462f69 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -106,6 +106,22 @@ called during init), and two regression tests verifying `bootstrap_builtin_types()` seeds correct data and `agents resource add git-checkout` succeeds. Includes Robot Framework regression tests. (#553) +- Added UKO Indexer for real-time index synchronization. `UKOIndexer` orchestrates + analysis of resources into UKO triples via pluggable `AnalyzerRegistry` and + simultaneously indexes into text, vector, and graph backends. Provenance metadata + (`ProvenanceMetadata`, `ProvenancedTriple`) is attached to every triple, tracking + source resource, file path, and temporal validity. Write-side index backend protocols + (`TextIndexBackend`, `VectorIndexBackend`, `GraphIndexBackend`) are distinct from the + existing read-side query protocols. Graceful degradation: if text or vector backends + are `None`, the corresponding indexing step is silently skipped. Index lifecycle: + `index_resource` (add), `remove_resource` (cleanup), `reindex_resource` (change). + `ContentReader` protocol decouples the indexer from filesystem I/O. + `IndexLifecycleHook` provides callbacks for indexing events. In-memory stub + implementations for all three backends. `ResourceFileWatcher` monitors filesystem + paths via watchdog and triggers re-indexing callbacks on file changes with + configurable debouncing and optional `RESOURCE_MODIFIED` EventBus emission. + Includes 166 Behave BDD scenarios, 9 Robot Framework integration tests, ASV + benchmarks, and reference documentation. (#578) - Added general-purpose domain event system under `cleveragents.infrastructure.events`. `EventType` StrEnum defines 38 typed event identifiers across 9 domains (plan lifecycle, decision, invariant, actor, diff --git a/benchmarks/uko_indexer_bench.py b/benchmarks/uko_indexer_bench.py new file mode 100644 index 000000000..3e428109b --- /dev/null +++ b/benchmarks/uko_indexer_bench.py @@ -0,0 +1,224 @@ +"""ASV benchmarks for UKO Indexer Real-time Index Synchronization. + +Measures the performance of: +- UKOIndexer.index_resource() pipeline (analyze → provenance → graph → text → vector) +- UKOIndexer.remove_resource() cleanup +- UKOIndexer.reindex_resource() lifecycle +- In-memory index backend operations +- Graceful degradation (no optional backends) +""" + +from __future__ import annotations + +import importlib +import sys +from pathlib import Path + +# Ensure the local *source* tree is importable even when ASV has an +# older build of the package installed. +_SRC = str(Path(__file__).resolve().parents[1] / "src") +if _SRC not in sys.path: + sys.path.insert(0, _SRC) + +# Force-reload so ASV picks up the source tree version. +import cleveragents # noqa: E402 + +importlib.reload(cleveragents) + +from cleveragents.application.services.uko_indexer import ( # noqa: E402 + UKOIndexer, +) +from cleveragents.domain.models.acms.analyzers import ( # noqa: E402 + AnalyzerRegistry, +) +from cleveragents.domain.models.acms.index_stubs import ( # noqa: E402 + InMemoryGraphIndexBackend, + InMemoryTextIndexBackend, + InMemoryVectorIndexBackend, +) +from cleveragents.domain.models.acms.python_analyzer import ( # noqa: E402 + PythonAnalyzer, +) +from cleveragents.domain.models.core.resource import Resource # noqa: E402 + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- + +ULID_1 = "01HQ8ZDRX50000000000000001" +ULID_2 = "01HQ8ZDRX50000000000000002" +PROJECT = "local/bench" + +_SMALL_PYTHON = 'import os\n\ndef hello():\n """Say hello."""\n pass\n' + +_MEDIUM_PYTHON = ( + '"""Module docstring."""\n\nimport os\nfrom pathlib import Path\n\n' + + "\n\n".join( + f'class Cls{i}:\n """Class {i}."""\n' + f" def method_{i}(self):\n" + f' """Method {i}."""\n pass\n' + for i in range(20) + ) +) + + +class _InMemoryContentReader: + """Content reader for benchmarks.""" + + def __init__(self) -> None: + self._content: dict[str, str] = {} + + def set_content(self, resource_id: str, content: str) -> None: + self._content[resource_id] = content + + def read_content(self, resource: Resource) -> str: + return self._content[resource.resource_id] + + +def _make_resource(resource_id: str) -> Resource: + return Resource( + resource_id=resource_id, + resource_type_name="git-checkout", + classification="physical", + location="src/bench.py", + ) + + +# --------------------------------------------------------------------------- +# Index pipeline benchmarks +# --------------------------------------------------------------------------- + + +class IndexResourceSuite: + """Benchmark UKOIndexer.index_resource() pipeline.""" + + timeout = 120 + + def setup(self) -> None: + self.resource = _make_resource(ULID_1) + + def time_index_small_python(self) -> None: + # Re-create indexer each time to avoid accumulation + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer()) + reader = _InMemoryContentReader() + reader.set_content(ULID_1, _SMALL_PYTHON) + indexer = UKOIndexer( + analyzer_registry=registry, + graph_backend=InMemoryGraphIndexBackend(), + text_backend=InMemoryTextIndexBackend(), + vector_backend=InMemoryVectorIndexBackend(), + content_reader=reader, + ) + indexer.index_resource(self.resource, project=PROJECT) + + def time_index_medium_python(self) -> None: + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer()) + reader = _InMemoryContentReader() + reader.set_content(ULID_1, _MEDIUM_PYTHON) + indexer = UKOIndexer( + analyzer_registry=registry, + graph_backend=InMemoryGraphIndexBackend(), + text_backend=InMemoryTextIndexBackend(), + vector_backend=InMemoryVectorIndexBackend(), + content_reader=reader, + ) + indexer.index_resource(self.resource, project=PROJECT) + + +class IndexGracefulDegradationSuite: + """Benchmark indexing without optional backends.""" + + timeout = 120 + + def setup(self) -> None: + self.resource = _make_resource(ULID_1) + + def time_index_graph_only(self) -> None: + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer()) + reader = _InMemoryContentReader() + reader.set_content(ULID_1, _SMALL_PYTHON) + indexer = UKOIndexer( + analyzer_registry=registry, + graph_backend=InMemoryGraphIndexBackend(), + content_reader=reader, + ) + indexer.index_resource(self.resource, project=PROJECT) + + +# --------------------------------------------------------------------------- +# Lifecycle benchmarks +# --------------------------------------------------------------------------- + + +class LifecycleSuite: + """Benchmark index lifecycle (add/remove/reindex).""" + + timeout = 120 + + def setup(self) -> None: + self.resource = _make_resource(ULID_1) + + def time_remove_resource(self) -> None: + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer()) + reader = _InMemoryContentReader() + reader.set_content(ULID_1, _SMALL_PYTHON) + indexer = UKOIndexer( + analyzer_registry=registry, + graph_backend=InMemoryGraphIndexBackend(), + text_backend=InMemoryTextIndexBackend(), + vector_backend=InMemoryVectorIndexBackend(), + content_reader=reader, + ) + indexer.index_resource(self.resource, project=PROJECT) + indexer.remove_resource(ULID_1, project=PROJECT) + + def time_reindex_resource(self) -> None: + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer()) + reader = _InMemoryContentReader() + reader.set_content(ULID_1, _SMALL_PYTHON) + indexer = UKOIndexer( + analyzer_registry=registry, + graph_backend=InMemoryGraphIndexBackend(), + text_backend=InMemoryTextIndexBackend(), + vector_backend=InMemoryVectorIndexBackend(), + content_reader=reader, + ) + indexer.index_resource(self.resource, project=PROJECT) + reader.set_content(ULID_1, _MEDIUM_PYTHON) + indexer.reindex_resource(self.resource, project=PROJECT) + + +# --------------------------------------------------------------------------- +# Backend operation benchmarks +# --------------------------------------------------------------------------- + + +class BackendOperationsSuite: + """Benchmark raw in-memory backend operations.""" + + timeout = 60 + + def time_text_index_and_search(self) -> None: + be = InMemoryTextIndexBackend() + be.index_document(PROJECT, "doc1", "hello world", {"k": "v"}) + be.search(PROJECT, "hello", limit=10) + + def time_vector_index_and_search(self) -> None: + be = InMemoryVectorIndexBackend() + be.index_embedding(PROJECT, "emb1", [1.0, 2.0, 3.0], {"k": "v"}) + be.search_similar(PROJECT, [1.0, 2.0, 3.0], limit=10) + + def time_graph_add_and_query(self) -> None: + be = InMemoryGraphIndexBackend() + be.add_triple(PROJECT, "s1", "p1", "o1") + be.query(PROJECT, "SELECT *") + + def time_graph_add_and_remove(self) -> None: + be = InMemoryGraphIndexBackend() + be.add_triple(PROJECT, "s1", "p1", "o1") + be.remove_triples(PROJECT, subject="s1", predicate=None, obj=None) diff --git a/docs/reference/uko_indexer.md b/docs/reference/uko_indexer.md new file mode 100644 index 000000000..fbae52ac5 --- /dev/null +++ b/docs/reference/uko_indexer.md @@ -0,0 +1,230 @@ +# UKO Indexer — Real-time Index Synchronization + +The UKO Indexer orchestrates the analysis of resources into UKO triples and +simultaneously indexes content into text, vector, and graph backends. It +implements the index lifecycle (add/change/remove/maintenance) with graceful +degradation when optional backends are unavailable. + +Based on `docs/specification.md` > ACMS > Real-time Index +Synchronization and Custom Index Backends. + +## Architecture + +``` +Resource ──► AnalyzerRegistry.get_for_resource() + │ + ▼ + ContentReader.read_content() + │ + ▼ + Analyzer.analyze(content, resource_uri) + │ + ▼ + attach_provenance(triples, resource, timestamp) + │ + ├──► GraphIndexBackend.add_triple() (required) + ├──► TextIndexBackend.index_document() (optional — graceful degradation) + └──► VectorIndexBackend.index_embedding() (optional — graceful degradation) +``` + +## UKOIndexer + +Main service class that produces UKO triples from resources and indexes them +into all available backends. + +### Constructor + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `analyzer_registry` | `AnalyzerRegistry` | Yes | Registry of domain analyzers | +| `graph_backend` | `GraphIndexBackend` | Yes | Backend for UKO triple storage | +| `text_backend` | `TextIndexBackend \| None` | No | Backend for full-text indexing | +| `vector_backend` | `VectorIndexBackend \| None` | No | Backend for embedding indexing | +| `content_reader` | `ContentReader \| None` | No | Reader for resource content (default: `LocationContentReader`) | +| `lifecycle_hook` | `IndexLifecycleHook \| None` | No | Callback for lifecycle events (default: `DefaultLifecycleHook`) | +| `max_triples` | `int` | No | Maximum triples per resource (default: `50_000`). Must be positive. | + +### Methods + +| Method | Signature | Returns | Description | +|--------|-----------|---------|-------------| +| `index_resource` | `(resource: Resource, *, project: str)` | `IndexResult` | Full indexing pipeline | +| `remove_resource` | `(resource_id: str, *, project: str)` | `None` | Remove from all indices | +| `reindex_resource` | `(resource: Resource, *, project: str)` | `IndexResult` | Remove + re-index | + +### Properties + +| Property | Type | Description | +|----------|------|-------------| +| `analyzer_registry` | `AnalyzerRegistry` | The analyzer registry | +| `indexed_resource_count` | `int` | Number of currently indexed resources | +| `has_text_backend` | `bool` | Whether text backend is available | +| `has_vector_backend` | `bool` | Whether vector backend is available | + +### Graceful Degradation + +If `text_backend` or `vector_backend` is `None`, the corresponding indexing +step is silently skipped. The `graph_backend` is always required. Error +counts from individual backend failures are recorded in `IndexResult.errors` +without aborting the entire indexing operation. + +## Index Backends (Write-side Protocols) + +These are **write-side** protocols for indexing. They are distinct from the +existing read-side query protocols (`TextBackend`, `VectorBackend`, +`GraphBackend`) in `backends.py`. + +### TextIndexBackend + +| Method | Signature | Returns | +|--------|-----------|---------| +| `index_document` | `(project: str, doc_id: str, content: str, metadata: dict[str, str])` | `IndexedDocument` | +| `search` | `(project: str, query: str, *, limit: int = 20)` | `list[SearchResult]` | +| `remove_document` | `(project: str, doc_id: str)` | `None` | +| `rebuild_index` | `(project: str)` | `None` | + +### VectorIndexBackend + +| Method | Signature | Returns | +|--------|-----------|---------| +| `index_embedding` | `(project: str, doc_id: str, embedding: list[float], metadata: dict[str, str])` | `None` | +| `search_similar` | `(project: str, query_embedding: list[float], limit: int = 20, min_relevance: float = 0.0)` | `list[SearchResult]` | +| `remove_embedding` | `(project: str, doc_id: str)` | `None` | + +### GraphIndexBackend + +| Method | Signature | Returns | +|--------|-----------|---------| +| `add_triple` | `(project: str, subject: str, predicate: str, obj: str)` | `None` | +| `query` | `(project: str, sparql: str)` | `list[dict[str, str]]` | +| `remove_triples` | `(project: str, subject: str \| None, predicate: str \| None, obj: str \| None)` | `None` | + +## Result Types + +### IndexedDocument + +| Field | Type | Description | +|-------|------|-------------| +| `project` | `str` | Namespaced project name | +| `doc_id` | `str` | Document identifier | +| `char_count` | `int` | Character count (non-negative, default 0) | + +### SearchResult + +| Field | Type | Description | +|-------|------|-------------| +| `doc_id` | `str` | Document identifier (non-empty) | +| `content` | `str` | Matched text snippet | +| `score` | `float` | Relevance score in `[0.0, 1.0]` | +| `metadata` | `dict[str, str]` | Backend-specific metadata | + +## Provenance + +### ProvenanceMetadata + +Tracks the origin and validity of indexed triples. + +| Field | Type | Description | +|-------|------|-------------| +| `source_resource` | `str` | Resource ULID that produced the triple | +| `source_path` | `str` | File path of the source resource | +| `source_range` | `str` | Line range (e.g. `"10-25"`), empty if whole file | +| `valid_from` | `datetime` | When the triple was indexed (defaults to `now(UTC)`) | +| `is_current` | `bool` | Whether the triple is still current (default: `True`) | + +### ProvenancedTriple + +Wrapper combining a `UKOTriple` with its `ProvenanceMetadata`. + +| Field | Type | Description | +|-------|------|-------------| +| `triple` | `UKOTriple` | The underlying UKO triple | +| `provenance` | `ProvenanceMetadata` | Origin and validity metadata | + +### IndexResult + +Summary returned by `index_resource()` and `reindex_resource()`. + +| Field | Type | Description | +|-------|------|-------------| +| `resource_id` | `str` | ULID of the indexed resource | +| `triple_count` | `int` | Number of triples produced | +| `text_docs_indexed` | `int` | Text documents indexed (0 or 1) | +| `embeddings_indexed` | `int` | Embeddings indexed (0 or 1) | +| `analyzer_domain` | `str` | Domain of the analyzer used | +| `errors` | `tuple[str, ...]` | Non-fatal error messages | + +## Protocols + +### ContentReader + +Protocol for reading resource content. Decouples the indexer from the +filesystem. + +| Method | Signature | Returns | +|--------|-----------|---------| +| `read_content` | `(resource: Resource)` | `str` | + +**Default implementation**: `LocationContentReader` — reads from +`resource.location` via the local filesystem. + +### IndexLifecycleHook + +Callback protocol for index lifecycle events. + +| Method | Signature | Returns | +|--------|-----------|---------| +| `on_indexed` | `(result: IndexResult)` | `None` | +| `on_removed` | `(resource_id: str, project: str)` | `None` | +| `on_error` | `(resource_id: str, error: str)` | `None` | + +**Default implementation**: `DefaultLifecycleHook` — logs events via +structlog. + +## In-Memory Stubs + +For testing, three in-memory stub implementations are provided: + +- `InMemoryTextIndexBackend` — dict-based full-text storage with substring search +- `InMemoryVectorIndexBackend` — dict-based embedding storage (no real similarity) +- `InMemoryGraphIndexBackend` — list-based triple storage with pattern matching + +## ResourceFileWatcher + +Monitors filesystem paths via `watchdog` and triggers re-indexing callbacks +on file changes with configurable debouncing. + +### Constructor + +| Parameter | Type | Required | Description | +|-----------|------|----------|-------------| +| `indexer` | `UKOIndexer` | Yes | Indexer for re-index on change | +| `project` | `str` | Yes | Project namespace | +| `debounce_seconds` | `float` | No | Debounce delay (default: `0.5`) | +| `on_change` | `Callable[[str, str, FileChangeType], None] \| None` | No | Optional change callback | +| `event_bus` | `EventBus \| None` | No | Optional event bus for `RESOURCE_MODIFIED` emission | + +### FileChangeType + +`StrEnum` with values: `CREATED`, `MODIFIED`, `DELETED`, `MOVED`. + +### Methods + +| Method | Signature | Returns | Description | +|--------|-----------|---------|-------------| +| `watch` | `(resource_id: str, path: Path)` | `None` | Start watching a file path for a resource | +| `unwatch` | `(resource_id: str)` | `None` | Stop watching a resource | +| `shutdown` | `()` | `None` | Stop all observers and clean up | + +## Known Limitations + +- **Placeholder vector embedding**: The vector backend receives a + constant `[1.0]` placeholder instead of a real embedding. Real + embedding model integration is tracked as a follow-up to issue #578. + +## Spec References + +- Real-time Index Synchronization: `specification.md` > ACMS > Real-time Index Synchronization +- Custom Index Backends: `specification.md` > ACMS > Custom Index Backends +- Domain Analyzers: `specification.md` > ACMS > Domain Analyzers +- Config keys: `specification.md` > Configuration Service > Index Backend Keys diff --git a/features/mocks/uko_indexer_mocks.py b/features/mocks/uko_indexer_mocks.py new file mode 100644 index 000000000..1717f87be --- /dev/null +++ b/features/mocks/uko_indexer_mocks.py @@ -0,0 +1,292 @@ +"""Test doubles for UKO Indexer BDD and Robot tests. + +Provides in-memory content readers, failing backends, tracking +lifecycle hooks, and event bus stubs used by +``features/steps/uko_indexer_steps.py``, +``robot/helper_uko_indexer.py``, and ``benchmarks/uko_indexer_bench.py``. +""" + +from __future__ import annotations + +import threading +from collections.abc import Callable +from typing import Any + +from cleveragents.domain.models.acms.index_backends import ( + IndexedDocument, + SearchResult, +) +from cleveragents.domain.models.acms.provenance import IndexResult +from cleveragents.domain.models.core.resource import Resource +from cleveragents.infrastructure.events.models import DomainEvent +from cleveragents.infrastructure.events.types import EventType + +# --------------------------------------------------------------------------- +# Content readers +# --------------------------------------------------------------------------- + + +class InMemoryContentReader: + """Content reader that returns pre-configured content strings.""" + + def __init__(self) -> None: + self._content: dict[str, str] = {} + + def set_content(self, resource_id: str, content: str) -> None: + self._content[resource_id] = content + + def read_content(self, resource: Resource) -> str: + if resource.resource_id in self._content: + return self._content[resource.resource_id] + raise OSError(f"No content for {resource.resource_id}") + + +class FailingContentReader: + """Content reader that always raises OSError.""" + + def read_content(self, resource: Resource) -> str: + raise OSError("Simulated read failure") + + +# --------------------------------------------------------------------------- +# Lifecycle hooks +# --------------------------------------------------------------------------- + + +class TrackingLifecycleHook: + """Lifecycle hook that records events for assertion.""" + + def __init__(self) -> None: + self.indexed_events: list[IndexResult] = [] + self.removed_events: list[tuple[str, str]] = [] + self.error_events: list[tuple[str, str]] = [] + + def on_indexed(self, result: IndexResult) -> None: + self.indexed_events.append(result) + + def on_removed(self, resource_id: str, project: str) -> None: + self.removed_events.append((resource_id, project)) + + def on_error(self, resource_id: str, error: str) -> None: + self.error_events.append((resource_id, error)) + + +# --------------------------------------------------------------------------- +# Failing backends (test doubles) +# --------------------------------------------------------------------------- + + +class FailingGraphBackend: + """Graph backend that raises on add_triple.""" + + def add_triple( + self, + project: str, + subject: str, + predicate: str, + obj: str, + ) -> None: + raise RuntimeError("Simulated graph failure") + + def remove_triples( + self, + project: str, + subject: str | None, + predicate: str | None, + obj: str | None, + ) -> None: + pass + + def query(self, project: str, sparql: str) -> list[dict[str, str]]: + return [] + + def triple_count(self, project: str | None = None) -> int: + return 0 + + +class FailingTextBackend: + """Text backend that raises on index_document.""" + + def index_document( + self, + project: str, + doc_id: str, + content: str, + metadata: dict[str, str], + ) -> IndexedDocument: + raise RuntimeError("Simulated text failure") + + def search( + self, + project: str, + query: str, + limit: int = 20, + ) -> list[SearchResult]: + return [] + + def remove_document(self, project: str, doc_id: str) -> None: + pass + + def rebuild_index(self, project: str) -> None: + pass + + @property + def document_count(self) -> int: + return 0 + + +class FailingVectorBackend: + """Vector backend that raises on index_embedding.""" + + def index_embedding( + self, + project: str, + doc_id: str, + embedding: list[float], + metadata: dict[str, str], + ) -> None: + raise RuntimeError("Simulated vector failure") + + def search_similar( + self, + project: str, + query_embedding: list[float], + limit: int = 20, + min_relevance: float = 0.0, + ) -> list[SearchResult]: + return [] + + def remove_embedding(self, project: str, doc_id: str) -> None: + pass + + @property + def embedding_count(self) -> int: + return 0 + + +# --------------------------------------------------------------------------- +# Event bus stubs +# --------------------------------------------------------------------------- + + +class TrackingEventBus: + """Minimal EventBus stub that records emitted events.""" + + def __init__(self) -> None: + self.events: list[DomainEvent] = [] + self._event = threading.Event() + + def emit(self, event: DomainEvent) -> None: + self.events.append(event) + self._event.set() + + def subscribe( + self, + event_type: EventType, + handler: Callable[..., Any], + ) -> None: + _ = event_type, handler # Not needed for tests + + def wait(self, timeout: float = 2.0) -> bool: + return self._event.wait(timeout=timeout) + + +# --------------------------------------------------------------------------- +# Backends that fail on removal (not on add) +# --------------------------------------------------------------------------- + + +class RemovalFailingGraphBackend: + """Graph backend that works for add but fails on remove_triples.""" + + def __init__(self) -> None: + self._triples: list[tuple[str, str, str, str]] = [] + + def add_triple(self, project: str, subject: str, predicate: str, obj: str) -> None: + self._triples.append((project, subject, predicate, obj)) + + def remove_triples( + self, project: str, subject: str | None, predicate: str | None, obj: str | None + ) -> None: + raise RuntimeError("Simulated graph removal failure") + + def query(self, project: str, sparql: str) -> list[dict[str, str]]: + return [] + + def triple_count(self, project: str | None = None) -> int: + return len(self._triples) + + +class RemovalFailingTextBackend: + """Text backend that works for add but fails on remove_document.""" + + def index_document( + self, project: str, doc_id: str, content: str, metadata: dict[str, str] + ) -> IndexedDocument: + return IndexedDocument(project=project, doc_id=doc_id, char_count=len(content)) + + def search(self, project: str, query: str, limit: int = 20) -> list[SearchResult]: + return [] + + def remove_document(self, project: str, doc_id: str) -> None: + raise RuntimeError("Simulated text removal failure") + + def rebuild_index(self, project: str) -> None: + pass + + @property + def document_count(self) -> int: + return 0 + + +class RemovalFailingVectorBackend: + """Vector backend that works for add but fails on remove_embedding.""" + + def index_embedding( + self, + project: str, + doc_id: str, + embedding: list[float], + metadata: dict[str, str], + ) -> None: + pass + + def search_similar( + self, + project: str, + query_embedding: list[float], + limit: int = 20, + min_relevance: float = 0.0, + ) -> list[SearchResult]: + return [] + + def remove_embedding(self, project: str, doc_id: str) -> None: + raise RuntimeError("Simulated vector removal failure") + + @property + def embedding_count(self) -> int: + return 0 + + +class SelectiveFailingGraphBackend: + """Graph backend that fails only on specific predicates.""" + + def __init__(self, fail_on: set[str] | None = None) -> None: + self._triples: list[tuple[str, str, str, str]] = [] + self._fail_on = fail_on or set() + + def add_triple(self, project: str, subject: str, predicate: str, obj: str) -> None: + if predicate in self._fail_on: + raise RuntimeError(f"Simulated failure on {predicate}") + self._triples.append((project, subject, predicate, obj)) + + def remove_triples( + self, project: str, subject: str | None, predicate: str | None, obj: str | None + ) -> None: + pass + + def query(self, project: str, sparql: str) -> list[dict[str, str]]: + return [] + + def triple_count(self, project: str | None = None) -> int: + return len(self._triples) diff --git a/features/steps/uko_indexer_backend_steps.py b/features/steps/uko_indexer_backend_steps.py new file mode 100644 index 000000000..67a61944d --- /dev/null +++ b/features/steps/uko_indexer_backend_steps.py @@ -0,0 +1,495 @@ +"""Text/Vector/Graph backend steps + Provenance + IndexResult steps.""" + +from __future__ import annotations + +from typing import Any + +from behave import then, when # type: ignore[import-untyped] +from pydantic import ValidationError + +from cleveragents.domain.models.acms.analyzers import UKOTriple +from cleveragents.domain.models.acms.index_backends import ( + IndexedDocument, + SearchResult, +) +from cleveragents.domain.models.acms.provenance import ( + IndexResult, + ProvenancedTriple, + ProvenanceMetadata, +) +from features.steps.uko_indexer_common import ULID_1 + +__all__: list[str] = [] + + +@when( + 'idx I index a text document "{doc_id}" with content "{content}" in project "{project}"' +) +def step_index_text_doc(context: Any, doc_id: str, content: str, project: str) -> None: + context.idx_text_backend.index_document(project, doc_id, content, {}) + + +@then('idx searching "{query}" in project "{project}" should return {count:d} result') +def step_search_text_singular( + context: Any, query: str, project: str, count: int +) -> None: + results = context.idx_text_backend.search(project, query) + assert len(results) == count, f"Expected {count} result, got {len(results)}" + context.idx_search_results = results + + +@then('idx searching "{query}" in project "{project}" should return {count:d} results') +def step_search_text_plural(context: Any, query: str, project: str, count: int) -> None: + results = context.idx_text_backend.search(project, query) + assert len(results) == count, f"Expected {count} results, got {len(results)}" + context.idx_search_results = results + + +@then('idx the search result doc_id should be "{doc_id}"') +def step_check_search_doc_id(context: Any, doc_id: str) -> None: + assert context.idx_search_results[0].doc_id == doc_id + + +@when('idx I remove text document "{doc_id}" from project "{project}"') +def step_remove_text_doc(context: Any, doc_id: str, project: str) -> None: + context.idx_text_backend.remove_document(project, doc_id) + + +@when('idx I rebuild text index for project "{project}"') +def step_rebuild_text_index(context: Any, project: str) -> None: + context.idx_text_backend.rebuild_index(project) + + +@then("idx indexing a text document with empty project should raise ValueError") +def step_text_empty_project(context: Any) -> None: + try: + context.idx_text_backend.index_document("", "doc", "content", {}) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx indexing a text document with empty content should raise ValueError") +def step_text_empty_content(context: Any) -> None: + try: + context.idx_text_backend.index_document("proj", "doc", "", {}) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx searching with empty query should raise ValueError") +def step_text_empty_query(context: Any) -> None: + try: + context.idx_text_backend.search("proj", "") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx searching with limit 0 should raise ValueError") +def step_text_zero_limit(context: Any) -> None: + try: + context.idx_text_backend.search("proj", "q", limit=0) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@when( + 'idx I index an embedding "{doc_id}" with vector "{vector_str}" in project "{project}"' +) +def step_index_embedding( + context: Any, doc_id: str, vector_str: str, project: str +) -> None: + embedding = [float(v) for v in vector_str.split(",")] + context.idx_vector_backend.index_embedding(project, doc_id, embedding, {}) + + +@then('idx searching similar in project "{project}" should return {count:d} result') +def step_search_similar_singular(context: Any, project: str, count: int) -> None: + results = context.idx_vector_backend.search_similar(project, [1.0]) + assert len(results) == count, f"Expected {count}, got {len(results)}" + + +@then('idx searching similar in project "{project}" should return {count:d} results') +def step_search_similar_plural(context: Any, project: str, count: int) -> None: + results = context.idx_vector_backend.search_similar(project, [1.0]) + assert len(results) == count, f"Expected {count}, got {len(results)}" + + +@then( + "idx searching similar in project" + ' "{project}" with min_relevance {threshold} should return {count:d} results' +) +def step_search_similar_min_relevance( + context: Any, + project: str, + threshold: str, + count: int, +) -> None: + results = context.idx_vector_backend.search_similar( + project, + [1.0], + min_relevance=float(threshold), + ) + assert len(results) == count, f"Expected {count}, got {len(results)}" + + +@then("idx searching similar with min_relevance {threshold} should raise ValueError") +def step_search_similar_invalid_min_relevance( + context: Any, + threshold: str, +) -> None: + try: + context.idx_vector_backend.search_similar( + "local/app", + [1.0], + min_relevance=float(threshold), + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@when('idx I remove embedding "{doc_id}" from project "{project}"') +def step_remove_embedding(context: Any, doc_id: str, project: str) -> None: + context.idx_vector_backend.remove_embedding(project, doc_id) + + +@then("idx indexing an embedding with empty vector should raise ValueError") +def step_vector_empty_embedding(context: Any) -> None: + try: + context.idx_vector_backend.index_embedding("proj", "doc", [], {}) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@when('idx I add a triple "{subj}" "{pred}" "{obj}" in project "{project}"') +def step_add_triple(context: Any, subj: str, pred: str, obj: str, project: str) -> None: + context.idx_graph_backend.add_triple(project, subj, pred, obj) + + +@then('idx querying project "{project}" should return {count:d} binding') +def step_query_graph_singular(context: Any, project: str, count: int) -> None: + results = context.idx_graph_backend.query(project, "SELECT * WHERE { ?s ?p ?o }") + assert len(results) == count, f"Expected {count}, got {len(results)}" + context.idx_graph_results = results + + +@then('idx querying project "{project}" should return {count:d} bindings') +def step_query_graph_plural(context: Any, project: str, count: int) -> None: + results = context.idx_graph_backend.query(project, "SELECT * WHERE { ?s ?p ?o }") + assert len(results) == count, f"Expected {count}, got {len(results)}" + context.idx_graph_results = results + + +@then('idx the binding should have subject "{subject}"') +def step_check_binding_subject(context: Any, subject: str) -> None: + assert context.idx_graph_results[0]["s"] == subject + + +@when('idx I remove triples with subject "{subject}" from project "{project}"') +def step_remove_triples(context: Any, subject: str, project: str) -> None: + context.idx_graph_backend.remove_triples( + project, subject=subject, predicate=None, obj=None + ) + + +@then("idx removing triples with all-None pattern should raise ValueError") +def step_graph_all_none(context: Any) -> None: + try: + context.idx_graph_backend.remove_triples("proj", None, None, None) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx adding a triple with empty project should raise ValueError") +def step_graph_empty_project(context: Any) -> None: + try: + context.idx_graph_backend.add_triple("", "s", "p", "o") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@when('idx I create provenance for resource "{resource_id}"') +def step_create_provenance(context: Any, resource_id: str) -> None: + context.idx_provenance = ProvenanceMetadata(source_resource=resource_id) + + +@then('idx the provenance should have source_resource "{resource_id}"') +def step_check_provenance_resource(context: Any, resource_id: str) -> None: + assert context.idx_provenance.source_resource == resource_id + + +@then("idx the provenance should be current") +def step_check_provenance_current(context: Any) -> None: + assert context.idx_provenance.is_current is True + + +@then("idx the provenance should have a valid_from timestamp") +def step_check_provenance_timestamp(context: Any) -> None: + assert context.idx_provenance.valid_from is not None + + +@then("idx creating provenance with empty source_resource should raise ValidationError") +def step_provenance_empty_resource(context: Any) -> None: + try: + ProvenanceMetadata(source_resource="") + assert False, "Expected ValidationError" # noqa: B011 + except ValidationError: + pass + + +@when('idx I create a provenanced triple for resource "{resource_id}"') +def step_create_provenanced_triple(context: Any, resource_id: str) -> None: + triple = UKOTriple( + subject_uri="uko://s", predicate="uko:type", object_uri="uko://C" + ) + provenance = ProvenanceMetadata(source_resource=resource_id) + context.idx_pt = ProvenancedTriple(triple=triple, provenance=provenance) + + +@then("idx the provenanced triple should have the triple") +def step_check_pt_triple(context: Any) -> None: + assert context.idx_pt.triple.subject_uri == "uko://s" + + +@then("idx the provenanced triple should have the provenance") +def step_check_pt_provenance(context: Any) -> None: + assert context.idx_pt.provenance.source_resource == ULID_1 + + +@then("idx mutating the provenance should raise ValidationError") +def step_provenance_frozen(context: Any) -> None: + try: + context.idx_provenance.is_current = False # type: ignore[misc] + assert False, "Expected error" # noqa: B011 + except ValidationError: + # Pydantic frozen models raise ValidationError on mutation + pass + # Verify the object wasn't actually mutated + assert context.idx_provenance.is_current is True + + +@when("idx I create an IndexResult with {triples:d} triples and {text:d} text doc") +def step_create_index_result(context: Any, triples: int, text: int) -> None: + context.idx_result = IndexResult( + resource_id=ULID_1, + triple_count=triples, + text_docs_indexed=text, + analyzer_domain="python", + ) + + +@then("idx the IndexResult should have triple_count {count:d}") +def step_check_ir_triples(context: Any, count: int) -> None: + assert context.idx_result.triple_count == count + + +@then("idx the IndexResult should have text_docs_indexed {count:d}") +def step_check_ir_text(context: Any, count: int) -> None: + assert context.idx_result.text_docs_indexed == count + + +@then('idx the IndexResult should have analyzer_domain "{domain}"') +def step_check_ir_domain(context: Any, domain: str) -> None: + assert context.idx_result.analyzer_domain == domain + + +@then( + "idx creating IndexResult with negative triple_count should raise ValidationError" +) +def step_ir_negative(context: Any) -> None: + try: + IndexResult(resource_id=ULID_1, triple_count=-1) + assert False, "Expected ValidationError" # noqa: B011 + except ValidationError: + pass + + +@then("idx creating SearchResult with score 1.5 should raise ValueError") +def step_search_result_bad_score(context: Any) -> None: + try: + SearchResult(doc_id="d", content="c", score=1.5) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx creating SearchResult with empty doc_id should raise ValueError") +def step_search_result_empty_id(context: Any) -> None: + try: + SearchResult(doc_id="", content="c", score=0.5) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx creating IndexedDocument with negative char_count should raise ValueError") +def step_indexed_doc_negative(context: Any) -> None: + try: + IndexedDocument(project="p", doc_id="d", char_count=-1) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx creating IndexedDocument with empty project should raise ValueError") +def step_indexed_doc_empty_project(context: Any) -> None: + try: + IndexedDocument(project="", doc_id="d") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then( + "idx the graph backend should contain triple" + ' with predicate "{predicate}" and object "{obj}"' +) +def step_graph_contains_triple(context: Any, predicate: str, obj: str) -> None: + backend = context.idx_graph_backend + project = getattr(context, "idx_project", "local/test") + bindings = backend.query(project, "SELECT *") + matched = [b for b in bindings if b["p"] == predicate and b["o"] == obj] + assert len(matched) > 0, ( + f"No triple with predicate={predicate!r} object={obj!r} found in {bindings}" + ) + + +@then('idx the graph backend should not contain triple with object "{obj}"') +def step_graph_not_contains_object(context: Any, obj: str) -> None: + backend = context.idx_graph_backend + project = getattr(context, "idx_project", "local/test") + bindings = backend.query(project, "SELECT *") + matched = [b for b in bindings if b["o"] == obj] + assert len(matched) == 0, ( + f"Found unexpected triple(s) with object={obj!r}: {matched}" + ) + + +@then('idx the provenance source_resource should be "{expected}"') +def step_provenance_source_resource(context: Any, expected: str) -> None: + pt = context.idx_pt + assert pt.provenance.source_resource == expected, ( + f"Expected source_resource={expected!r}, got {pt.provenance.source_resource!r}" + ) + + +@then('idx the provenance source_path should be "{expected}"') +def step_provenance_source_path(context: Any, expected: str) -> None: + pt = context.idx_pt + assert pt.provenance.source_path == expected, ( + f"Expected source_path={expected!r}, got {pt.provenance.source_path!r}" + ) + + +@then( + "idx indexing a text document with whitespace-only project should raise ValueError" +) +def step_text_whitespace_project(context: Any) -> None: + try: + context.idx_text_backend.index_document(" ", "doc", "content", {}) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then( + "idx indexing a text document with whitespace-only content should raise ValueError" +) +def step_text_whitespace_content(context: Any) -> None: + try: + context.idx_text_backend.index_document("proj", "doc", " ", {}) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx adding a triple with whitespace-only subject should raise ValueError") +def step_graph_whitespace_subject(context: Any) -> None: + try: + context.idx_graph_backend.add_triple("proj", " ", "p", "o") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then('idx the text backend should contain doc_id "{doc_id}" in project "{project}"') +def step_text_contains_doc(context: Any, doc_id: str, project: str) -> None: + key = (project, doc_id) + assert key in context.idx_text_backend._docs, ( + f"doc_id {doc_id!r} not found in text backend for project {project!r}" + ) + + +@then('idx the vector backend should contain doc_id "{doc_id}" in project "{project}"') +def step_vector_contains_doc(context: Any, doc_id: str, project: str) -> None: + key = (project, doc_id) + assert key in context.idx_vector_backend._embeddings, ( + f"doc_id {doc_id!r} not found in vector backend for project {project!r}" + ) + + +@then("idx creating IndexResult with empty resource_id should raise ValidationError") +def step_ir_empty_resource_id(context: Any) -> None: + try: + IndexResult(resource_id="") + assert False, "Expected ValidationError" # noqa: B011 + except ValidationError: + pass + + +@then("idx creating SearchResult with score -0.1 should raise ValueError") +def step_search_result_negative_score(context: Any) -> None: + try: + SearchResult(doc_id="d", content="c", score=-0.1) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx creating SearchResult with NaN score should raise ValueError") +def step_search_result_nan_score(context: Any) -> None: + try: + SearchResult(doc_id="d", content="c", score=float("nan")) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx removing triples with empty-string subject should raise ValueError") +def step_graph_empty_string_subject(context: Any) -> None: + try: + context.idx_graph_backend.remove_triples( + "proj", subject="", predicate=None, obj=None + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx removing triples with empty-string predicate should raise ValueError") +def step_graph_empty_string_predicate(context: Any) -> None: + try: + context.idx_graph_backend.remove_triples( + "proj", subject="s", predicate="", obj=None + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx removing triples with empty-string obj should raise ValueError") +def step_graph_empty_string_obj(context: Any) -> None: + try: + context.idx_graph_backend.remove_triples( + "proj", subject="s", predicate=None, obj="" + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass diff --git a/features/steps/uko_indexer_common.py b/features/steps/uko_indexer_common.py new file mode 100644 index 000000000..92cfbaf7b --- /dev/null +++ b/features/steps/uko_indexer_common.py @@ -0,0 +1,62 @@ +"""Shared helpers, constants, and Background steps for UKO Indexer behave tests. + +All steps prefixed with ``idx`` to avoid AmbiguousStep collisions with +other feature files. +""" + +from __future__ import annotations + +from typing import Any + +from behave import given # type: ignore[import-untyped] + +from cleveragents.domain.models.acms.analyzers import AnalyzerRegistry +from cleveragents.domain.models.acms.index_stubs import ( + InMemoryGraphIndexBackend, + InMemoryTextIndexBackend, + InMemoryVectorIndexBackend, +) +from cleveragents.domain.models.core.resource import Resource +from features.mocks.uko_indexer_mocks import InMemoryContentReader + +__all__: list[str] = [] + +ULID_1 = "01HQ8ZDRX50000000000000001" +DEFAULT_PROJECT = "local/test" + + +# --------------------------------------------------------------------------- +# Test helpers +# --------------------------------------------------------------------------- + + +def _make_resource( + resource_id: str, + location: str | None = "src/main.py", + resource_type_name: str = "git-checkout", +) -> Resource: + """Create a minimal Resource for testing.""" + return Resource( + resource_id=resource_id, + resource_type_name=resource_type_name, + classification="physical", + location=location, + ) + + +# --------------------------------------------------------------------------- +# Background steps +# --------------------------------------------------------------------------- + + +@given("idx a clean analyzer registry") +def step_clean_registry(context: Any) -> None: + context.idx_registry = AnalyzerRegistry() + context.idx_content_reader = InMemoryContentReader() + + +@given("idx in-memory index backends") +def step_in_memory_backends(context: Any) -> None: + context.idx_text_backend = InMemoryTextIndexBackend() + context.idx_vector_backend = InMemoryVectorIndexBackend() + context.idx_graph_backend = InMemoryGraphIndexBackend() diff --git a/features/steps/uko_indexer_core_steps.py b/features/steps/uko_indexer_core_steps.py new file mode 100644 index 000000000..87ddbf954 --- /dev/null +++ b/features/steps/uko_indexer_core_steps.py @@ -0,0 +1,500 @@ +"""Analyzer lookup + core indexing + lifecycle + reindex steps. + +All steps prefixed with ``idx`` to avoid AmbiguousStep collisions with +other feature files. +""" + +from __future__ import annotations + +from typing import Any + +from behave import given, then, when # type: ignore[import-untyped] + +from cleveragents.application.services.uko_indexer import UKOIndexer +from cleveragents.domain.models.acms.markdown_analyzer import MarkdownAnalyzer +from cleveragents.domain.models.acms.python_analyzer import PythonAnalyzer +from features.mocks.uko_indexer_mocks import ( + FailingContentReader, + FailingGraphBackend, + FailingTextBackend, + FailingVectorBackend, + TrackingLifecycleHook, +) +from features.steps.uko_indexer_common import ( + DEFAULT_PROJECT, + ULID_1, + _make_resource, +) + +__all__: list[str] = [] + +# AnalyzerRegistry.get_for_resource steps + + +@given("idx a PythonAnalyzer is registered") +def step_register_python(context: Any) -> None: + context.idx_registry.register(PythonAnalyzer()) + + +@given("idx a PythonAnalyzer is registered with priority {priority:d}") +def step_register_python_priority(context: Any, priority: int) -> None: + context.idx_registry.register(PythonAnalyzer(), priority=priority) + + +@given( + 'idx an alternative Python analyzer "{name}" is registered with priority {priority:d}' +) +def step_register_alt_python(context: Any, name: str, priority: int) -> None: + """Register a stub ``.py`` analyzer with custom *name* as domain.""" + from cleveragents.domain.models.acms.analyzers import UKOTriple + + class _Alt: + supported_extensions = frozenset({".py"}) + domain = name + + def analyze(self, content: str, resource_uri: str) -> list[UKOTriple]: + return [] + + context.idx_registry.register(_Alt(), priority=priority) + + +@given("idx a MarkdownAnalyzer is registered") +def step_register_markdown(context: Any) -> None: + context.idx_registry.register(MarkdownAnalyzer()) + + +@when('idx I look up analyzer for a resource with location "{location}"') +def step_lookup_by_location(context: Any, location: str) -> None: + resource = _make_resource(ULID_1, location=location) + context.idx_found_analyzer = context.idx_registry.get_for_resource(resource) + + +@when("idx I look up analyzer for a resource with no location") +def step_lookup_no_location(context: Any) -> None: + resource = _make_resource(ULID_1, location=None) + context.idx_found_analyzer = context.idx_registry.get_for_resource(resource) + + +@then('idx the analyzer domain should be "{domain}"') +def step_check_analyzer_domain(context: Any, domain: str) -> None: + assert context.idx_found_analyzer is not None, "Expected an analyzer" + assert context.idx_found_analyzer.domain == domain + + +@then("idx no analyzer should be found") +def step_no_analyzer(context: Any) -> None: + assert context.idx_found_analyzer is None + + +# UKOIndexer setup steps + + +@given("idx a UKOIndexer with all backends") +def step_indexer_all(context: Any) -> None: + reader = context.idx_content_reader + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=reader, + ) + + +@given("idx a UKOIndexer without text backend") +def step_indexer_no_text(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=None, + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + ) + + +@given("idx a UKOIndexer without vector backend") +def step_indexer_no_vector(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=None, + content_reader=context.idx_content_reader, + ) + + +@given("idx a UKOIndexer with only graph backend") +def step_indexer_graph_only(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=None, + vector_backend=None, + content_reader=context.idx_content_reader, + ) + + +@given("idx a UKOIndexer with a failing content reader") +def step_indexer_failing_reader(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=FailingContentReader(), + ) + + +@given("idx a UKOIndexer with a tracking lifecycle hook") +def step_indexer_tracking_hook(context: Any) -> None: + context.idx_hook = TrackingLifecycleHook() + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + lifecycle_hook=context.idx_hook, + ) + + +@given("idx a UKOIndexer with a tracking lifecycle hook and failing reader") +def step_indexer_tracking_failing(context: Any) -> None: + context.idx_hook = TrackingLifecycleHook() + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=FailingContentReader(), + lifecycle_hook=context.idx_hook, + ) + + +@given("idx a UKOIndexer with a failing graph backend") +def step_indexer_failing_graph(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=FailingGraphBackend(), # type: ignore[arg-type] + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + ) + + +@given("idx a UKOIndexer with a failing text backend") +def step_indexer_failing_text(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=FailingTextBackend(), # type: ignore[arg-type] + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + ) + + +@given("idx a UKOIndexer with a failing vector backend") +def step_indexer_failing_vector(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=FailingVectorBackend(), # type: ignore[arg-type] + content_reader=context.idx_content_reader, + ) + + +# UKOIndexer action steps + + +@when('idx I index a Python resource "{rid}" with content "{content}"') +def step_index_python(context: Any, rid: str, content: str) -> None: + resource = _make_resource(rid, location="src/main.py") + # Unescape literal \n from Gherkin steps into real newlines + real_content = content.replace("\\n", "\n") + context.idx_content_reader.set_content(rid, real_content) + context.idx_result = context.idx_indexer.index_resource( + resource, + project=DEFAULT_PROJECT, + ) + + +@when('idx I index a Markdown resource "{rid}" with content "{content}"') +def step_index_markdown(context: Any, rid: str, content: str) -> None: + resource = _make_resource(rid, location="README.md") + real_content = content.replace("\\n", "\n") + context.idx_content_reader.set_content(rid, real_content) + context.idx_result = context.idx_indexer.index_resource( + resource, + project=DEFAULT_PROJECT, + ) + + +@when('idx I index a CSV resource "{rid}" with location "{location}"') +def step_index_csv(context: Any, rid: str, location: str) -> None: + resource = _make_resource(rid, location=location) + context.idx_content_reader.set_content(rid, "a,b,c") + context.idx_result = context.idx_indexer.index_resource( + resource, + project=DEFAULT_PROJECT, + ) + + +@when('idx I remove resource "{rid}" from project "{project}"') +def step_remove_resource(context: Any, rid: str, project: str) -> None: + context.idx_indexer.remove_resource(rid, project=project) + + +@when('idx I reindex resource "{rid}" with content "{content}"') +def step_reindex_resource(context: Any, rid: str, content: str) -> None: + resource = _make_resource(rid, location="src/main.py") + real_content = content.replace("\\n", "\n") + context.idx_content_reader.set_content(rid, real_content) + context.idx_result = context.idx_indexer.reindex_resource( + resource, + project=DEFAULT_PROJECT, + ) + + +@when( + 'idx I index Python resource "{rid}" under project "{project}" content "{content}"' +) +def step_index_python_in_project( + context: Any, rid: str, project: str, content: str +) -> None: + resource = _make_resource(rid, location="src/main.py") + real_content = content.replace("\\n", "\n") + context.idx_content_reader.set_content(rid, real_content) + context.idx_result = context.idx_indexer.index_resource( + resource, + project=project, + ) + + +@when('idx I reindex resource "{rid}" under project "{project}" content "{content}"') +def step_reindex_in_project( + context: Any, + rid: str, + project: str, + content: str, +) -> None: + resource = _make_resource(rid, location="src/main.py") + real_content = content.replace("\\n", "\n") + context.idx_content_reader.set_content(rid, real_content) + context.idx_result = context.idx_indexer.reindex_resource( + resource, + project=project, + ) + + +# UKOIndexer assertion steps + + +@then("idx the index result should have triple_count greater than 0") +def step_check_triples_gt0(context: Any) -> None: + assert context.idx_result.triple_count > 0, ( + f"Expected >0 triples, got {context.idx_result.triple_count}" + ) + + +@then("idx the index result should have triple_count {count:d}") +def step_check_exact_triples(context: Any, count: int) -> None: + assert context.idx_result.triple_count == count + + +@then("idx the index result should have text_docs_indexed {count:d}") +def step_check_text_docs(context: Any, count: int) -> None: + assert context.idx_result.text_docs_indexed == count + + +@then("idx the index result should have embeddings_indexed {count:d}") +def step_check_embeddings(context: Any, count: int) -> None: + assert context.idx_result.embeddings_indexed == count + + +@then('idx the index result should have analyzer_domain "{domain}"') +def step_check_domain(context: Any, domain: str) -> None: + assert context.idx_result.analyzer_domain == domain + + +@then("idx the index result should have {count:d} errors") +def step_check_error_count(context: Any, count: int) -> None: + assert len(context.idx_result.errors) == count, ( + f"Expected {count} errors, got {len(context.idx_result.errors)}: " + f"{context.idx_result.errors}" + ) + + +@then('idx the index result error should mention "{word}"') +def step_check_error_message(context: Any, word: str) -> None: + assert any(word.lower() in e.lower() for e in context.idx_result.errors), ( + f"No error mentioning '{word}' in {context.idx_result.errors}" + ) + + +@then('idx the index result errors should mention "{word}"') +def step_check_errors_mention(context: Any, word: str) -> None: + assert any(word.lower() in e.lower() for e in context.idx_result.errors), ( + f"No error mentioning '{word}' in {context.idx_result.errors}" + ) + + +@then('idx the graph backend should have triples for project "{project}"') +def step_graph_has_triples(context: Any, project: str) -> None: + count = context.idx_graph_backend.triple_count(project) + assert count > 0, f"Expected triples, got {count}" + + +@then('idx the graph backend should have {count:d} triples for project "{project}"') +def step_graph_triple_count(context: Any, count: int, project: str) -> None: + actual = context.idx_graph_backend.triple_count(project) + assert actual == count, f"Expected {count}, got {actual}" + + +@then("idx the text backend should have {count:d} document") +def step_text_doc_count_singular(context: Any, count: int) -> None: + assert context.idx_text_backend.document_count == count + + +@then("idx the text backend should have {count:d} documents") +def step_text_doc_count_plural(context: Any, count: int) -> None: + assert context.idx_text_backend.document_count == count + + +@then("idx the vector backend should have {count:d} embedding") +def step_vector_count_singular(context: Any, count: int) -> None: + assert context.idx_vector_backend.embedding_count == count + + +@then("idx the vector backend should have {count:d} embeddings") +def step_vector_count_plural(context: Any, count: int) -> None: + assert context.idx_vector_backend.embedding_count == count + + +@then("idx the indexer should track {count:d} indexed resource") +def step_indexed_count_singular(context: Any, count: int) -> None: + assert context.idx_indexer.indexed_resource_count == count + + +@then("idx the indexer should track {count:d} indexed resources") +def step_indexed_count_plural(context: Any, count: int) -> None: + assert context.idx_indexer.indexed_resource_count == count + + +# Validation error steps + + +@then("idx indexing with empty project should raise ValueError") +def step_index_empty_project(context: Any) -> None: + resource = _make_resource(ULID_1) + try: + context.idx_indexer.index_resource(resource, project="") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx removing with empty resource_id should raise ValueError") +def step_remove_empty_rid(context: Any) -> None: + try: + context.idx_indexer.remove_resource("", project=DEFAULT_PROJECT) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx removing with empty project should raise ValueError") +def step_remove_empty_project(context: Any) -> None: + try: + context.idx_indexer.remove_resource(ULID_1, project="") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +# Lifecycle hook steps + + +@then("idx the lifecycle hook should have received {count:d} on_indexed event") +def step_hook_indexed_singular(context: Any, count: int) -> None: + assert len(context.idx_hook.indexed_events) == count + + +@then("idx the lifecycle hook should have received {count:d} on_indexed events") +def step_hook_indexed_plural(context: Any, count: int) -> None: + actual = len(context.idx_hook.indexed_events) + assert actual == count, f"Expected {count} on_indexed events, got {actual}" + + +@then("idx the lifecycle hook should have received {count:d} on_error events") +def step_hook_error(context: Any, count: int) -> None: + assert len(context.idx_hook.error_events) == count + + +@then("idx the lifecycle hook should have received {count:d} on_removed event") +def step_hook_removed(context: Any, count: int) -> None: + assert len(context.idx_hook.removed_events) == count + + +@then("idx the lifecycle hook should have received {count:d} on_removed events") +def step_hook_removed_plural(context: Any, count: int) -> None: + actual = len(context.idx_hook.removed_events) + assert actual == count, f"Expected {count} on_removed events, got {actual}" + + +# Constructor validation steps + + +@then("idx creating UKOIndexer with invalid registry should raise TypeError") +def step_invalid_registry(context: Any) -> None: + try: + UKOIndexer( + analyzer_registry="not a registry", # type: ignore[arg-type] + graph_backend=context.idx_graph_backend, + ) + assert False, "Expected TypeError" # noqa: B011 + except TypeError: + pass + + +@then("idx the indexer should have text backend") +def step_has_text(context: Any) -> None: + assert context.idx_indexer.has_text_backend is True + + +@then("idx the indexer should have vector backend") +def step_has_vector(context: Any) -> None: + assert context.idx_indexer.has_vector_backend is True + + +@then("idx the indexer should not have text backend") +def step_no_text(context: Any) -> None: + assert context.idx_indexer.has_text_backend is False + + +@then("idx the indexer should not have vector backend") +def step_no_vector(context: Any) -> None: + assert context.idx_indexer.has_vector_backend is False + + +# Reindex validation steps + + +@then("idx reindexing with empty project should raise ValueError") +def step_reindex_empty_project(context: Any) -> None: + resource = _make_resource(ULID_1) + try: + context.idx_indexer.reindex_resource(resource, project="") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx reindexing with whitespace-only project should raise ValueError") +def step_reindex_whitespace_project(context: Any) -> None: + resource = _make_resource(ULID_1) + try: + context.idx_indexer.reindex_resource(resource, project=" ") + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass diff --git a/features/steps/uko_indexer_coverage_steps.py b/features/steps/uko_indexer_coverage_steps.py new file mode 100644 index 000000000..bd548f2f8 --- /dev/null +++ b/features/steps/uko_indexer_coverage_steps.py @@ -0,0 +1,496 @@ +"""Validation edge cases + coverage scenarios for UKO Indexer behave tests.""" + +from __future__ import annotations + +import shutil +import tempfile +import threading +from pathlib import Path +from typing import Any + +from behave import given, then, when # type: ignore[import-untyped] + +from cleveragents.application.services.uko_indexer import UKOIndexer +from cleveragents.domain.models.acms.analyzers import ( + AnalyzerRegistry, + UKOTriple, +) +from cleveragents.domain.models.acms.index_backends import IndexedDocument +from cleveragents.domain.models.acms.provenance import ( + IndexResult, + ProvenancedTriple, + ProvenanceMetadata, +) +from cleveragents.domain.models.acms.python_analyzer import PythonAnalyzer +from cleveragents.domain.models.core.resource import Resource +from features.mocks.uko_indexer_mocks import ( + FailingContentReader, + InMemoryContentReader, +) +from features.steps.uko_indexer_common import ( + DEFAULT_PROJECT, + ULID_1, + _make_resource, +) + +__all__: list[str] = [] + + +@then("idx LocationContentReader should reject path with dotdot component") +def step_lcr_dotdot(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + reader = LocationContentReader() + resource = _make_resource(ULID_1, location="../etc/passwd") + try: + reader.read_content(resource) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "traversal" in str(exc).lower() + + +@then("idx LocationContentReader should reject resource with no location") +def step_lcr_no_location(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + reader = LocationContentReader() + resource = _make_resource(ULID_1, location=None) + try: + reader.read_content(resource) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "no location" in str(exc).lower() + + +@then("idx LocationContentReader should reject max_content_size of 0") +def step_lcr_zero_size(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + try: + LocationContentReader(max_content_size=0) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "positive" in str(exc).lower() + + +@then("idx LocationContentReader should read a real file successfully") +def step_lcr_read_real(context: Any) -> None: + import tempfile + + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + with tempfile.NamedTemporaryFile( + mode="w", suffix=".py", delete=False, encoding="utf-8" + ) as f: + f.write("x = 1\n") + f.flush() + path = f.name + + from pathlib import Path + + reader = LocationContentReader(base_dir=Path(path).parent) + resource = _make_resource(ULID_1, location=path) + content = reader.read_content(resource) + assert content == "x = 1\n" + Path(path).unlink() + + +@then("idx LocationContentReader should reject non-existent file") +def step_lcr_nonexistent(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + reader = LocationContentReader() + resource = _make_resource(ULID_1, location="/tmp/nonexistent_file_abc123.py") + try: + reader.read_content(resource) + assert False, "Expected OSError" # noqa: B011 + except OSError: + pass + + +@then("idx LocationContentReader should reject path escaping base_dir") +def step_lcr_escape_base(context: Any) -> None: + import tempfile + from pathlib import Path + + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + # Create a temp dir to use as base_dir + base = Path(tempfile.mkdtemp()) + reader = LocationContentReader(base_dir=base) + # Try to read /etc/hostname which is outside the base + resource = _make_resource(ULID_1, location="/etc/hostname") + try: + reader.read_content(resource) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "escapes" in str(exc).lower() + base.rmdir() + + +@then("idx LocationContentReader should reject a non-regular file") +def step_lcr_non_regular_file(context: Any) -> None: + import os + import tempfile + from pathlib import Path + + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + tmp_dir = Path(tempfile.mkdtemp()) + fifo_path = tmp_dir / "testpipe" + os.mkfifo(str(fifo_path)) + reader = LocationContentReader(base_dir=tmp_dir) + resource = _make_resource(ULID_1, location=str(fifo_path)) + try: + reader.read_content(resource) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert ( + "regular file" in str(exc).lower() + or "not a regular file" in str(exc).lower() + ), str(exc) + finally: + fifo_path.unlink() + tmp_dir.rmdir() + + +@then("idx LocationContentReader should reject content exceeding max size") +def step_lcr_max_content_size(context: Any) -> None: + import tempfile + from pathlib import Path + + from cleveragents.application.services.uko_indexer_protocols import ( + LocationContentReader, + ) + + tmp_dir = Path(tempfile.mkdtemp()) + big_file = tmp_dir / "big.txt" + big_file.write_text("x" * 100, encoding="utf-8") + reader = LocationContentReader(base_dir=tmp_dir, max_content_size=10) + resource = _make_resource(ULID_1, location=str(big_file)) + try: + reader.read_content(resource) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "exceeds" in str(exc).lower(), str(exc) + finally: + big_file.unlink() + tmp_dir.rmdir() + + +@then("idx creating UKOIndexer with invalid graph_backend should raise TypeError") +def step_idx_invalid_graph(context: Any) -> None: + try: + UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend="not_a_backend", # type: ignore[arg-type] + ) + assert False, "Expected TypeError" # noqa: B011 + except TypeError: + pass + + +@then("idx the indexer analyzer_registry should be the same registry") +def step_idx_registry_prop(context: Any) -> None: + assert context.idx_indexer.analyzer_registry is context.idx_registry + + +@given("idx a lifecycle hook that raises on on_indexed") +def step_idx_hook_fail_indexed(context: Any) -> None: + class _FailOnIndexed: + def on_indexed(self, result: IndexResult) -> None: + raise RuntimeError("hook boom") + + def on_removed(self, resource_id: str, project: str) -> None: + pass + + def on_error(self, resource_id: str, error: str) -> None: + pass + + context.idx_failing_hook = _FailOnIndexed() + + +@given("idx a lifecycle hook that raises on on_removed") +def step_idx_hook_fail_removed(context: Any) -> None: + class _FailOnRemoved: + def on_indexed(self, result: IndexResult) -> None: + pass + + def on_removed(self, resource_id: str, project: str) -> None: + raise RuntimeError("hook boom") + + def on_error(self, resource_id: str, error: str) -> None: + pass + + context.idx_failing_hook = _FailOnRemoved() + + +@given("idx a lifecycle hook that raises on on_error") +def step_idx_hook_fail_error(context: Any) -> None: + class _FailOnError: + def on_indexed(self, result: IndexResult) -> None: + pass + + def on_removed(self, resource_id: str, project: str) -> None: + pass + + def on_error(self, resource_id: str, error: str) -> None: + raise RuntimeError("hook boom") + + context.idx_failing_hook = _FailOnError() + + +@given("idx a content reader that raises on read") +def step_idx_failing_reader(context: Any) -> None: + context.idx_content_reader = FailingContentReader() + + +@when("idx I create an indexer with the failing hook") +def step_idx_make_with_hook(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + lifecycle_hook=context.idx_failing_hook, + ) + + +@then('idx removing resource "{rid}" should not raise') +def step_idx_remove_no_raise(context: Any, rid: str) -> None: + context.idx_indexer.remove_resource(rid, project=DEFAULT_PROJECT) + + +@then("idx the index result should have errors") +def step_idx_result_has_errors(context: Any) -> None: + assert len(context.idx_result.errors) > 0, "Expected errors" + + +@given("idx an analyzer that raises RuntimeError") +def step_idx_failing_analyzer(context: Any) -> None: + class _FailingAnalyzer: + domain = "failing" + supported_extensions = frozenset({".py"}) + + def analyze( + self, + content: str, + resource_uri: str, + ) -> list[UKOTriple]: + raise RuntimeError("analyzer boom") + + context.idx_registry = AnalyzerRegistry() + context.idx_registry.register(_FailingAnalyzer()) # type: ignore[arg-type] + + +@given("idx an analyzer that produces {n:d} triples") +def step_idx_many_triples_analyzer(context: Any, n: int) -> None: + class _ManyTriplesAnalyzer: + domain = "many" + supported_extensions = frozenset({".py"}) + + def analyze( + self, + content: str, + resource_uri: str, + ) -> list[UKOTriple]: + return [ + UKOTriple( + subject_uri=f"uko://s{i}", + predicate="rdf:type", + object_uri="uko://Thing", + ) + for i in range(n) + ] + + context.idx_registry = AnalyzerRegistry() + context.idx_registry.register(_ManyTriplesAnalyzer()) # type: ignore[arg-type] + + +@when("idx I create an indexer with max_triples {n:d}") +def step_idx_make_max_triples(context: Any, n: int) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + max_triples=n, + ) + + +@then("idx the index result triple_count should be at most {n:d}") +def step_idx_triple_count_capped(context: Any, n: int) -> None: + assert context.idx_result.triple_count <= n, ( + f"Expected at most {n} triples, got {context.idx_result.triple_count}" + ) + + +@given("idx an AnalyzerRegistry with the PythonAnalyzer") +def step_registry_with_python(context: Any) -> None: + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer()) + context.idx_registry = registry + + +@when('idx I analyze content "{content}" with resource URI "{resource_uri}"') +def step_analyze_content(context: Any, content: str, resource_uri: str) -> None: + real_content = content.replace("\\n", "\n") + analyzer = context.idx_registry.get_for_extension(".py") + assert analyzer is not None + context.idx_triples = analyzer.analyze(real_content, resource_uri) + + +@when( + "idx I wrap the first triple with provenance" + ' for resource "{resource_id}" at path "{path}"' +) +def step_wrap_provenance(context: Any, resource_id: str, path: str) -> None: + assert len(context.idx_triples) > 0, "No triples to wrap" + prov = ProvenanceMetadata( + source_resource=resource_id, + source_path=path, + ) + context.idx_pt = ProvenancedTriple( + triple=context.idx_triples[0], + provenance=prov, + ) + + +@given('idx a content reader with resource "{rid}" content "{content}"') +def step_idx_content_reader_setup(context: Any, rid: str, content: str) -> None: + reader = InMemoryContentReader() + reader.set_content(rid, content.replace("\\n", "\n")) + context.idx_content_reader = reader + + +@when("idx I create an indexer") +def step_idx_create_indexer(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + ) + + +@when('idx I index resource "{rid}" in project "{project}"') +@then('idx I index resource "{rid}" in project "{project}"') +def step_idx_index_resource_generic(context: Any, rid: str, project: str) -> None: + resource = Resource( + resource_id=rid, + resource_type_name="git-checkout", + location="src/example.py", + classification="physical", + ) + context.idx_result = context.idx_indexer.index_resource( + resource, + project=project, + ) + + +@then("idx creating IndexedDocument with empty doc_id should raise ValueError") +def step_indexed_doc_empty_docid(context: Any) -> None: + try: + IndexedDocument(project="p", doc_id="", char_count=0) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx creating ProvenanceMetadata with invalid source_range should raise") +def step_prov_invalid_source_range(context: Any) -> None: + try: + ProvenanceMetadata(source_resource="r1", source_range="not-valid") + assert False, "Expected ValueError" # noqa: B011 + except Exception: + pass + + +@then("idx creating ProvenanceMetadata with empty source_range should succeed") +def step_prov_empty_source_range(context: Any) -> None: + pm = ProvenanceMetadata(source_resource="r1", source_range="") + assert pm.source_range == "" + + +@then("idx file watcher should cancel prior dest timer on rapid sequence") +def step_fw_cancel_dest_timer(context: Any) -> None: + """File B has pending timer, then move A->B cancels it (line 375).""" + from watchdog.events import FileModifiedEvent, FileMovedEvent + + from cleveragents.application.services.resource_file_watcher import ( + ResourceFileWatcher, + ) + + ev = threading.Event() + + def cb(rid: str, _p: str, _ct: str) -> None: + ev.set() + + td = tempfile.mkdtemp() + try: + fa = Path(td) / "a.py" + fb = Path(td) / "b.py" + fa.write_text("x=1\n", encoding="utf-8") + fb.write_text("y=1\n", encoding="utf-8") + w = ResourceFileWatcher(on_change=cb, debounce_seconds=10.0) + w.watch(fa, resource_id="01HQ8ZDRX50000000000000088", project="local/t") + w.watch(fb, resource_id="01HQ8ZDRX50000000000000089", project="local/t") + w.start() + # Creates pending timer for B + w._handle_fs_event(FileModifiedEvent(str(fb.resolve()))) + assert str(fb.resolve()) in w._pending_timers + # Move A -> B: should cancel B's pending timer (line 375) + w._handle_fs_event(FileMovedEvent(str(fa.resolve()), str(fb.resolve()))) + # The move creates a new timer for dest (B); wait for it + w._debounce_seconds = 0.05 # speed up for test + w._handle_fs_event(FileModifiedEvent(str(fb.resolve()))) + assert ev.wait(timeout=2.0) + w.stop() + finally: + shutil.rmtree(td, ignore_errors=True) + + +@then("idx file watcher EventBus move event should include dest_path") +def step_fw_eventbus_move_dest(context: Any) -> None: + """Verify EventBus event for move includes dest_path in details.""" + from watchdog.events import FileMovedEvent + + from cleveragents.application.services.resource_file_watcher import ( + ResourceFileWatcher, + ) + from features.mocks.uko_indexer_mocks import TrackingEventBus + + td = tempfile.mkdtemp() + try: + fp = Path(td) / "src.py" + fp.write_text("x=1\n", encoding="utf-8") + dp = str((Path(td) / "dst.py").resolve()) + bus = TrackingEventBus() + w = ResourceFileWatcher(event_bus=bus, debounce_seconds=0.05) + w.watch(fp, resource_id="01HQ8ZDRX50000000000000066", project="local/t") + w.start() + w._handle_fs_event(FileMovedEvent(str(fp.resolve()), dp)) + assert bus.wait(timeout=2.0) + details = bus.events[-1].details + assert "dest_path" in details, f"dest_path missing from {details}" + assert details["dest_path"] == dp + w.stop() + finally: + shutil.rmtree(td, ignore_errors=True) diff --git a/features/steps/uko_indexer_edge_steps.py b/features/steps/uko_indexer_edge_steps.py new file mode 100644 index 000000000..828f3009b --- /dev/null +++ b/features/steps/uko_indexer_edge_steps.py @@ -0,0 +1,472 @@ +"""Edge-case coverage steps for UKO Indexer behave tests. + +Covers removal-failure paths, max_triples validation, reindex on fresh +resource, index_graph internals (empty obj, rdfs:label, provenance), +index_stubs edge cases, and index_backends protocol base methods. + +All steps prefixed with ``idx`` to avoid AmbiguousStep collisions with +other feature files. +""" + +from __future__ import annotations + +from typing import Any + +from behave import given, then, when # type: ignore[import-untyped] + +from cleveragents.application.services.uko_indexer import UKOIndexer +from cleveragents.domain.models.acms.analyzers import UKOTriple +from cleveragents.domain.models.acms.index_stubs import ( + InMemoryGraphIndexBackend, + InMemoryTextIndexBackend, + InMemoryVectorIndexBackend, +) +from cleveragents.domain.models.acms.provenance import ( + ProvenancedTriple, + ProvenanceMetadata, +) +from features.mocks.uko_indexer_mocks import ( + RemovalFailingGraphBackend, + RemovalFailingTextBackend, + RemovalFailingVectorBackend, + SelectiveFailingGraphBackend, +) +from features.steps.uko_indexer_common import ULID_1, _make_resource + +__all__: list[str] = [] + + +# --------------------------------------------------------------------------- +# Removal-failing backend steps +# --------------------------------------------------------------------------- + + +@given("idx a UKOIndexer with removal-failing graph backend") +def step_indexer_removal_failing_graph(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=RemovalFailingGraphBackend(), + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + ) + + +@given("idx a UKOIndexer with removal-failing text backend") +def step_indexer_removal_failing_text(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=RemovalFailingTextBackend(), + vector_backend=context.idx_vector_backend, + content_reader=context.idx_content_reader, + ) + + +@given("idx a UKOIndexer with removal-failing vector backend") +def step_indexer_removal_failing_vector(context: Any) -> None: + context.idx_indexer = UKOIndexer( + analyzer_registry=context.idx_registry, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=RemovalFailingVectorBackend(), + content_reader=context.idx_content_reader, + ) + + +# --------------------------------------------------------------------------- +# max_triples validation +# --------------------------------------------------------------------------- + + +@then("idx creating a UKOIndexer with max_triples 0 should raise ValueError") +def step_max_triples_zero(context: Any) -> None: + from cleveragents.domain.models.acms.analyzers import AnalyzerRegistry + + try: + UKOIndexer( + analyzer_registry=AnalyzerRegistry(), + graph_backend=InMemoryGraphIndexBackend(), + max_triples=0, + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "positive" in str(exc).lower() + + +# --------------------------------------------------------------------------- +# reindex_resource on fresh resource +# --------------------------------------------------------------------------- + + +@when('idx I reindex resource "{rid}" in project "{project}"') +def step_reindex_resource(context: Any, rid: str, project: str) -> None: + resource = _make_resource(rid) + context.idx_result = context.idx_indexer.reindex_resource( + resource, + project=project, + ) + + +# --------------------------------------------------------------------------- +# index_graph internals edge cases +# --------------------------------------------------------------------------- + + +@then("idx index_graph should skip triples with empty object") +def step_index_graph_empty_obj(context: Any) -> None: + from cleveragents.application.services.uko_indexer_internals import index_graph + + backend = InMemoryGraphIndexBackend() + # Use model_construct to bypass the validator that requires at + # least one of object_uri / object_value to be non-empty. This + # tests the defensive guard inside index_graph. + triple = UKOTriple.model_construct( + subject_uri="urn:x", + predicate="uko:name", + object_uri="", + object_value="", + confidence=1.0, + ) + prov = ProvenanceMetadata(source_resource=ULID_1, source_path="f.py") + pt = ProvenancedTriple.model_construct(triple=triple, provenance=prov) + errors: list[str] = [] + stored, subjects = index_graph( + backend, + "local/test", + _make_resource(ULID_1), + [pt], + errors, + "uko://resource/" + ULID_1, + ) + assert stored == 0 + assert len(subjects) == 0 + + +@then("idx index_graph should tolerate rdfs:label failure") +def step_index_graph_label_failure(context: Any) -> None: + from cleveragents.application.services.uko_indexer_internals import index_graph + + backend = SelectiveFailingGraphBackend(fail_on={"rdfs:label"}) + triple = UKOTriple( + subject_uri="urn:x", + predicate="uko:name", + object_uri="urn:target", + object_value="Target Label", + ) + pt = ProvenancedTriple( + triple=triple, + provenance=ProvenanceMetadata(source_resource=ULID_1, source_path="f.py"), + ) + errors: list[str] = [] + stored, _subjects = index_graph( + backend, + "local/test", + _make_resource(ULID_1), + [pt], + errors, + "uko://resource/" + ULID_1, + ) + assert stored == 1 + assert any("label" in e.lower() for e in errors) + + +@then("idx index_graph should tolerate provenance link failure") +def step_index_graph_provenance_failure(context: Any) -> None: + from cleveragents.application.services.uko_indexer_internals import index_graph + + backend = SelectiveFailingGraphBackend(fail_on={"uko:sourceResource"}) + triple = UKOTriple( + subject_uri="urn:x", + predicate="uko:name", + object_uri="urn:target", + ) + pt = ProvenancedTriple( + triple=triple, + provenance=ProvenanceMetadata(source_resource=ULID_1, source_path="f.py"), + ) + errors: list[str] = [] + stored, _subjects = index_graph( + backend, + "local/test", + _make_resource(ULID_1), + [pt], + errors, + "uko://resource/" + ULID_1, + ) + assert stored == 1 + assert any("provenance" in e.lower() for e in errors) + + +# --------------------------------------------------------------------------- +# index_stubs edge cases +# --------------------------------------------------------------------------- + + +@then("idx text search should stop at the limit") +def step_text_search_limit(context: Any) -> None: + backend = InMemoryTextIndexBackend() + backend.index_document("p", "d1", "hello world", {}) + backend.index_document("p", "d2", "hello again", {}) + backend.index_document("p", "d3", "hello third", {}) + results = backend.search("p", "hello", limit=2) + assert len(results) == 2 + + +@then("idx vector search_similar with empty embedding should raise ValueError") +def step_vector_empty_embedding(context: Any) -> None: + backend = InMemoryVectorIndexBackend() + try: + backend.search_similar("p", []) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "non-empty" in str(exc).lower() + + +@then("idx vector search_similar with limit 0 should raise ValueError") +def step_vector_zero_limit(context: Any) -> None: + backend = InMemoryVectorIndexBackend() + try: + backend.search_similar("p", [1.0], limit=0) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert "positive" in str(exc).lower() + + +@then("idx vector search_similar filters by project") +def step_vector_project_filter(context: Any) -> None: + backend = InMemoryVectorIndexBackend() + backend.index_embedding("proj_a", "d1", [1.0], {}) + backend.index_embedding("proj_b", "d2", [1.0], {}) + results = backend.search_similar("proj_a", [1.0], limit=10) + assert len(results) == 1 + assert results[0].doc_id == "d1" + + +@then("idx graph remove_triples on missing project should be no-op") +def step_graph_remove_missing_project(context: Any) -> None: + backend = InMemoryGraphIndexBackend() + backend.remove_triples("nonexistent", subject="s", predicate=None, obj=None) + assert backend.triple_count() == 0 + + +@then("idx graph triple_count without project should sum all projects") +def step_graph_triple_count_all(context: Any) -> None: + backend = InMemoryGraphIndexBackend() + backend.add_triple("p1", "s1", "pred", "o1") + backend.add_triple("p2", "s2", "pred", "o2") + assert backend.triple_count(project=None) == 2 + assert backend.triple_count(project="p1") == 1 + + +# --------------------------------------------------------------------------- +# index_backends protocol base method coverage +# --------------------------------------------------------------------------- + + +@then("idx all index backend protocol base methods should be callable") +def step_protocol_base_methods(context: Any) -> None: + from cleveragents.domain.models.acms.index_backends import ( + GraphIndexBackend, + TextIndexBackend, + VectorIndexBackend, + ) + + class _TextSub(TextIndexBackend): + pass + + class _VectorSub(VectorIndexBackend): + pass + + class _GraphSub(GraphIndexBackend): + pass + + t = _TextSub() # type: ignore[abstract] + t.index_document("p", "d", "c", {}) + t.search("p", "q", limit=1) + t.remove_document("p", "d") + t.rebuild_index("p") + + v = _VectorSub() # type: ignore[abstract] + v.index_embedding("p", "d", [1.0], {}) + v.search_similar("p", [1.0], limit=1) + v.remove_embedding("p", "d") + + g = _GraphSub() # type: ignore[abstract] + g.add_triple("p", "s", "pred", "o") + g.query("p", "SELECT ?s WHERE {}") + g.remove_triples("p", subject="s", predicate=None, obj=None) + + +# --------------------------------------------------------------------------- +# min_relevance filtering (P3-3) +# --------------------------------------------------------------------------- + + +@then("idx vector search_similar with min_relevance above 1.0 raises ValueError") +def step_vector_min_relevance_above(context: Any) -> None: + backend = InMemoryVectorIndexBackend() + backend.index_embedding("p", "d1", [1.0], {}) + try: + backend.search_similar("p", [1.0], limit=10, min_relevance=1.1) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx vector search_similar with min_relevance 1.0 includes exact matches") +def step_vector_min_relevance_exact(context: Any) -> None: + backend = InMemoryVectorIndexBackend() + backend.index_embedding("p", "d1", [1.0], {}) + results = backend.search_similar("p", [1.0], limit=10, min_relevance=1.0) + assert len(results) == 1 + assert results[0].doc_id == "d1" + + +@then("idx vector search_similar with min_relevance 0.0 includes all matches") +def step_vector_min_relevance_zero(context: Any) -> None: + backend = InMemoryVectorIndexBackend() + backend.index_embedding("p", "d1", [1.0], {}) + backend.index_embedding("p", "d2", [2.0], {}) + results = backend.search_similar("p", [1.0], limit=10, min_relevance=0.0) + assert len(results) == 2 + + +# --------------------------------------------------------------------------- +# Stub size-limit enforcement (P2-16) +# --------------------------------------------------------------------------- + + +@then("idx text stub should raise RuntimeError when max_entries exceeded") +def step_text_stub_size_limit(context: Any) -> None: + backend = InMemoryTextIndexBackend(max_entries=2) + backend.index_document("p", "d1", "content1", {}) + backend.index_document("p", "d2", "content2", {}) + try: + backend.index_document("p", "d3", "content3", {}) + assert False, "Expected RuntimeError" # noqa: B011 + except RuntimeError as exc: + assert "max_entries" in str(exc) + + +@then("idx vector stub should raise RuntimeError when max_entries exceeded") +def step_vector_stub_size_limit(context: Any) -> None: + backend = InMemoryVectorIndexBackend(max_entries=2) + backend.index_embedding("p", "d1", [1.0], {}) + backend.index_embedding("p", "d2", [2.0], {}) + try: + backend.index_embedding("p", "d3", [3.0], {}) + assert False, "Expected RuntimeError" # noqa: B011 + except RuntimeError as exc: + assert "max_entries" in str(exc) + + +@then("idx graph stub should raise RuntimeError when max_entries exceeded") +def step_graph_stub_size_limit(context: Any) -> None: + backend = InMemoryGraphIndexBackend(max_entries=2) + backend.add_triple("p", "s1", "pred", "o1") + backend.add_triple("p", "s2", "pred", "o2") + try: + backend.add_triple("p", "s3", "pred", "o3") + assert False, "Expected RuntimeError" # noqa: B011 + except RuntimeError as exc: + assert "max_entries" in str(exc) + + +# --------------------------------------------------------------------------- +# Provenance metadata persistence (P2-22, P2-27) +# --------------------------------------------------------------------------- + + +@then("idx index_graph should store provenance metadata triples") +def step_index_graph_provenance_metadata(context: Any) -> None: + from cleveragents.application.services.uko_indexer_internals import index_graph + + backend = InMemoryGraphIndexBackend() + triple = UKOTriple( + subject_uri="urn:x", + predicate="uko:name", + object_uri="urn:target", + object_value="TargetLabel", + confidence=0.85, + ) + prov = ProvenanceMetadata( + source_resource=ULID_1, + source_path="src/example.py", + source_range="10-25", + ) + pt = ProvenancedTriple(triple=triple, provenance=prov) + errors: list[str] = [] + stored, _subjects = index_graph( + backend, + "local/test", + _make_resource(ULID_1), + [pt], + errors, + "uko://resource/" + ULID_1, + ) + assert stored == 1 + + # Query all triples for the project + bindings = backend.query("local/test", "SELECT * WHERE { ?s ?p ?o }") + + # Build a predicate -> object map for the subject + preds = {b["p"]: b["o"] for b in bindings if b["s"] == "urn:x"} + + # Data triple + assert preds.get("uko:name") == "urn:target" + # Provenance metadata + assert preds.get("uko:sourceResource") == "uko://resource/" + ULID_1 + assert preds.get("uko:sourcePath") == "src/example.py" + assert preds.get("uko:sourceRange") == "10-25" + assert "uko:validFrom" in preds # ISO timestamp + assert preds.get("uko:isCurrent") == "true" + # Confidence < 1.0 should be stored + assert preds.get("uko:confidence") == "0.85" + + +# --------------------------------------------------------------------------- +# Concurrency — thread-safe indexing (#58) +# --------------------------------------------------------------------------- + + +@then("idx concurrent indexing of different resources should be thread-safe") +def step_idx_concurrent_indexing(context: Any) -> None: + import threading + + from cleveragents.domain.models.acms.python_analyzer import PythonAnalyzer + from features.mocks.uko_indexer_mocks import InMemoryContentReader + + reg = context.idx_registry + reg.register(PythonAnalyzer(), priority=10) + reader = InMemoryContentReader() + indexer = UKOIndexer( + analyzer_registry=reg, + graph_backend=context.idx_graph_backend, + text_backend=context.idx_text_backend, + vector_backend=context.idx_vector_backend, + content_reader=reader, + ) + results: list[Any] = [None] * 4 + errs: list[Exception] = [] + + rids = [f"01HQ8ZDRX5000000000000{i:04d}" for i in range(4)] + + def _do(i: int) -> None: + rid = rids[i] + reader.set_content(rid, f"def f{i}(): pass\n") + try: + results[i] = indexer.index_resource( + _make_resource(rid, f"src/m{i}.py"), + project="local/app", + ) + except Exception as e: + errs.append(e) + + ts = [threading.Thread(target=_do, args=(i,)) for i in range(4)] + for t in ts: + t.start() + for t in ts: + t.join(timeout=10) + assert not errs, f"Concurrent errors: {errs}" + assert indexer.indexed_resource_count == 4 diff --git a/features/steps/uko_indexer_protocol_steps.py b/features/steps/uko_indexer_protocol_steps.py new file mode 100644 index 000000000..f1dad042e --- /dev/null +++ b/features/steps/uko_indexer_protocol_steps.py @@ -0,0 +1,282 @@ +"""Protocol stubs, UKOTriple validation, and direct internal-call steps. + +Coverage scenarios that exercise Protocol ``raise NotImplementedError`` +stubs, Pydantic validator error paths, and internal helper functions +called directly (bypassing the public API) to ensure slipcover traces +the code. + +All steps prefixed with ``idx`` to avoid AmbiguousStep collisions. +""" + +from __future__ import annotations + +from typing import Any + +from behave import then # type: ignore[import-untyped] + +from cleveragents.domain.models.acms.analyzers import UKOTriple +from cleveragents.domain.models.acms.provenance import IndexResult +from features.steps.uko_indexer_common import ULID_1, _make_resource + +__all__: list[str] = [] + + +# --------------------------------------------------------------------------- +# Protocol stub NotImplementedError steps +# --------------------------------------------------------------------------- + + +@then( + "idx calling ContentReader.read_content directly should raise NotImplementedError" +) +def step_proto_content_reader(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + ContentReader, + ) + + class _Bare(ContentReader): # type: ignore[misc] + pass + + bare = _Bare() # type: ignore[abstract] + try: + bare.read_content(_make_resource(ULID_1)) + assert False, "Expected NotImplementedError" # noqa: B011 + except NotImplementedError: + pass + + +@then( + "idx calling IndexLifecycleHook.on_indexed directly" + " should raise NotImplementedError" +) +def step_proto_hook_indexed(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + IndexLifecycleHook, + ) + + class _Bare(IndexLifecycleHook): # type: ignore[misc] + pass + + bare = _Bare() # type: ignore[abstract] + try: + bare.on_indexed(IndexResult(resource_id=ULID_1)) + assert False, "Expected NotImplementedError" # noqa: B011 + except NotImplementedError: + pass + + +@then( + "idx calling IndexLifecycleHook.on_removed directly" + " should raise NotImplementedError" +) +def step_proto_hook_removed(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + IndexLifecycleHook, + ) + + class _Bare(IndexLifecycleHook): # type: ignore[misc] + pass + + bare = _Bare() # type: ignore[abstract] + try: + bare.on_removed(ULID_1, "p") + assert False, "Expected NotImplementedError" # noqa: B011 + except NotImplementedError: + pass + + +@then( + "idx calling IndexLifecycleHook.on_error directly should raise NotImplementedError" +) +def step_proto_hook_error(context: Any) -> None: + from cleveragents.application.services.uko_indexer_protocols import ( + IndexLifecycleHook, + ) + + class _Bare(IndexLifecycleHook): # type: ignore[misc] + pass + + bare = _Bare() # type: ignore[abstract] + try: + bare.on_error(ULID_1, "err") + assert False, "Expected NotImplementedError" # noqa: B011 + except NotImplementedError: + pass + + +@then( + "idx calling AnalyzerProtocol.supported_extensions directly" + " should raise NotImplementedError" +) +def step_proto_analyzer_ext(context: Any) -> None: + from cleveragents.domain.models.acms.analyzers import AnalyzerProtocol + + class _Bare(AnalyzerProtocol): # type: ignore[misc] + pass + + bare = _Bare() # type: ignore[abstract] + try: + _ = bare.supported_extensions + assert False, "Expected NotImplementedError" # noqa: B011 + except NotImplementedError: + pass + + +@then("idx calling AnalyzerProtocol.domain directly should raise NotImplementedError") +def step_proto_analyzer_domain(context: Any) -> None: + from cleveragents.domain.models.acms.analyzers import AnalyzerProtocol + + class _Bare(AnalyzerProtocol): # type: ignore[misc] + pass + + bare = _Bare() # type: ignore[abstract] + try: + _ = bare.domain + assert False, "Expected NotImplementedError" # noqa: B011 + except NotImplementedError: + pass + + +@then("idx calling AnalyzerProtocol.analyze directly should raise NotImplementedError") +def step_proto_analyzer_analyze(context: Any) -> None: + from cleveragents.domain.models.acms.analyzers import AnalyzerProtocol + + class _Bare(AnalyzerProtocol): # type: ignore[misc] + pass + + bare = _Bare() # type: ignore[abstract] + try: + bare.analyze("content", "uko://r") + assert False, "Expected NotImplementedError" # noqa: B011 + except NotImplementedError: + pass + + +# --------------------------------------------------------------------------- +# UKOTriple validation edge cases +# --------------------------------------------------------------------------- + + +@then("idx creating UKOTriple with whitespace-only subject_uri should raise ValueError") +def step_triple_ws_subject(context: Any) -> None: + try: + UKOTriple( + subject_uri=" ", + predicate="p", + object_uri="uko://o", + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx creating UKOTriple with whitespace-only predicate should raise ValueError") +def step_triple_ws_predicate(context: Any) -> None: + try: + UKOTriple( + subject_uri="uko://s", + predicate=" ", + object_uri="uko://o", + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then( + "idx creating UKOTriple with empty object_uri and empty object_value" + " should raise ValueError" +) +def step_triple_no_object(context: Any) -> None: + try: + UKOTriple( + subject_uri="uko://s", + predicate="rdf:type", + object_uri="", + object_value="", + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +# --------------------------------------------------------------------------- +# fire_on_removed direct call +# --------------------------------------------------------------------------- + + +@then("idx calling fire_on_removed with a raising hook should not raise") +def step_fire_on_removed_direct(context: Any) -> None: + from cleveragents.application.services.uko_indexer_internals import ( + fire_on_removed, + ) + + class _RaisingHook: + def on_indexed(self, result: IndexResult) -> None: + pass + + def on_removed(self, resource_id: str, project: str) -> None: + raise RuntimeError("boom") + + def on_error(self, resource_id: str, error: str) -> None: + pass + + # Should not raise — fire_on_removed swallows the exception + fire_on_removed( + _RaisingHook(), # type: ignore[arg-type] + ULID_1, + "local/test", + ) + + +# --------------------------------------------------------------------------- +# AnalyzerRegistry validation and listing steps +# --------------------------------------------------------------------------- + + +@then("idx registering a non-protocol object should raise TypeError") +def step_registry_non_protocol(context: Any) -> None: + from cleveragents.domain.models.acms.analyzers import AnalyzerRegistry + + registry = AnalyzerRegistry() + try: + registry.register("not_an_analyzer") # type: ignore[arg-type] + assert False, "Expected TypeError" # noqa: B011 + except TypeError: + pass + + +@then("idx registering an analyzer with empty extensions should raise ValueError") +def step_registry_empty_extensions(context: Any) -> None: + from cleveragents.domain.models.acms.analyzers import AnalyzerRegistry + + class _NoExtensions: + supported_extensions: frozenset[str] = frozenset() + domain = "empty" + + def analyze(self, content: str, resource_uri: str) -> list[UKOTriple]: + return [] + + registry = AnalyzerRegistry() + try: + registry.register(_NoExtensions()) # type: ignore[arg-type] + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx looking up empty extension should return None") +def step_registry_empty_ext(context: Any) -> None: + assert context.idx_registry.get_for_extension("") is None + + +@then('idx list_extensions should include ".py"') +def step_registry_list_ext(context: Any) -> None: + exts = context.idx_registry.list_extensions() + assert ".py" in exts, f".py not in {exts}" + + +@then("idx list_analyzers should return {n:d} analyzer") +def step_registry_list_analyzers(context: Any, n: int) -> None: + analyzers = context.idx_registry.list_analyzers() + assert len(analyzers) == n, f"Expected {n} analyzers, got {len(analyzers)}" diff --git a/features/steps/uko_indexer_watcher_steps.py b/features/steps/uko_indexer_watcher_steps.py new file mode 100644 index 000000000..3c4f73882 --- /dev/null +++ b/features/steps/uko_indexer_watcher_steps.py @@ -0,0 +1,486 @@ +"""ResourceFileWatcher steps for UKO Indexer behave tests.""" + +from __future__ import annotations + +import shutil +import tempfile +import threading +import time +from pathlib import Path +from typing import Any + +from behave import given, then, when # type: ignore[import-untyped] +from watchdog.events import ( + DirModifiedEvent, + FileCreatedEvent, + FileDeletedEvent, + FileModifiedEvent, + FileMovedEvent, +) + +from cleveragents.application.services.resource_file_watcher import ( + FileChangeType, + ResourceFileWatcher, +) +from cleveragents.infrastructure.events.types import EventType +from features.mocks.uko_indexer_mocks import TrackingEventBus + +__all__: list[str] = [] + + +def _cleanup_watcher(context: Any) -> None: + """Stop the file watcher if running (behave cleanup hook).""" + watcher = getattr(context, "fw_watcher", None) + if watcher is not None and watcher.is_running: + watcher.stop() + tmpdir = getattr(context, "fw_tmpdir_obj", None) + if tmpdir is not None: + tmpdir.cleanup() + + +@given("idx a ResourceFileWatcher with default settings") +def step_fw_default(context: Any) -> None: + context.fw_watcher = ResourceFileWatcher() + context.add_cleanup(_cleanup_watcher, context) + + +@given("idx a ResourceFileWatcher with callback tracking and debounce {secs}") +def step_fw_tracking(context: Any, secs: str) -> None: + context.fw_changes = [] # list of (resource_id, project, change_type) + context.fw_callback_event = threading.Event() + + def _on_change(rid: str, proj: str, ct: FileChangeType) -> None: + context.fw_changes.append((rid, proj, ct)) + context.fw_callback_event.set() + + context.fw_watcher = ResourceFileWatcher( + on_change=_on_change, + debounce_seconds=float(secs), + ) + context.add_cleanup(_cleanup_watcher, context) + + +@given("idx a ResourceFileWatcher with EventBus tracking and debounce {secs}") +def step_fw_eventbus(context: Any, secs: str) -> None: + context.fw_event_bus = TrackingEventBus() + context.fw_watcher = ResourceFileWatcher( + event_bus=context.fw_event_bus, + debounce_seconds=float(secs), + ) + context.add_cleanup(_cleanup_watcher, context) + + +@given("idx a ResourceFileWatcher with a failing callback and debounce {secs}") +def step_fw_failing_cb(context: Any, secs: str) -> None: + def _boom(rid: str, proj: str, ct: FileChangeType) -> None: + _ = rid, proj, ct + msg = "intentional test failure" + raise RuntimeError(msg) + + context.fw_watcher = ResourceFileWatcher( + on_change=_boom, + debounce_seconds=float(secs), + ) + context.add_cleanup(_cleanup_watcher, context) + + +@given('idx a temporary watched file "{name}" with content "{content}"') +def step_fw_temp_file(context: Any, name: str, content: str) -> None: + context.fw_tmpdir_obj = tempfile.TemporaryDirectory() + tmpdir = Path(context.fw_tmpdir_obj.name) + fpath = tmpdir / name + fpath.write_text(content) + context.fw_temp_file = fpath + + +@given("idx a temporary watched directory") +def step_fw_temp_dir(context: Any) -> None: + context.fw_tmpdir_obj = tempfile.TemporaryDirectory() + context.fw_temp_dir = Path(context.fw_tmpdir_obj.name) + + +@when("idx the file watcher is started") +@given("idx the file watcher is started") +def step_fw_start(context: Any) -> None: + context.fw_watcher.start() + + +@when("idx the file watcher is stopped") +def step_fw_stop(context: Any) -> None: + context.fw_watcher.stop() + + +@then("idx the file watcher should be running") +def step_fw_running(context: Any) -> None: + assert context.fw_watcher.is_running, "Watcher should be running" + + +@then("idx the file watcher should not be running") +def step_fw_not_running(context: Any) -> None: + assert not context.fw_watcher.is_running, "Watcher should not be running" + + +@when('idx the watcher registers the file for resource "{rid}" project "{proj}"') +@given('idx the watcher registers the file for resource "{rid}" project "{proj}"') +def step_fw_register(context: Any, rid: str, proj: str) -> None: + context.fw_watcher.watch( + context.fw_temp_file, + resource_id=rid, + project=proj, + ) + + +@when("idx the watcher unregisters the watched file") +def step_fw_unregister(context: Any) -> None: + context.fw_watcher.unwatch(context.fw_temp_file) + + +@then("idx the watcher should have {n:d} watched paths") +def step_fw_count(context: Any, n: int) -> None: + actual = len(context.fw_watcher.watched_paths) + assert actual == n, f"Expected {n} watched paths, got {actual}" + + +@when("idx a FileModifiedEvent is simulated for the watched file") +def step_fw_sim_modified(context: Any) -> None: + event = FileModifiedEvent(str(context.fw_temp_file.resolve())) + context.fw_watcher._handle_fs_event(event) + + +@when("idx a FileDeletedEvent is simulated for the watched file") +def step_fw_sim_deleted(context: Any) -> None: + event = FileDeletedEvent(str(context.fw_temp_file.resolve())) + context.fw_watcher._handle_fs_event(event) + + +@when("idx 5 FileModifiedEvents are simulated rapidly for the watched file") +def step_fw_sim_rapid(context: Any) -> None: + for _ in range(5): + event = FileModifiedEvent(str(context.fw_temp_file.resolve())) + context.fw_watcher._handle_fs_event(event) + + +@when("idx we wait {secs} seconds for debounce") +def step_fw_wait(context: Any, secs: str) -> None: + time.sleep(float(secs)) + + +@then("idx the on_change callback should fire within {secs:d} seconds") +def step_fw_cb_wait(context: Any, secs: int) -> None: + fired = context.fw_callback_event.wait(timeout=float(secs)) + assert fired, f"Callback did not fire within {secs}s" + + +@then('idx the callback should receive resource "{rid}" project "{proj}" change "{ct}"') +def step_fw_cb_check(context: Any, rid: str, proj: str, ct: str) -> None: + assert len(context.fw_changes) > 0, "No callback invocations recorded" + last = context.fw_changes[-1] + assert last[0] == rid, f"resource_id: expected {rid}, got {last[0]}" + assert last[1] == proj, f"project: expected {proj}, got {last[1]}" + assert last[2] == ct, f"change_type: expected {ct}, got {last[2]}" + + +@then("idx the on_change callback should have fired exactly {n:d} time") +@then("idx the on_change callback should have fired exactly {n:d} times") +def step_fw_cb_count(context: Any, n: int) -> None: + actual = len(context.fw_changes) + assert actual == n, f"Expected {n} callback invocations, got {actual}" + + +@then( + "idx the EventBus should receive a RESOURCE_MODIFIED event within {secs:d} seconds" +) +def step_fw_eb_wait(context: Any, secs: int) -> None: + received = context.fw_event_bus.wait(timeout=float(secs)) + assert received, f"No EventBus event within {secs}s" + assert len(context.fw_event_bus.events) > 0 + assert context.fw_event_bus.events[-1].event_type == EventType.RESOURCE_MODIFIED + + +@then('idx the event details should contain resource_id "{rid}"') +def step_fw_eb_detail(context: Any, rid: str) -> None: + last = context.fw_event_bus.events[-1] + assert last.details.get("resource_id") == rid + + +@then('idx watching path "{path}" should raise ValueError containing "{substr}"') +def step_fw_watch_bad_path(context: Any, path: str, substr: str) -> None: + try: + context.fw_watcher.watch( + Path(path), + resource_id="r", + project="p", + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert substr in str(exc), f"'{substr}' not in '{exc}'" + + +@then( + 'idx watching the temporary directory should raise ValueError containing "{substr}"' +) +def step_fw_watch_dir(context: Any, substr: str) -> None: + try: + context.fw_watcher.watch( + context.fw_temp_dir, + resource_id="r", + project="p", + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError as exc: + assert substr in str(exc), f"'{substr}' not in '{exc}'" + + +@then("idx creating ResourceFileWatcher with debounce -1.0 should raise ValueError") +def step_fw_bad_debounce(context: Any) -> None: + try: + ResourceFileWatcher(debounce_seconds=-1.0) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@when('idx a FileModifiedEvent is simulated for path "{path}"') +def step_fw_sim_unwatched(context: Any, path: str) -> None: + event = FileModifiedEvent(path) + context.fw_watcher._handle_fs_event(event) + + +@when("idx a directory event is simulated for the watched path") +def step_fw_sim_dir_event(context: Any) -> None: + event = DirModifiedEvent(str(context.fw_temp_file.resolve())) + context.fw_watcher._handle_fs_event(event) + + +@then("idx watching with empty resource_id should raise ValueError") +def step_fw_watch_empty_rid(context: Any) -> None: + try: + context.fw_watcher.watch( + context.fw_temp_file, + resource_id="", + project="p", + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx watching with empty project should raise ValueError") +def step_fw_watch_empty_project(context: Any) -> None: + try: + context.fw_watcher.watch( + context.fw_temp_file, + resource_id="r", + project="", + ) + assert False, "Expected ValueError" # noqa: B011 + except ValueError: + pass + + +@then("idx starting the watcher again should raise RuntimeError") +def step_fw_double_start(context: Any) -> None: + try: + context.fw_watcher.start() + assert False, "Expected RuntimeError" # noqa: B011 + except RuntimeError: + pass + + +@when("idx I stop the file watcher") +@then("idx I stop the file watcher") +def step_fw_explicit_stop(context: Any) -> None: + context.fw_watcher.stop() + + +@then("idx the debounce_seconds should be {value}") +def step_fw_debounce_prop(context: Any, value: str) -> None: + assert context.fw_watcher.debounce_seconds == float(value), ( + f"Expected {value}, got {context.fw_watcher.debounce_seconds}" + ) + + +@when("idx a FileCreatedEvent is simulated for the watched file") +def step_fw_sim_created(context: Any) -> None: + event = FileCreatedEvent(str(context.fw_temp_file.resolve())) + context.fw_watcher._handle_fs_event(event) + + +@when("idx a FileMovedEvent is simulated for the watched file") +def step_fw_sim_moved(context: Any) -> None: + event = FileMovedEvent( + str(context.fw_temp_file.resolve()), + str(context.fw_temp_file.resolve()) + ".bak", + ) + context.fw_watcher._handle_fs_event(event) + + +@given("idx a ResourceFileWatcher with auto_reindex disabled") +def step_fw_auto_reindex_disabled(context: Any) -> None: + context.fw_watcher = ResourceFileWatcher(auto_reindex=False) + context.add_cleanup(_cleanup_watcher, context) + + +@then("idx the auto_reindex property should be false") +def step_fw_auto_reindex_prop(context: Any) -> None: + assert context.fw_watcher.auto_reindex is False + + +@given("idx a ResourceFileWatcher with a directly-failing callback and debounce {secs}") +def step_fw_direct_fail_cb(context: Any, secs: str) -> None: + def _boom(rid: str, proj: str, ct: FileChangeType) -> None: + _ = rid, proj, ct + raise RuntimeError("intentional") + + context.fw_watcher = ResourceFileWatcher( + on_change=_boom, + debounce_seconds=float(secs), + ) + context.add_cleanup(_cleanup_watcher, context) + + +@when("idx _fire_change is called directly with failing callback") +def step_fw_fire_change_fail_cb(context: Any) -> None: + path = str(context.fw_temp_file.resolve()) + context.fw_watcher._fire_change( + path, + "res-001", + "local/test", + FileChangeType.MODIFIED, + ) + + +@given("idx a ResourceFileWatcher with a failing event bus and debounce {secs}") +def step_fw_failing_event_bus(context: Any, secs: str) -> None: + class _FailingBus: + def emit(self, event: object) -> None: + raise RuntimeError("bus boom") + + def subscribe(self, event_type: object, handler: object) -> None: + pass + + context.fw_watcher = ResourceFileWatcher( + event_bus=_FailingBus(), # type: ignore[arg-type] + debounce_seconds=float(secs), + ) + context.add_cleanup(_cleanup_watcher, context) + + +@when("idx _fire_change is called directly with failing event bus") +def step_fw_fire_change_fail_bus(context: Any) -> None: + path = str(context.fw_temp_file.resolve()) + context.fw_watcher._fire_change( + path, + "res-001", + "local/test", + FileChangeType.MODIFIED, + ) + + +@when("idx _fire_change is called directly after stop") +def step_fw_fire_change_after_stop(context: Any) -> None: + path = str(context.fw_temp_file.resolve()) + context.fw_watcher._fire_change( + path, + "res-001", + "local/test", + FileChangeType.MODIFIED, + ) + + +@when("idx a _ResourceChangeHandler is used to route events") +def step_fw_handler_routes(context: Any) -> None: + from cleveragents.application.services.resource_file_watcher import ( + _ResourceChangeHandler, + ) + + handler = _ResourceChangeHandler(context.fw_watcher) + path = str(context.fw_temp_file.resolve()) + handler.on_modified(FileModifiedEvent(path)) + handler.on_created(FileCreatedEvent(path)) + handler.on_deleted(FileDeletedEvent(path)) + handler.on_moved(FileMovedEvent(path, path + ".bak")) + + +@then("idx file watcher should fire both callback and event_bus on change") +def step_callback_and_event_bus_combo(context: Any) -> None: + td = tempfile.mkdtemp() + try: + fp = Path(td) / "combo.py" + fp.write_text("x=1\n") + chs: list[str] = [] + ev = threading.Event() + uid = "01HQ8ZDRX50000000000000001" + bus = TrackingEventBus() + + def cb(r: str, _p: str, _ct: FileChangeType) -> None: + chs.append(r) + ev.set() + + w = ResourceFileWatcher( + on_change=cb, + event_bus=bus, + debounce_seconds=0.01, + ) + w.start() + w.watch(fp, resource_id=uid, project="local/test") + w._handle_fs_event(FileModifiedEvent(str(fp.resolve()))) + assert ev.wait(timeout=2.0) and chs[-1] == uid + assert bus.wait(timeout=2.0) and len(bus.events) >= 1 + w.stop() + finally: + shutil.rmtree(td, ignore_errors=True) + + +@then("idx file watcher should schedule new dir on cross-directory move") +def step_fw_cross_dir_move(context: Any) -> None: + ev = threading.Event() + + def cb(rid: str, _p: str, _ct: str) -> None: + ev.set() + + td = tempfile.mkdtemp() + try: + sub = Path(td) / "sub" + sub.mkdir() + fp = Path(td) / "orig.py" + fp.write_text("x=1\n", encoding="utf-8") + dest = sub / "moved.py" + w = ResourceFileWatcher(on_change=cb, debounce_seconds=0.05) + w.watch(fp, resource_id="01HQ8ZDRX50000000000000077", project="local/t") + w.start() + w._handle_fs_event(FileMovedEvent(str(fp.resolve()), str(dest.resolve()))) + assert ev.wait(timeout=2.0) + assert str(sub.resolve()) in w._dir_watches + w.stop() + finally: + shutil.rmtree(td, ignore_errors=True) + + +@then("idx file watcher should handle move then modify sequence") +def step_fw_move_then_modify(context: Any) -> None: + chs: list[str] = [] + ev = threading.Event() + uid = "01HQ8ZDRX50000000000000099" + + def cb(rid: str, _p: str, _ct: str) -> None: + chs.append(rid) + ev.set() + + td = tempfile.mkdtemp() + try: + fp = Path(td) / "src.py" + fp.write_text("x=1\n", encoding="utf-8") + dp = str((Path(td) / "dst.py").resolve()) + w = ResourceFileWatcher(on_change=cb, debounce_seconds=0.05) + w.watch(fp, resource_id=uid, project="local/test") + w.start() + w._handle_fs_event(FileMovedEvent(str(fp.resolve()), dp)) + assert ev.wait(timeout=2.0), "Move cb" + ev.clear() + w._handle_fs_event(FileModifiedEvent(dp)) + assert ev.wait(timeout=2.0), "Modify cb" + w.stop() + assert len(chs) >= 2 and chs[0] == uid and chs[1] == uid + finally: + shutil.rmtree(td, ignore_errors=True) diff --git a/features/uko_indexer.feature b/features/uko_indexer.feature new file mode 100644 index 000000000..5ed460743 --- /dev/null +++ b/features/uko_indexer.feature @@ -0,0 +1,1083 @@ +Feature: UKO Indexer — Real-time Index Synchronization + As a context pipeline consumer + I need the UKOIndexer to analyze resources and index into backends + So that context strategies can query pre-built indices + + Background: + Given idx a clean analyzer registry + And idx in-memory index backends + + # ================================================================= + # Index backend protocols and stubs + # ================================================================= + + Scenario: InMemoryTextIndexBackend indexes and searches documents + When idx I index a text document "doc1" with content "hello world" in project "local/app" + Then idx searching "hello" in project "local/app" should return 1 result + And idx the search result doc_id should be "doc1" + + Scenario: InMemoryTextIndexBackend returns empty for non-matching query + When idx I index a text document "doc1" with content "hello world" in project "local/app" + Then idx searching "foobar" in project "local/app" should return 0 results + + Scenario: InMemoryTextIndexBackend scopes by project + When idx I index a text document "doc1" with content "hello world" in project "local/app" + Then idx searching "hello" in project "local/other" should return 0 results + + Scenario: InMemoryTextIndexBackend removes documents + When idx I index a text document "doc1" with content "hello world" in project "local/app" + And idx I remove text document "doc1" from project "local/app" + Then idx searching "hello" in project "local/app" should return 0 results + + Scenario: InMemoryTextIndexBackend rebuilds index + When idx I index a text document "doc1" with content "hello world" in project "local/app" + And idx I rebuild text index for project "local/app" + Then idx searching "hello" in project "local/app" should return 0 results + + Scenario: InMemoryTextIndexBackend rejects empty project + Then idx indexing a text document with empty project should raise ValueError + + Scenario: InMemoryTextIndexBackend rejects empty content + Then idx indexing a text document with empty content should raise ValueError + + Scenario: InMemoryTextIndexBackend rejects empty query + Then idx searching with empty query should raise ValueError + + Scenario: InMemoryTextIndexBackend rejects non-positive limit + Then idx searching with limit 0 should raise ValueError + + Scenario: InMemoryVectorIndexBackend indexes and searches embeddings + When idx I index an embedding "emb1" with vector "1.0,2.0,3.0" in project "local/app" + Then idx searching similar in project "local/app" should return 1 result + + Scenario: InMemoryVectorIndexBackend removes embeddings + When idx I index an embedding "emb1" with vector "1.0,2.0,3.0" in project "local/app" + And idx I remove embedding "emb1" from project "local/app" + Then idx searching similar in project "local/app" should return 0 results + + Scenario: InMemoryVectorIndexBackend rejects out-of-range min_relevance + When idx I index an embedding "emb1" with vector "1.0,2.0,3.0" in project "local/app" + Then idx searching similar with min_relevance 1.1 should raise ValueError + + Scenario: InMemoryVectorIndexBackend rejects empty embedding + Then idx indexing an embedding with empty vector should raise ValueError + + Scenario: InMemoryGraphIndexBackend adds and queries triples + When idx I add a triple "uko://s" "uko:type" "uko://Class" in project "local/app" + Then idx querying project "local/app" should return 1 binding + And idx the binding should have subject "uko://s" + + Scenario: InMemoryGraphIndexBackend removes triples by subject + When idx I add a triple "uko://s" "uko:type" "uko://Class" in project "local/app" + And idx I add a triple "uko://s" "uko:name" "Foo" in project "local/app" + And idx I remove triples with subject "uko://s" from project "local/app" + Then idx querying project "local/app" should return 0 bindings + + Scenario: InMemoryGraphIndexBackend rejects all-None pattern + Then idx removing triples with all-None pattern should raise ValueError + + Scenario: InMemoryGraphIndexBackend rejects empty project + Then idx adding a triple with empty project should raise ValueError + + # ================================================================= + # Provenance model + # ================================================================= + + Scenario: ProvenanceMetadata has required fields + When idx I create provenance for resource "01HQ8ZDRX50000000000000001" + Then idx the provenance should have source_resource "01HQ8ZDRX50000000000000001" + And idx the provenance should be current + And idx the provenance should have a valid_from timestamp + + Scenario: ProvenanceMetadata rejects empty source_resource + Then idx creating provenance with empty source_resource should raise ValidationError + + Scenario: ProvenancedTriple wraps triple with provenance + When idx I create a provenanced triple for resource "01HQ8ZDRX50000000000000001" + Then idx the provenanced triple should have the triple + And idx the provenanced triple should have the provenance + + Scenario: ProvenanceMetadata is frozen + When idx I create provenance for resource "01HQ8ZDRX50000000000000001" + Then idx mutating the provenance should raise ValidationError + + Scenario: IndexResult tracks indexing summary + When idx I create an IndexResult with 5 triples and 1 text doc + Then idx the IndexResult should have triple_count 5 + And idx the IndexResult should have text_docs_indexed 1 + And idx the IndexResult should have analyzer_domain "python" + + Scenario: IndexResult rejects negative triple_count + Then idx creating IndexResult with negative triple_count should raise ValidationError + + # ================================================================= + # AnalyzerRegistry.get_for_resource + # ================================================================= + + Scenario: AnalyzerRegistry.get_for_resource finds analyzer by location extension + Given idx a PythonAnalyzer is registered + When idx I look up analyzer for a resource with location "src/main.py" + Then idx the analyzer domain should be "python" + + Scenario: AnalyzerRegistry.get_for_resource returns None for unknown extension + Given idx a PythonAnalyzer is registered + When idx I look up analyzer for a resource with location "data.csv" + Then idx no analyzer should be found + + Scenario: AnalyzerRegistry.get_for_resource returns None for no location + Given idx a PythonAnalyzer is registered + When idx I look up analyzer for a resource with no location + Then idx no analyzer should be found + + Scenario: AnalyzerRegistry.get_for_resource finds MarkdownAnalyzer + Given idx a MarkdownAnalyzer is registered + When idx I look up analyzer for a resource with location "README.md" + Then idx the analyzer domain should be "markdown" + + Scenario: AnalyzerRegistry higher-priority analyzer supersedes lower-priority + Given idx a clean analyzer registry + And idx a PythonAnalyzer is registered with priority 0 + And idx an alternative Python analyzer "python-advanced" is registered with priority 10 + When idx I look up analyzer for a resource with location "main.py" + Then idx the analyzer domain should be "python-advanced" + + Scenario: AnalyzerRegistry lower-priority analyzer does not supersede higher-priority + Given idx a clean analyzer registry + And idx an alternative Python analyzer "python-advanced" is registered with priority 10 + And idx a PythonAnalyzer is registered with priority 0 + When idx I look up analyzer for a resource with location "main.py" + Then idx the analyzer domain should be "python-advanced" + + # ================================================================= + # UKOIndexer core + # ================================================================= + + Scenario: UKOIndexer indexes a Python resource into all backends + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the index result should have triple_count greater than 0 + And idx the index result should have text_docs_indexed 1 + And idx the index result should have embeddings_indexed 1 + And idx the index result should have analyzer_domain "python" + And idx the index result should have 0 errors + And idx the graph backend should have triples for project "local/test" + And idx the text backend should have 1 document + And idx the vector backend should have 1 embedding + And idx the indexer should track 1 indexed resource + + Scenario: UKOIndexer indexes a Markdown resource + Given idx a MarkdownAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Markdown resource "01HQ8ZDRX50000000000000002" with content "# Title\nSome text" + Then idx the index result should have triple_count greater than 0 + And idx the index result should have analyzer_domain "markdown" + + Scenario: UKOIndexer returns none-domain when no analyzer matches + And idx a UKOIndexer with all backends + When idx I index a CSV resource "01HQ8ZDRX50000000000000003" with location "data.csv" + Then idx the index result should have analyzer_domain "none" + And idx the index result should have triple_count 0 + + Scenario: UKOIndexer degrades gracefully without text backend + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer without text backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "x = 1" + Then idx the index result should have text_docs_indexed 0 + And idx the index result should have 0 errors + + Scenario: UKOIndexer degrades gracefully without vector backend + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer without vector backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "x = 1" + Then idx the index result should have embeddings_indexed 0 + And idx the index result should have 0 errors + + Scenario: UKOIndexer degrades gracefully without text and vector backends + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with only graph backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "x = 1" + Then idx the index result should have text_docs_indexed 0 + And idx the index result should have embeddings_indexed 0 + And idx the index result should have triple_count greater than 0 + + Scenario: UKOIndexer reports content read errors without side effects + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with a failing content reader + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "x = 1" + Then idx the index result should have 1 errors + And idx the index result error should mention "read" + And idx the indexer should track 0 indexed resources + And idx the text backend should have 0 documents + And idx the vector backend should have 0 embeddings + + Scenario: UKOIndexer rejects empty project + Given idx a UKOIndexer with all backends + Then idx indexing with empty project should raise ValueError + + # ================================================================= + # UKOIndexer remove and reindex + # ================================================================= + + Scenario: UKOIndexer removes a resource from all backends + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I remove resource "01HQ8ZDRX50000000000000001" from project "local/test" + Then idx the graph backend should have 0 triples for project "local/test" + And idx the text backend should have 0 documents + And idx the vector backend should have 0 embeddings + And idx the indexer should track 0 indexed resources + + Scenario: UKOIndexer reindexes a resource + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I reindex resource "01HQ8ZDRX50000000000000001" with content "class Bar:\n pass" + Then idx the index result should have triple_count greater than 0 + And idx the indexer should track 1 indexed resource + + Scenario: UKOIndexer remove rejects empty resource_id + Given idx a UKOIndexer with all backends + Then idx removing with empty resource_id should raise ValueError + + Scenario: UKOIndexer remove rejects empty project + Given idx a UKOIndexer with all backends + Then idx removing with empty project should raise ValueError + + # ================================================================= + # Lifecycle hooks + # ================================================================= + + Scenario: DefaultLifecycleHook receives on_indexed event + Given idx a UKOIndexer with a tracking lifecycle hook + And idx a PythonAnalyzer is registered + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "x = 1" + Then idx the lifecycle hook should have received 1 on_indexed event + And idx the lifecycle hook should have received 0 on_error events + + Scenario: DefaultLifecycleHook receives on_removed event + Given idx a UKOIndexer with a tracking lifecycle hook + And idx a PythonAnalyzer is registered + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "x = 1" + And idx I remove resource "01HQ8ZDRX50000000000000001" from project "local/test" + Then idx the lifecycle hook should have received 1 on_removed event + + Scenario: Lifecycle hook receives on_error for read failure + Given idx a UKOIndexer with a tracking lifecycle hook and failing reader + And idx a PythonAnalyzer is registered + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "x = 1" + Then idx the lifecycle hook should have received 1 on_error events + + # ================================================================= + # UKOIndexer constructor validation + # ================================================================= + + Scenario: UKOIndexer rejects invalid analyzer_registry type + Then idx creating UKOIndexer with invalid registry should raise TypeError + + Scenario: UKOIndexer has_text_backend property + Given idx a UKOIndexer with all backends + Then idx the indexer should have text backend + And idx the indexer should have vector backend + + Scenario: UKOIndexer without optional backends reports availability + Given idx a UKOIndexer with only graph backend + Then idx the indexer should not have text backend + And idx the indexer should not have vector backend + + # ================================================================= + # SearchResult and IndexedDocument models + # ================================================================= + + Scenario: SearchResult validates score range + Then idx creating SearchResult with score 1.5 should raise ValueError + + Scenario: SearchResult rejects empty doc_id + Then idx creating SearchResult with empty doc_id should raise ValueError + + Scenario: IndexedDocument validates non-negative char_count + Then idx creating IndexedDocument with negative char_count should raise ValueError + + Scenario: IndexedDocument rejects empty project + Then idx creating IndexedDocument with empty project should raise ValueError + + # ================================================================= + # Review-discovered edge cases + # ================================================================= + + Scenario: UKOIndexer double-index is idempotent (no stale accumulation) + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the indexer should track 1 indexed resource + And idx the text backend should have 1 document + And idx the vector backend should have 1 embedding + + Scenario: UKOIndexer reindex replaces old triples with new + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I reindex resource "01HQ8ZDRX50000000000000001" with content "class Bar:\n pass" + Then idx the indexer should track 1 indexed resource + And idx the graph backend should contain triple with predicate "rdfs:label" and object "Bar" + And idx the graph backend should not contain triple with object "Foo" + + Scenario: UKOIndexer remove of never-indexed resource is no-op + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I remove resource "01HQ8ZDRX50000000000000099" from project "local/test" + Then idx the indexer should track 0 indexed resources + + Scenario: ProvenancedTriple has correct source_resource value + Given idx a PythonAnalyzer is registered + And idx an AnalyzerRegistry with the PythonAnalyzer + When idx I analyze content "x = 1" with resource URI "uko://resource/01HQ8ZDRX50000000000000001" + And idx I wrap the first triple with provenance for resource "01HQ8ZDRX50000000000000001" at path "src/main.py" + Then idx the provenance source_resource should be "01HQ8ZDRX50000000000000001" + And idx the provenance source_path should be "src/main.py" + + # ================================================================= + # Bug-fix verifications (deep review round 2) + # ================================================================= + + Scenario: Idempotent re-index does NOT fire on_removed lifecycle event + Given idx a UKOIndexer with a tracking lifecycle hook + And idx a PythonAnalyzer is registered + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Bar:\n pass" + Then idx the lifecycle hook should have received 0 on_removed events + And idx the lifecycle hook should have received 2 on_indexed events + + Scenario: Cross-project re-index cleans up old project data + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index Python resource "01HQ8ZDRX50000000000000001" under project "org/old" content "class Foo:\n pass" + And idx I index Python resource "01HQ8ZDRX50000000000000001" under project "org/new" content "class Bar:\n pass" + Then idx the graph backend should have 0 triples for project "org/old" + And idx the graph backend should have triples for project "org/new" + And idx the text backend should have 1 document + And idx the indexer should track 1 indexed resource + + Scenario: Reindex with different project cleans old project + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index Python resource "01HQ8ZDRX50000000000000001" under project "org/old" content "class Foo:\n pass" + And idx I reindex resource "01HQ8ZDRX50000000000000001" under project "org/new" content "class Bar:\n pass" + Then idx the graph backend should have 0 triples for project "org/old" + And idx the graph backend should have triples for project "org/new" + And idx the indexer should track 1 indexed resource + + Scenario: Whitespace-only project is rejected by text backend + Then idx indexing a text document with whitespace-only project should raise ValueError + + Scenario: Whitespace-only content is rejected by text backend + Then idx indexing a text document with whitespace-only content should raise ValueError + + Scenario: Whitespace-only subject is rejected by graph backend + Then idx adding a triple with whitespace-only subject should raise ValueError + + # ================================================================= + # Backend error path scenarios (graceful degradation) + # ================================================================= + + Scenario: UKOIndexer handles graph backend error gracefully + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with a failing graph backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the index result should have triple_count 0 + And idx the index result errors should mention "Graph backend" + + Scenario: UKOIndexer handles text backend error gracefully + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with a failing text backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the index result should have text_docs_indexed 0 + And idx the index result errors should mention "Text backend" + And idx the index result should have triple_count greater than 0 + + Scenario: UKOIndexer handles vector backend error gracefully + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with a failing vector backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the index result should have embeddings_indexed 0 + And idx the index result errors should mention "Vector backend" + And idx the index result should have triple_count greater than 0 + + # ================================================================= + # Strengthened content assertions + # ================================================================= + + Scenario: Text backend stores correct doc_id and content after indexing + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the text backend should contain doc_id "uko://resource/01HQ8ZDRX50000000000000001" in project "local/test" + + Scenario: Vector backend stores correct doc_id after indexing + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the vector backend should contain doc_id "uko://resource/01HQ8ZDRX50000000000000001" in project "local/test" + + Scenario: Graph backend stores sourceResource provenance link + Given idx a PythonAnalyzer is registered + And idx a UKOIndexer with all backends + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + Then idx the graph backend should contain triple with predicate "uko:sourceResource" and object "uko://resource/01HQ8ZDRX50000000000000001" + + # ================================================================= + # Additional negative validation tests + # ================================================================= + + Scenario: IndexResult rejects empty resource_id + Then idx creating IndexResult with empty resource_id should raise ValidationError + + Scenario: SearchResult rejects negative score + Then idx creating SearchResult with score -0.1 should raise ValueError + + Scenario: Reindex rejects empty project + Given idx a UKOIndexer with all backends + Then idx reindexing with empty project should raise ValueError + + Scenario: Reindex rejects whitespace-only project + Given idx a UKOIndexer with all backends + Then idx reindexing with whitespace-only project should raise ValueError + + # ================================================================= + # LocationContentReader tests (P2-5) + # ================================================================= + + Scenario: LocationContentReader rejects path with dotdot components + Then idx LocationContentReader should reject path with dotdot component + + Scenario: LocationContentReader rejects resource with no location + Then idx LocationContentReader should reject resource with no location + + Scenario: LocationContentReader rejects non-positive max_content_size + Then idx LocationContentReader should reject max_content_size of 0 + + Scenario: LocationContentReader reads a real file + Then idx LocationContentReader should read a real file successfully + + Scenario: LocationContentReader rejects non-existent file + Then idx LocationContentReader should reject non-existent file + + Scenario: LocationContentReader rejects path escaping base_dir + Then idx LocationContentReader should reject path escaping base_dir + + Scenario: SearchResult rejects NaN score + Then idx creating SearchResult with NaN score should raise ValueError + + # ================================================================= + # File-watching integration (P1-1: file-watching acceptance criterion) + # ================================================================= + + @file-watching + Scenario: ResourceFileWatcher starts and stops cleanly + Given idx a ResourceFileWatcher with default settings + When idx the file watcher is started + Then idx the file watcher should be running + When idx the file watcher is stopped + Then idx the file watcher should not be running + + @file-watching + Scenario: Watching a file path registers it for monitoring + Given idx a ResourceFileWatcher with default settings + And idx a temporary watched file "test.py" with content "x = 1" + When idx the watcher registers the file for resource "res-001" project "local/test" + Then idx the watcher should have 1 watched paths + + @file-watching + Scenario: Unwatching a file path removes it from monitoring + Given idx a ResourceFileWatcher with default settings + And idx a temporary watched file "test.py" with content "x = 1" + When idx the watcher registers the file for resource "res-001" project "local/test" + And idx the watcher unregisters the watched file + Then idx the watcher should have 0 watched paths + + @file-watching + Scenario: File modification triggers on_change callback + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a FileModifiedEvent is simulated for the watched file + Then idx the on_change callback should fire within 2 seconds + And idx the callback should receive resource "res-001" project "local/test" change "modified" + + @file-watching + Scenario: File deletion triggers on_change callback + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a FileDeletedEvent is simulated for the watched file + Then idx the on_change callback should fire within 2 seconds + And idx the callback should receive resource "res-001" project "local/test" change "deleted" + + @file-watching + Scenario: Debouncing coalesces rapid changes into one callback + Given idx a ResourceFileWatcher with callback tracking and debounce 0.1 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx 5 FileModifiedEvents are simulated rapidly for the watched file + Then idx the on_change callback should fire within 2 seconds + And idx the on_change callback should have fired exactly 1 time + + @file-watching + Scenario: EventBus receives RESOURCE_MODIFIED event on file change + Given idx a ResourceFileWatcher with EventBus tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a FileModifiedEvent is simulated for the watched file + Then idx the EventBus should receive a RESOURCE_MODIFIED event within 2 seconds + And idx the event details should contain resource_id "res-001" + + @file-watching + Scenario: Callback errors do not crash the watcher + Given idx a ResourceFileWatcher with a failing callback and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a FileModifiedEvent is simulated for the watched file + And idx we wait 0.2 seconds for debounce + Then idx the file watcher should be running + + @file-watching + Scenario: Watching a non-existent path raises ValueError + Given idx a ResourceFileWatcher with default settings + Then idx watching path "/nonexistent/path.py" should raise ValueError containing "does not exist" + + @file-watching + Scenario: Watching a directory path raises ValueError + Given idx a ResourceFileWatcher with default settings + And idx a temporary watched directory + Then idx watching the temporary directory should raise ValueError containing "not a file" + + @file-watching + Scenario: Negative debounce_seconds raises ValueError + Then idx creating ResourceFileWatcher with debounce -1.0 should raise ValueError + + @file-watching + Scenario: Unwatched file changes are ignored + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx the file watcher is started + When idx a FileModifiedEvent is simulated for path "/tmp/unwatched.py" + And idx we wait 0.2 seconds for debounce + Then idx the on_change callback should have fired exactly 0 times + + @file-watching + Scenario: Directory events are ignored by the watcher + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a directory event is simulated for the watched path + And idx we wait 0.2 seconds for debounce + Then idx the on_change callback should have fired exactly 0 times + + # ================================================================= + # Coverage — file-watcher validation and edge cases + # ================================================================= + + @file-watching + Scenario: Watch rejects empty resource_id + Given idx a ResourceFileWatcher with default settings + And idx a temporary watched file "test.py" with content "x = 1" + Then idx watching with empty resource_id should raise ValueError + + @file-watching + Scenario: Watch rejects empty project + Given idx a ResourceFileWatcher with default settings + And idx a temporary watched file "test.py" with content "x = 1" + Then idx watching with empty project should raise ValueError + + @file-watching + Scenario: Starting an already-running watcher raises RuntimeError + Given idx a ResourceFileWatcher with default settings + And idx the file watcher is started + Then idx starting the watcher again should raise RuntimeError + And idx I stop the file watcher + + @file-watching + Scenario: Stopping a non-running watcher is a no-op + Given idx a ResourceFileWatcher with default settings + When idx I stop the file watcher + Then idx the file watcher should not be running + + @file-watching + Scenario: debounce_seconds property returns configured value + Given idx a ResourceFileWatcher with callback tracking and debounce 0.25 + Then idx the debounce_seconds should be 0.25 + + @file-watching + Scenario: File creation triggers on_change with created type + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a FileCreatedEvent is simulated for the watched file + Then idx the on_change callback should fire within 2 seconds + And idx the callback should receive resource "res-001" project "local/test" change "created" + + @file-watching + Scenario: File move triggers on_change with moved type + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a FileMovedEvent is simulated for the watched file + Then idx the on_change callback should fire within 2 seconds + And idx the callback should receive resource "res-001" project "local/test" change "moved" + + @file-watching + Scenario: Unwatch cancels pending debounce timer + Given idx a ResourceFileWatcher with callback tracking and debounce 5.0 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a FileModifiedEvent is simulated for the watched file + And idx the watcher unregisters the watched file + And idx we wait 0.2 seconds for debounce + Then idx the on_change callback should have fired exactly 0 times + + # ================================================================= + # Coverage — UKOIndexer validation and error paths + # ================================================================= + + Scenario: UKOIndexer rejects invalid graph_backend type + Then idx creating UKOIndexer with invalid graph_backend should raise TypeError + + Scenario: UKOIndexer analyzer_registry property returns the registry + Given idx a content reader with resource "01HQ8ZDRX50000000000000001" content "x = 1" + When idx I create an indexer + Then idx the indexer analyzer_registry should be the same registry + + Scenario: Lifecycle hook error on on_indexed does not crash indexer + Given idx a PythonAnalyzer is registered + And idx a content reader with resource "01HQ8ZDRX50000000000000001" content "x = 1\n" + And idx a lifecycle hook that raises on on_indexed + When idx I create an indexer with the failing hook + And idx I index resource "01HQ8ZDRX50000000000000001" in project "local/test" + Then idx the index result should have 0 errors + + Scenario: Lifecycle hook error on on_removed does not crash indexer + Given idx a content reader with resource "01HQ8ZDRX50000000000000001" content "x = 1\n" + And idx a lifecycle hook that raises on on_removed + When idx I create an indexer with the failing hook + And idx I index resource "01HQ8ZDRX50000000000000001" in project "local/test" + Then idx removing resource "01HQ8ZDRX50000000000000001" should not raise + + Scenario: Lifecycle hook error on on_error does not crash indexer + Given idx a PythonAnalyzer is registered + And idx a lifecycle hook that raises on on_error + And idx a content reader that raises on read + When idx I create an indexer with the failing hook + And idx I index resource "01HQ8ZDRX50000000000000001" in project "local/test" + Then idx the index result should have errors + + Scenario: Analyzer exception returns IndexResult with error + Given idx a content reader with resource "01HQ8ZDRX50000000000000001" content "x = 1\n" + And idx an analyzer that raises RuntimeError + When idx I create an indexer + And idx I index resource "01HQ8ZDRX50000000000000001" in project "local/test" + Then idx the index result should have errors + And idx the index result error should mention "Analyzer error" + + Scenario: max_triples caps analyzer output + Given idx a content reader with resource "01HQ8ZDRX50000000000000001" content "x = 1\n" + And idx an analyzer that produces 100 triples + When idx I create an indexer with max_triples 5 + And idx I index resource "01HQ8ZDRX50000000000000001" in project "local/test" + Then idx the index result triple_count should be at most 5 + + # ================================================================= + # Coverage — LocationContentReader edge cases + # ================================================================= + + Scenario: LocationContentReader rejects non-regular file (e.g. named pipe) + Then idx LocationContentReader should reject a non-regular file + + Scenario: LocationContentReader rejects content exceeding max size + Then idx LocationContentReader should reject content exceeding max size + + # ================================================================= + # Coverage — ResourceFileWatcher start with pre-registered paths + # ================================================================= + + @file-watching + Scenario: Starting watcher with pre-registered paths schedules them + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx the file watcher is started + Then idx the file watcher should be running + And idx the watcher should have 1 watched paths + + # ================================================================= + # Coverage — IndexedDocument validation + # ================================================================= + + Scenario: IndexedDocument rejects empty doc_id + Then idx creating IndexedDocument with empty doc_id should raise ValueError + + # ================================================================= + # Coverage — Protocol stub NotImplementedError + # ================================================================= + + Scenario: ContentReader protocol raises NotImplementedError + Then idx calling ContentReader.read_content directly should raise NotImplementedError + + Scenario: IndexLifecycleHook protocol raises NotImplementedError on on_indexed + Then idx calling IndexLifecycleHook.on_indexed directly should raise NotImplementedError + + Scenario: IndexLifecycleHook protocol raises NotImplementedError on on_removed + Then idx calling IndexLifecycleHook.on_removed directly should raise NotImplementedError + + Scenario: IndexLifecycleHook protocol raises NotImplementedError on on_error + Then idx calling IndexLifecycleHook.on_error directly should raise NotImplementedError + + Scenario: AnalyzerProtocol raises NotImplementedError on supported_extensions + Then idx calling AnalyzerProtocol.supported_extensions directly should raise NotImplementedError + + Scenario: AnalyzerProtocol raises NotImplementedError on domain + Then idx calling AnalyzerProtocol.domain directly should raise NotImplementedError + + Scenario: AnalyzerProtocol raises NotImplementedError on analyze + Then idx calling AnalyzerProtocol.analyze directly should raise NotImplementedError + + # ================================================================= + # Coverage — UKOTriple validation edge cases + # ================================================================= + + Scenario: UKOTriple rejects whitespace-only subject_uri + Then idx creating UKOTriple with whitespace-only subject_uri should raise ValueError + + Scenario: UKOTriple rejects whitespace-only predicate + Then idx creating UKOTriple with whitespace-only predicate should raise ValueError + + Scenario: UKOTriple rejects both object_uri and object_value empty + Then idx creating UKOTriple with empty object_uri and empty object_value should raise ValueError + + # ================================================================= + # Coverage — AnalyzerRegistry edge cases + # ================================================================= + + Scenario: AnalyzerRegistry.get_for_resource returns None for extensionless location + Given idx a PythonAnalyzer is registered + When idx I look up analyzer for a resource with location "Makefile" + Then idx no analyzer should be found + + # ================================================================= + # Coverage — AnalyzerRegistry validation and listing + # ================================================================= + + Scenario: AnalyzerRegistry.register rejects non-protocol object + Then idx registering a non-protocol object should raise TypeError + + Scenario: AnalyzerRegistry.register rejects analyzer with empty extensions + Then idx registering an analyzer with empty extensions should raise ValueError + + Scenario: AnalyzerRegistry.get_for_extension with empty string returns None + Then idx looking up empty extension should return None + + Scenario: AnalyzerRegistry.list_extensions returns registered extensions + Given idx a PythonAnalyzer is registered + Then idx list_extensions should include ".py" + + Scenario: AnalyzerRegistry.list_analyzers returns registered analyzers + Given idx a PythonAnalyzer is registered + Then idx list_analyzers should return 1 analyzer + + # ================================================================= + # Coverage — fire_on_removed direct call with raising hook + # ================================================================= + + Scenario: fire_on_removed swallows hook exception + Then idx calling fire_on_removed with a raising hook should not raise + + # ================================================================= + # Coverage — ResourceFileWatcher additional paths + # ================================================================= + + @file-watching + Scenario: ResourceFileWatcher with auto_reindex disabled does not start observer + Given idx a ResourceFileWatcher with auto_reindex disabled + When idx the file watcher is started + Then idx the file watcher should not be running + + @file-watching + Scenario: ResourceFileWatcher auto_reindex property reflects construction + Given idx a ResourceFileWatcher with auto_reindex disabled + Then idx the auto_reindex property should be false + + @file-watching + Scenario: _fire_change with failing callback logs error but does not crash + Given idx a ResourceFileWatcher with a directly-failing callback and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx _fire_change is called directly with failing callback + Then idx the file watcher should be running + + @file-watching + Scenario: _fire_change with failing event_bus logs error but does not crash + Given idx a ResourceFileWatcher with a failing event bus and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx _fire_change is called directly with failing event bus + Then idx the file watcher should be running + + @file-watching + Scenario: _fire_change after stop is a no-op + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx the file watcher is stopped + And idx _fire_change is called directly after stop + Then idx the on_change callback should have fired exactly 0 times + + @file-watching + Scenario: _ResourceChangeHandler routes events to watcher + Given idx a ResourceFileWatcher with callback tracking and debounce 0.05 + And idx a temporary watched file "test.py" with content "x = 1" + And idx the file watcher is started + And idx the watcher registers the file for resource "res-001" project "local/test" + When idx a _ResourceChangeHandler is used to route events + Then idx the on_change callback should fire within 2 seconds + + # ================================================================= + # Coverage — removal backend failure paths + # ================================================================= + + @coverage + Scenario: remove_resource tolerates graph backend failure + Given idx a clean analyzer registry + And idx in-memory index backends + And idx a PythonAnalyzer is registered + And idx a UKOIndexer with removal-failing graph backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I remove resource "01HQ8ZDRX50000000000000001" from project "local/test" + Then idx the indexer should track 0 indexed resources + + @coverage + Scenario: remove_resource tolerates text backend failure + Given idx a clean analyzer registry + And idx in-memory index backends + And idx a PythonAnalyzer is registered + And idx a UKOIndexer with removal-failing text backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I remove resource "01HQ8ZDRX50000000000000001" from project "local/test" + Then idx the indexer should track 0 indexed resources + + @coverage + Scenario: remove_resource tolerates vector backend failure + Given idx a clean analyzer registry + And idx in-memory index backends + And idx a PythonAnalyzer is registered + And idx a UKOIndexer with removal-failing vector backend + When idx I index a Python resource "01HQ8ZDRX50000000000000001" with content "class Foo:\n pass" + And idx I remove resource "01HQ8ZDRX50000000000000001" from project "local/test" + Then idx the indexer should track 0 indexed resources + + # ================================================================= + # Coverage — max_triples validation + # ================================================================= + + @coverage + Scenario: max_triples of zero raises ValueError + Then idx creating a UKOIndexer with max_triples 0 should raise ValueError + + # ================================================================= + # Coverage — reindex_resource on a fresh (not previously indexed) resource + # ================================================================= + + @coverage + Scenario: reindex_resource on a never-indexed resource succeeds + Given idx a clean analyzer registry + And idx in-memory index backends + And idx a PythonAnalyzer is registered + And idx a content reader with resource "01HQ8ZDRX50000000000000001" content "x = 1\n" + When idx I create an indexer + And idx I reindex resource "01HQ8ZDRX50000000000000001" in project "local/test" + Then idx the index result should have triple_count greater than 0 + + # ================================================================= + # Coverage — index_graph edge cases (internals) + # ================================================================= + + @coverage + Scenario: index_graph skips triple with no object_uri and no object_value + Then idx index_graph should skip triples with empty object + + @coverage + Scenario: index_graph handles rdfs:label failure for dual-object triple + Then idx index_graph should tolerate rdfs:label failure + + @coverage + Scenario: index_graph handles provenance link failure + Then idx index_graph should tolerate provenance link failure + + # ================================================================= + # Coverage — index_stubs edge cases + # ================================================================= + + @coverage + Scenario: InMemoryTextIndexBackend search respects limit + Then idx text search should stop at the limit + + @coverage + Scenario: InMemoryVectorIndexBackend rejects empty query_embedding + Then idx vector search_similar with empty embedding should raise ValueError + + @coverage + Scenario: InMemoryVectorIndexBackend rejects zero limit + Then idx vector search_similar with limit 0 should raise ValueError + + @coverage + Scenario: InMemoryVectorIndexBackend search filters by project + Then idx vector search_similar filters by project + + @coverage + Scenario: InMemoryGraphIndexBackend remove_triples on missing project is no-op + Then idx graph remove_triples on missing project should be no-op + + @coverage + Scenario: InMemoryGraphIndexBackend triple_count across all projects + Then idx graph triple_count without project should sum all projects + + # ================================================================= + # Coverage — index_backends protocol base methods + # ================================================================= + + @coverage + Scenario: Protocol base method bodies are exercisable + Then idx all index backend protocol base methods should be callable + + # ================================================================= + # Coverage — min_relevance filtering (P3-3) + # ================================================================= + + @coverage + Scenario: InMemoryVectorIndexBackend min_relevance above 1.0 raises ValueError + Then idx vector search_similar with min_relevance above 1.0 raises ValueError + + @coverage + Scenario: InMemoryVectorIndexBackend min_relevance 1.0 includes exact matches + Then idx vector search_similar with min_relevance 1.0 includes exact matches + + @coverage + Scenario: InMemoryVectorIndexBackend min_relevance 0.0 includes all matches + Then idx vector search_similar with min_relevance 0.0 includes all matches + + # ================================================================= + # Coverage — stub size limits (P2-16) + # ================================================================= + + @coverage + Scenario: InMemoryTextIndexBackend enforces max_entries limit + Then idx text stub should raise RuntimeError when max_entries exceeded + + @coverage + Scenario: InMemoryVectorIndexBackend enforces max_entries limit + Then idx vector stub should raise RuntimeError when max_entries exceeded + + @coverage + Scenario: InMemoryGraphIndexBackend enforces max_entries limit + Then idx graph stub should raise RuntimeError when max_entries exceeded + + # ================================================================= + # Coverage — provenance metadata persistence (P2-22, P2-27) + # ================================================================= + + @coverage + Scenario: index_graph stores full provenance metadata and confidence + Then idx index_graph should store provenance metadata triples + + # ================================================================= + # Coverage — callback + event_bus combo (P3-#18) + # ================================================================= + + @coverage + Scenario: ResourceFileWatcher fires both callback and EventBus + Then idx file watcher should fire both callback and event_bus on change + + # ================================================================= + # Coverage — source_range validation (#28) + # ================================================================= + + @coverage + Scenario: ProvenanceMetadata rejects invalid source_range format + Then idx creating ProvenanceMetadata with invalid source_range should raise + + @coverage + Scenario: ProvenanceMetadata accepts empty source_range + Then idx creating ProvenanceMetadata with empty source_range should succeed + + # ================================================================= + # Coverage — move to new directory (#4) + # ================================================================= + + @coverage + Scenario: FileMovedEvent to new directory schedules observer for dest parent + Then idx file watcher should schedule new dir on cross-directory move + + # ================================================================= + # Coverage — concurrency (#58) + # ================================================================= + + @coverage + Scenario: Concurrent indexing of different resources is thread-safe + Then idx concurrent indexing of different resources should be thread-safe + + # ================================================================= + # Coverage — move then modify sequence (#59) + # ================================================================= + + @coverage + Scenario: FileMovedEvent followed by FileModifiedEvent on new path + Then idx file watcher should handle move then modify sequence + + # ================================================================= + # Coverage — empty-string filter validation in remove_triples + # ================================================================= + + @coverage + Scenario: Graph remove_triples rejects empty-string subject filter + Given idx a clean analyzer registry + And idx in-memory index backends + Then idx removing triples with empty-string subject should raise ValueError + + @coverage + Scenario: Graph remove_triples rejects empty-string predicate filter + Given idx a clean analyzer registry + And idx in-memory index backends + Then idx removing triples with empty-string predicate should raise ValueError + + @coverage + Scenario: Graph remove_triples rejects empty-string obj filter + Given idx a clean analyzer registry + And idx in-memory index backends + Then idx removing triples with empty-string obj should raise ValueError + + # ================================================================= + # Coverage — dest timer cancel and EventBus move dest_path + # ================================================================= + + @coverage @file-watching + Scenario: Rapid create-then-move cancels prior dest timer + Given idx a clean analyzer registry + And idx in-memory index backends + Then idx file watcher should cancel prior dest timer on rapid sequence + + @coverage @file-watching + Scenario: EventBus move event includes dest_path in details + Given idx a clean analyzer registry + And idx in-memory index backends + Then idx file watcher EventBus move event should include dest_path diff --git a/robot/helper_uko_indexer.py b/robot/helper_uko_indexer.py new file mode 100644 index 000000000..66e2a28f2 --- /dev/null +++ b/robot/helper_uko_indexer.py @@ -0,0 +1,491 @@ +"""Robot Framework helper for UKO Indexer smoke tests. + +Provides a CLI-style interface for Robot to invoke UKOIndexer creation, +index lifecycle operations, graceful degradation, and provenance tracking. +Exit code 0 = success, 1 = failure. + +Usage: + python robot/helper_uko_indexer.py +""" + +from __future__ import annotations + +import sys +from collections.abc import Callable +from pathlib import Path + +# Ensure the src directory is on the import path. +_SRC = str(Path(__file__).resolve().parents[1] / "src") +if _SRC not in sys.path: + sys.path.insert(0, _SRC) + +from cleveragents.application.services.uko_indexer import UKOIndexer # noqa: E402 +from cleveragents.application.services.uko_indexer_protocols import ( # noqa: E402 + ContentReader, + DefaultLifecycleHook, + LocationContentReader, +) +from cleveragents.domain.models.acms.analyzers import ( # noqa: E402 + AnalyzerRegistry, +) +from cleveragents.domain.models.acms.index_backends import ( # noqa: E402 + IndexedDocument, + SearchResult, +) +from cleveragents.domain.models.acms.index_stubs import ( # noqa: E402 + InMemoryGraphIndexBackend, + InMemoryTextIndexBackend, + InMemoryVectorIndexBackend, +) +from cleveragents.domain.models.acms.provenance import ( # noqa: E402 + IndexResult, + ProvenanceMetadata, +) +from cleveragents.domain.models.acms.python_analyzer import ( # noqa: E402 + PythonAnalyzer, +) +from cleveragents.domain.models.core.resource import Resource # noqa: E402 + +# Ensure features/ is importable for shared mocks +_FEATURES = str(Path(__file__).resolve().parents[1]) +if _FEATURES not in sys.path: + sys.path.insert(0, _FEATURES) + +from features.mocks.uko_indexer_mocks import InMemoryContentReader # noqa: E402 + +ULID_1 = "01HQ8ZDRX50000000000000001" +ULID_2 = "01HQ8ZDRX50000000000000002" +PROJECT = "local/test" + + +def _make_resource( + resource_id: str, + location: str = "src/example.py", +) -> Resource: + """Create a minimal Resource for testing.""" + return Resource( + resource_id=resource_id, + resource_type_name="git-checkout", + location=location, + classification="physical", + ) + + +def _make_indexer( + *, + text: bool = True, + vector: bool = True, +) -> tuple[ + UKOIndexer, + InMemoryContentReader, + InMemoryTextIndexBackend | None, + InMemoryVectorIndexBackend | None, + InMemoryGraphIndexBackend, +]: + """Create a fully-wired UKOIndexer with in-memory backends.""" + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer()) + reader = InMemoryContentReader() + graph = InMemoryGraphIndexBackend() + text_be = InMemoryTextIndexBackend() if text else None + vector_be = InMemoryVectorIndexBackend() if vector else None + indexer = UKOIndexer( + analyzer_registry=registry, + graph_backend=graph, + text_backend=text_be, + vector_backend=vector_be, + content_reader=reader, + ) + return ( + indexer, + reader, + text_be, + vector_be, + graph, + ) + + +def cmd_index_resource() -> int: + """Test basic index_resource pipeline.""" + try: + indexer, reader, _text, _vec, _graph = _make_indexer() + resource = _make_resource(ULID_1) + code = '"""Module doc."""\nclass Foo:\n """Class doc."""\n pass\n' + reader.set_content(ULID_1, code) + result = indexer.index_resource(resource, project=PROJECT) + + assert result.resource_id == ULID_1 + assert result.triple_count > 0 + assert result.analyzer_domain == "python" + assert result.text_docs_indexed == 1 + assert result.embeddings_indexed == 1 + assert len(result.errors) == 0 + assert indexer.indexed_resource_count == 1 + + print("uko-indexer-index-resource-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-index-resource-fail: {exc}") + return 1 + + +def cmd_remove_resource() -> int: + """Test remove_resource cleanup including backend state.""" + try: + indexer, reader, text_be, vec_be, graph_be = _make_indexer() + resource = _make_resource(ULID_1) + reader.set_content(ULID_1, "x = 1\n") + indexer.index_resource(resource, project=PROJECT) + assert indexer.indexed_resource_count == 1 + + # Verify backends have data before removal + assert graph_be.triple_count(PROJECT) > 0 + assert text_be is not None and text_be.document_count > 0 + assert vec_be is not None and vec_be.embedding_count > 0 + + indexer.remove_resource(ULID_1, project=PROJECT) + assert indexer.indexed_resource_count == 0 + + # Verify backends are cleaned up after removal + assert graph_be.triple_count(PROJECT) == 0 + assert text_be.document_count == 0 + assert vec_be.embedding_count == 0 + + print("uko-indexer-remove-resource-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-remove-resource-fail: {exc}") + return 1 + + +def cmd_reindex_resource() -> int: + """Test reindex_resource (remove + re-index).""" + try: + indexer, reader, _text, _vec, _graph = _make_indexer() + resource = _make_resource(ULID_1) + reader.set_content(ULID_1, "x = 1\n") + result1 = indexer.index_resource(resource, project=PROJECT) + assert result1.triple_count > 0 + + # Update content and reindex + reader.set_content( + ULID_1, + '"""Reindexed."""\nclass Bar:\n pass\n', + ) + result2 = indexer.reindex_resource(resource, project=PROJECT) + assert result2.resource_id == ULID_1 + assert result2.triple_count > 0 + assert indexer.indexed_resource_count == 1 + + print("uko-indexer-reindex-resource-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-reindex-resource-fail: {exc}") + return 1 + + +def cmd_graceful_degradation() -> int: + """Test graceful degradation without text/vector backends.""" + try: + indexer, reader, _, _, _graph = _make_indexer(text=False, vector=False) + resource = _make_resource(ULID_1) + reader.set_content(ULID_1, "y = 2\n") + result = indexer.index_resource(resource, project=PROJECT) + + assert result.triple_count > 0 + assert result.text_docs_indexed == 0 + assert result.embeddings_indexed == 0 + assert not indexer.has_text_backend + assert not indexer.has_vector_backend + assert len(result.errors) == 0 + + print("uko-indexer-graceful-degradation-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-graceful-degradation-fail: {exc}") + return 1 + + +def cmd_provenance() -> int: + """Test provenance metadata is attached to triples.""" + try: + # Verify ProvenanceMetadata construction + prov = ProvenanceMetadata( + source_resource=ULID_1, + source_path="src/example.py", + ) + assert prov.is_current is True + assert prov.valid_from is not None + + # Verify provenance is attached during indexing via IndexResult + indexer, reader, _, _, _graph = _make_indexer() + resource = _make_resource(ULID_1) + reader.set_content(ULID_1, "z = 3\n") + result = indexer.index_resource(resource, project=PROJECT) + assert result.triple_count > 0 + assert result.resource_id == ULID_1 + + print("uko-indexer-provenance-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-provenance-fail: {exc}") + return 1 + + +def cmd_index_backends() -> int: + """Test in-memory index backend operations.""" + try: + # Text backend + text_be = InMemoryTextIndexBackend() + text_be.index_document(PROJECT, "doc1", "hello world", {"key": "val"}) + results = text_be.search(PROJECT, "hello", limit=5) + assert len(results) == 1 + assert results[0].doc_id == "doc1" + text_be.remove_document(PROJECT, "doc1") + results = text_be.search(PROJECT, "hello", limit=5) + assert len(results) == 0 + + # Vector backend + vec_be = InMemoryVectorIndexBackend() + vec_be.index_embedding(PROJECT, "emb1", [1.0, 2.0, 3.0], {"key": "val"}) + results = vec_be.search_similar(PROJECT, [1.0, 2.0, 3.0], limit=5) + assert len(results) == 1 + assert results[0].doc_id == "emb1" + vec_be.remove_embedding(PROJECT, "emb1") + results = vec_be.search_similar(PROJECT, [1.0, 2.0, 3.0], limit=5) + assert len(results) == 0 + + # Graph backend + graph_be = InMemoryGraphIndexBackend() + graph_be.add_triple(PROJECT, "s1", "p1", "o1") + bindings = graph_be.query(PROJECT, "SELECT * WHERE { ?s ?p ?o }") + assert len(bindings) == 1 + assert bindings[0]["s"] == "s1" + assert bindings[0]["p"] == "p1" + assert bindings[0]["o"] == "o1" + graph_be.remove_triples(PROJECT, subject="s1", predicate=None, obj=None) + bindings = graph_be.query(PROJECT, "SELECT * WHERE { ?s ?p ?o }") + assert len(bindings) == 0 + + # Validation + doc = IndexedDocument(project=PROJECT, doc_id="d1", char_count=10) + assert doc.char_count == 10 + + sr = SearchResult(doc_id="d1", score=0.5, content="test") + assert sr.score == 0.5 + + print("uko-indexer-index-backends-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-index-backends-fail: {exc}") + return 1 + + +def cmd_validation() -> int: + """Test input validation guards.""" + try: + indexer, reader, _, _, _ = _make_indexer() + resource = _make_resource(ULID_1) + reader.set_content(ULID_1, "a = 1\n") + + # Empty project + try: + indexer.index_resource(resource, project="") + print("uko-indexer-validation-fail: no ValueError for empty project") + return 1 + except ValueError: + pass + + # Empty resource_id on remove + try: + indexer.remove_resource("", project=PROJECT) + print("uko-indexer-validation-fail: no ValueError for empty resource_id") + return 1 + except ValueError: + pass + + # Invalid analyzer_registry type + try: + UKOIndexer( + analyzer_registry="not_a_registry", # type: ignore[arg-type] + graph_backend=InMemoryGraphIndexBackend(), + ) + print("uko-indexer-validation-fail: no TypeError for invalid registry") + return 1 + except TypeError: + pass + + # SearchResult score out of range + try: + SearchResult(doc_id="x", score=1.5, content="test") + print("uko-indexer-validation-fail: no ValueError for score > 1.0") + return 1 + except ValueError: + pass + + print("uko-indexer-validation-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-validation-fail: {exc}") + return 1 + + +def cmd_protocol_compliance() -> int: + """Verify protocol compliance for key types.""" + try: + reader = InMemoryContentReader() + assert isinstance(reader, ContentReader) + + hook = DefaultLifecycleHook() + # DefaultLifecycleHook logs but doesn't fail + result = IndexResult( + resource_id=ULID_1, + analyzer_domain="python", + triple_count=5, + ) + hook.on_indexed(result) + hook.on_removed(ULID_1, PROJECT) + hook.on_error(ULID_1, "test error") + + # LocationContentReader protocol compliance + loc_reader = LocationContentReader() + assert isinstance(loc_reader, ContentReader) + + print("uko-indexer-protocol-compliance-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-protocol-compliance-fail: {exc}") + return 1 + + +def cmd_file_watching() -> int: + """Test ResourceFileWatcher lifecycle, change detection, and EventBus.""" + import threading + import time + + from watchdog.events import FileDeletedEvent, FileModifiedEvent + + from cleveragents.application.services.resource_file_watcher import ( + FileChangeType, + ResourceFileWatcher, + ) + from cleveragents.infrastructure.events.models import DomainEvent + from cleveragents.infrastructure.events.types import EventType + + class _MiniEventBus: + def __init__(self) -> None: + self.events: list[DomainEvent] = [] + self._ev = threading.Event() + + def emit(self, event: DomainEvent) -> None: + self.events.append(event) + self._ev.set() + + def subscribe(self, event_type: EventType, handler: object) -> None: + _ = event_type, handler + + try: + import tempfile as _tf + from pathlib import Path as _P + + tmpdir = _tf.mkdtemp() + fpath = _P(tmpdir) / "watched.py" + fpath.write_text("x = 1\n") + + changes: list[tuple[str, str, FileChangeType]] = [] + cb_event = threading.Event() + + def _on_change(rid: str, proj: str, ct: FileChangeType) -> None: + changes.append((rid, proj, ct)) + cb_event.set() + + bus = _MiniEventBus() + watcher = ResourceFileWatcher( + on_change=_on_change, + event_bus=bus, + debounce_seconds=0.05, + ) + + # 1. Lifecycle + assert not watcher.is_running + watcher.start() + assert watcher.is_running + + # 2. Watch a file + watcher.watch(fpath, resource_id=ULID_1, project=PROJECT) + assert len(watcher.watched_paths) == 1 + + # 3. Simulate modification + ev = FileModifiedEvent(str(fpath.resolve())) + watcher._handle_fs_event(ev) + assert cb_event.wait(timeout=2.0), "Callback did not fire" + assert changes[-1][0] == ULID_1 + assert changes[-1][2] == FileChangeType.MODIFIED + + # 4. EventBus received event + time.sleep(0.1) + assert len(bus.events) > 0 + assert bus.events[-1].event_type == EventType.RESOURCE_MODIFIED + assert bus.events[-1].details["resource_id"] == ULID_1 + + # 5. Simulate deletion + cb_event.clear() + ev2 = FileDeletedEvent(str(fpath.resolve())) + watcher._handle_fs_event(ev2) + assert cb_event.wait(timeout=2.0), "Callback did not fire for delete" + assert changes[-1][2] == FileChangeType.DELETED + + # 6. Unwatch + watcher.unwatch(fpath) + assert len(watcher.watched_paths) == 0 + + # 7. Stop + watcher.stop() + assert not watcher.is_running + + # Cleanup + fpath.unlink(missing_ok=True) + _P(tmpdir).rmdir() + + print("uko-indexer-file-watching-ok") + return 0 + except Exception as exc: + print(f"uko-indexer-file-watching-fail: {exc}") + import traceback + + traceback.print_exc() + return 1 + + +COMMANDS: dict[str, Callable[[], int]] = { + "index-resource": cmd_index_resource, + "remove-resource": cmd_remove_resource, + "reindex-resource": cmd_reindex_resource, + "graceful-degradation": cmd_graceful_degradation, + "provenance": cmd_provenance, + "index-backends": cmd_index_backends, + "validation": cmd_validation, + "protocol-compliance": cmd_protocol_compliance, + "file-watching": cmd_file_watching, +} + + +def main() -> int: + """Entry point called by Robot Framework ``Run Process``.""" + if len(sys.argv) < 2: + print("Usage: helper_uko_indexer.py ") + print(f"Commands: {', '.join(COMMANDS)}") + return 1 + + command: str = sys.argv[1] + handler = COMMANDS.get(command) + if handler is None: + print(f"Unknown command: {command}") + print(f"Commands: {', '.join(COMMANDS)}") + return 1 + + return handler() + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/robot/uko_indexer.robot b/robot/uko_indexer.robot new file mode 100644 index 000000000..a31c4ff02 --- /dev/null +++ b/robot/uko_indexer.robot @@ -0,0 +1,81 @@ +*** Settings *** +Documentation Smoke tests for UKO Indexer Real-time Index Synchronization +Resource ${CURDIR}/common.resource +Suite Setup Setup Test Environment +Suite Teardown Cleanup Test Environment + +*** Variables *** +${HELPER} ${CURDIR}/helper_uko_indexer.py + +*** Test Cases *** +Index Resource Pipeline + [Documentation] Verify UKOIndexer indexes a resource into graph, text, and vector backends + ${result}= Run Process ${PYTHON} ${HELPER} index-resource cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-index-resource-ok + +Remove Resource Cleanup + [Documentation] Verify UKOIndexer removes a resource from all indices + ${result}= Run Process ${PYTHON} ${HELPER} remove-resource cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-remove-resource-ok + +Reindex Resource Lifecycle + [Documentation] Verify reindex_resource removes old data then re-indexes + ${result}= Run Process ${PYTHON} ${HELPER} reindex-resource cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-reindex-resource-ok + +Graceful Degradation Without Optional Backends + [Documentation] Verify indexing works without text and vector backends + ${result}= Run Process ${PYTHON} ${HELPER} graceful-degradation cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-graceful-degradation-ok + +Provenance Metadata Attached + [Documentation] Verify provenance metadata is attached to indexed triples + ${result}= Run Process ${PYTHON} ${HELPER} provenance cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-provenance-ok + +Index Backend Operations + [Documentation] Verify in-memory text, vector, and graph backend CRUD operations + ${result}= Run Process ${PYTHON} ${HELPER} index-backends cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-index-backends-ok + +Input Validation Guards + [Documentation] Verify validation rejects empty project, empty resource_id, invalid types + ${result}= Run Process ${PYTHON} ${HELPER} validation cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-validation-ok + +Protocol Compliance + [Documentation] Verify ContentReader and IndexLifecycleHook protocol compliance + ${result}= Run Process ${PYTHON} ${HELPER} protocol-compliance cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-protocol-compliance-ok + +File Watching + [Documentation] Verify ResourceFileWatcher lifecycle, change detection, and EventBus + ${result}= Run Process ${PYTHON} ${HELPER} file-watching cwd=${WORKSPACE} + Log ${result.stdout} + Log ${result.stderr} + Should Be Equal As Integers ${result.rc} 0 + Should Contain ${result.stdout} uko-indexer-file-watching-ok diff --git a/src/cleveragents/application/container.py b/src/cleveragents/application/container.py index fe32ffb8b..56b2215f3 100644 --- a/src/cleveragents/application/container.py +++ b/src/cleveragents/application/container.py @@ -41,6 +41,9 @@ from cleveragents.application.services.project_service import ProjectService from cleveragents.application.services.repo_indexing_service import ( RepoIndexingService, ) +from cleveragents.application.services.resource_file_watcher import ( + ResourceFileWatcher, +) from cleveragents.application.services.resource_registry_service import ( ResourceRegistryService, ) @@ -49,8 +52,15 @@ from cleveragents.application.services.skeleton_compressor import ( ) from cleveragents.application.services.subplan_service import SubplanService from cleveragents.application.services.trace_service import TraceService +from cleveragents.application.services.uko_indexer import UKOIndexer from cleveragents.application.services.vector_store_service import VectorStoreService from cleveragents.config.settings import Settings, get_settings +from cleveragents.domain.models.acms.analyzers import AnalyzerRegistry +from cleveragents.domain.models.acms.index_stubs import ( + InMemoryGraphIndexBackend, + InMemoryTextIndexBackend, + InMemoryVectorIndexBackend, +) from cleveragents.domain.models.acms.stubs import ( InMemoryGraphBackend, InMemoryTextBackend, @@ -222,6 +232,40 @@ def _build_checkpoint_service( ) +def _resolve_auto_reindex() -> bool: + """Read ``index.auto-reindex`` from ConfigService, defaulting to True.""" + try: + from cleveragents.application.services.config_service import ( + ConfigService, + ) + + svc = ConfigService() + resolved = svc.resolve("index.auto-reindex") + return bool(resolved.value) if resolved.value is not None else True + except Exception: + return True + + +def _build_analyzer_registry() -> AnalyzerRegistry: + """Build an AnalyzerRegistry with built-in analyzers registered.""" + from cleveragents.domain.models.acms.python_analyzer import PythonAnalyzer + + registry = AnalyzerRegistry() + registry.register(PythonAnalyzer(), priority=10) + return registry + + +def _build_resource_file_watcher( + event_bus: ReactiveEventBus, +) -> ResourceFileWatcher: + """Build a ResourceFileWatcher with ConfigService integration.""" + auto_reindex = _resolve_auto_reindex() + return ResourceFileWatcher( + event_bus=event_bus, + auto_reindex=auto_reindex, + ) + + def _build_trace_service( database_url: str, settings: Settings | None = None, @@ -440,6 +484,26 @@ class Container(containers.DeclarativeContainer): vector_backend = providers.Singleton(InMemoryVectorBackend) graph_backend = providers.Singleton(InMemoryGraphBackend) + # ACMS UKO Indexer — write-side index backends (#578) + analyzer_registry = providers.Singleton(_build_analyzer_registry) + index_text_backend = providers.Singleton(InMemoryTextIndexBackend) + index_vector_backend = providers.Singleton(InMemoryVectorIndexBackend) + index_graph_backend = providers.Singleton(InMemoryGraphIndexBackend) + uko_indexer = providers.Singleton( + UKOIndexer, + analyzer_registry=analyzer_registry, + graph_backend=index_graph_backend, + text_backend=index_text_backend, + vector_backend=index_vector_backend, + ) + + # ACMS ResourceFileWatcher — watches resource files for changes (#578). + # Reads ``index.auto-reindex`` from ConfigService at construction. + resource_file_watcher = providers.Singleton( + _build_resource_file_watcher, + event_bus=event_bus, + ) + # Reactive routing stream_router = providers.Singleton(ReactiveStreamRouter) langgraph_bridge = providers.Singleton( diff --git a/src/cleveragents/application/services/__init__.py b/src/cleveragents/application/services/__init__.py index adc0d9a1f..7eeb31b18 100644 --- a/src/cleveragents/application/services/__init__.py +++ b/src/cleveragents/application/services/__init__.py @@ -169,6 +169,15 @@ from cleveragents.application.services.tool_registry_service import ( from cleveragents.application.services.trace_service import ( TraceService, ) +from cleveragents.application.services.uko_indexer import ( + UKOIndexer, +) +from cleveragents.application.services.uko_indexer_protocols import ( + ContentReader, + DefaultLifecycleHook, + IndexLifecycleHook, + LocationContentReader, +) from cleveragents.application.services.uko_loader import ( UKOLoader, UKOValidationError, @@ -208,6 +217,7 @@ __all__ = [ "ConfigService", "ContainerUnavailableError", "ContentHashDeduplicator", + "ContentReader", "ContextFragment", "CoordinationResult", "CoordinatorConfig", @@ -220,6 +230,7 @@ __all__ = [ "DecompositionNode", "DecompositionResult", "DecompositionService", + "DefaultLifecycleHook", "DefaultValidationRunner", "DependencyClosureComputer", "DependencyCycleRule", @@ -234,7 +245,9 @@ __all__ = [ "FusionEngine", "FusionResult", "GreedyKnapsackPacker", + "IndexLifecycleHook", "InvariantService", + "LocationContentReader", "MaxDepthResolver", "MergeConflictError", "MissingImportRule", @@ -287,6 +300,7 @@ __all__ = [ "TemporalArchaeologyStrategy", "ToolRegistryService", "TraceService", + "UKOIndexer", "UKOLoader", "UKOValidationError", "ValidationAttachment", diff --git a/src/cleveragents/application/services/resource_file_watcher.py b/src/cleveragents/application/services/resource_file_watcher.py new file mode 100644 index 000000000..2f9ffb4e2 --- /dev/null +++ b/src/cleveragents/application/services/resource_file_watcher.py @@ -0,0 +1,486 @@ +"""Resource file watcher for real-time index synchronization. + +Monitors filesystem paths associated with indexed resources using watchdog. +When file modifications are detected, the watcher debounces rapid changes +and triggers callbacks for incremental re-indexing. + +Spec reference: Architecture > ACMS > Real-time Index Synchronization + - Index Lifecycle: "Code changed → File modification detected → + Immediate incremental update → Indices reflect latest state" + - watchdog >= 4.0.0 (specification.md line ~41680) + - ``index.auto-reindex`` config key (specification.md line ~28711) +""" + +from __future__ import annotations + +import threading +from collections.abc import Callable +from enum import StrEnum +from pathlib import Path +from typing import Any + +import structlog +from watchdog.events import ( + FileCreatedEvent, + FileDeletedEvent, + FileModifiedEvent, + FileMovedEvent, + FileSystemEvent, + FileSystemEventHandler, +) +from watchdog.observers import Observer # type: ignore[import-untyped] + +from cleveragents.infrastructure.events.models import DomainEvent +from cleveragents.infrastructure.events.protocol import EventBus +from cleveragents.infrastructure.events.types import EventType + +_logger = structlog.get_logger(__name__) + + +def _sanitize_log_value(value: str) -> str: + """Replace control characters in *value* to prevent log injection.""" + return value.replace("\n", "\\n").replace("\r", "\\r") + + +class FileChangeType(StrEnum): + """Type of file system change detected by the watcher.""" + + CREATED = "created" + MODIFIED = "modified" + DELETED = "deleted" + MOVED = "moved" + + +DEFAULT_DEBOUNCE_SECONDS: float = 0.5 +"""Default debounce interval in seconds.""" + + +class ResourceFileWatcher: + """Watches filesystem paths for resource changes and triggers callbacks. + + Uses watchdog to monitor files associated with indexed resources. + When a modification is detected, the watcher debounces rapid changes + and invokes the registered callback with the affected resource details. + Optionally emits ``RESOURCE_MODIFIED`` domain events via an EventBus. + + Parameters + ---------- + on_change: + Callback invoked when a watched resource file changes. + Receives ``(resource_id, project, change_type)`` arguments. + event_bus: + Optional EventBus for emitting ``RESOURCE_MODIFIED`` events. + debounce_seconds: + Minimum interval between callback invocations for the same path. + Rapid changes within this window are coalesced into a single + callback. Default 0.5 seconds. + """ + + def __init__( + self, + *, + on_change: Callable[[str, str, FileChangeType], None] | None = None, + event_bus: EventBus | None = None, + debounce_seconds: float = DEFAULT_DEBOUNCE_SECONDS, + auto_reindex: bool = True, + ) -> None: + if debounce_seconds < 0: + msg = f"debounce_seconds must be >= 0, got {debounce_seconds}" + raise ValueError(msg) + self._on_change = on_change + self._event_bus = event_bus + self._debounce_seconds = debounce_seconds + self._auto_reindex = auto_reindex + self._observer: Observer | None = None # type: ignore[valid-type] + self._lock = threading.Lock() + # resolved path (str) -> (resource_id, project) + self._watched_paths: dict[str, tuple[str, str]] = {} + # parent directory (str) -> watchdog ObservedWatch handle + self._dir_watches: dict[str, Any] = {} + # resolved path (str) -> pending debounce Timer + self._pending_timers: dict[str, threading.Timer] = {} + self._running = False + + _logger.debug( + "file_watcher.initialized", + debounce_seconds=debounce_seconds, + auto_reindex=auto_reindex, + has_callback=on_change is not None, + has_event_bus=event_bus is not None, + ) + + # ------------------------------------------------------------------ + # Path management + # ------------------------------------------------------------------ + + def watch(self, path: Path, *, resource_id: str, project: str) -> None: + """Register a file path for monitoring. + + Parameters + ---------- + path: + Filesystem path to watch. Must exist and be a regular file. + resource_id: + The resource ID associated with this file. + project: + The project scope for this resource. + + Raises + ------ + ValueError: + If the path does not exist, is not a file, or arguments are + empty. + """ + if not resource_id or not resource_id.strip(): + msg = "resource_id must be non-empty" + raise ValueError(msg) + if not project or not project.strip(): + msg = "project must be non-empty" + raise ValueError(msg) + + resolved = path.resolve() + if not resolved.exists(): + msg = f"Path does not exist: {resolved}" + raise ValueError(msg) + if not resolved.is_file(): + msg = f"Path is not a file: {resolved}" + raise ValueError(msg) + + resolved_str = str(resolved) + parent_str = str(resolved.parent) + + with self._lock: + self._watched_paths[resolved_str] = (resource_id, project) + if ( + self._running + and self._observer is not None + and parent_str not in self._dir_watches + ): + handler = _ResourceChangeHandler(self) + watch_handle = self._observer.schedule( + handler, + parent_str, + recursive=False, + ) + self._dir_watches[parent_str] = watch_handle + + _logger.debug( + "file_watcher.path_registered", + path=_sanitize_log_value(resolved_str), + resource_id=_sanitize_log_value(resource_id), + project=_sanitize_log_value(project), + ) + + def unwatch(self, path: Path) -> None: + """Stop monitoring a file path. + + Parameters + ---------- + path: + The filesystem path to stop watching. + """ + resolved = path.resolve() + resolved_str = str(resolved) + parent_str = str(resolved.parent) + + with self._lock: + self._watched_paths.pop(resolved_str, None) + timer = self._pending_timers.pop(resolved_str, None) + if timer is not None: + timer.cancel() + # Unschedule directory if no more files in it are watched + still_watching = any( + str(Path(p).parent) == parent_str for p in self._watched_paths + ) + if not still_watching and parent_str in self._dir_watches: + if self._observer is not None: + self._observer.unschedule(self._dir_watches[parent_str]) + del self._dir_watches[parent_str] + + _logger.debug( + "file_watcher.path_unregistered", + path=_sanitize_log_value(resolved_str), + ) + + # ------------------------------------------------------------------ + # Lifecycle + # ------------------------------------------------------------------ + + def start(self) -> None: + """Start the watchdog observer. + + If ``auto_reindex`` was set to ``False`` at construction + (corresponding to ``index.auto-reindex = false`` in config), + the observer is **not** started and a debug log is emitted. + + Raises + ------ + RuntimeError: + If the watcher is already running. + """ + if not self._auto_reindex: + _logger.info( + "file_watcher.disabled", + reason="index.auto-reindex is false", + ) + return + + with self._lock: + if self._running: + msg = "File watcher is already running" + raise RuntimeError(msg) + observer = Observer() + observer.daemon = True + + handler = _ResourceChangeHandler(self) + seen_dirs: set[str] = set() + for path_str in self._watched_paths: + parent_str = str(Path(path_str).parent) + if parent_str not in seen_dirs: + watch_handle = observer.schedule( + handler, + parent_str, + recursive=False, + ) + self._dir_watches[parent_str] = watch_handle + seen_dirs.add(parent_str) + + observer.start() + self._observer = observer + self._running = True + + _logger.info( + "file_watcher.started", + watched_paths=len(self._watched_paths), + watched_dirs=len(self._dir_watches), + ) + + def stop(self) -> None: + """Stop the watchdog observer and cancel pending timers.""" + observer: Observer | None = None # type: ignore[valid-type] + with self._lock: + if not self._running: + return + for timer in self._pending_timers.values(): + timer.cancel() + self._pending_timers.clear() + observer = self._observer + self._observer = None + self._dir_watches.clear() + self._running = False + + # Join outside lock to avoid deadlock with timer callbacks. + if observer is not None: + observer.stop() + observer.join(timeout=5.0) + if observer.is_alive(): + _logger.warning( + "file_watcher.observer_join_timeout", + hint="Observer thread did not stop within 5 seconds", + ) + + _logger.info("file_watcher.stopped") + + @property + def is_running(self) -> bool: + """Whether the watcher is currently active.""" + return self._running + + @property + def watched_paths(self) -> dict[str, tuple[str, str]]: + """Snapshot of watched paths mapped to ``(resource_id, project)``.""" + with self._lock: + return dict(self._watched_paths) + + @property + def debounce_seconds(self) -> float: + """Current debounce interval.""" + return self._debounce_seconds + + @property + def auto_reindex(self) -> bool: + """Whether automatic re-indexing on file changes is enabled. + + Reflects the ``index.auto-reindex`` configuration key passed + at construction time. + """ + return self._auto_reindex + + # ------------------------------------------------------------------ + # Internal event handling + # ------------------------------------------------------------------ + + def _handle_fs_event(self, event: FileSystemEvent) -> None: + """Process a watchdog filesystem event (called from observer thread).""" + if event.is_directory: + return + + src_path = str(Path(str(event.src_path)).resolve()) + + if isinstance(event, FileCreatedEvent): + change_type = FileChangeType.CREATED + elif isinstance(event, FileModifiedEvent): + change_type = FileChangeType.MODIFIED + elif isinstance(event, FileDeletedEvent): + change_type = FileChangeType.DELETED + elif isinstance(event, FileMovedEvent): + change_type = FileChangeType.MOVED + else: + return + + # Single lock acquisition covers lookup, path-move update, and + # timer scheduling atomically — prevents races with unwatch(). + with self._lock: + entry = self._watched_paths.get(src_path) + if entry is None: + return + resource_id, project = entry + + # On move, update watched-paths so subsequent events on the + # new location are still tracked for this resource. + # Also schedule the destination's parent directory if it + # isn't already being watched (fixes missed events when a + # file moves to a new directory). + dest_path: str | None = None + if isinstance(event, FileMovedEvent): + dest_path = str(Path(str(event.dest_path)).resolve()) + self._watched_paths.pop(src_path, None) + self._watched_paths[dest_path] = (resource_id, project) + dest_parent = str(Path(dest_path).parent) + if ( + self._running + and self._observer is not None + and dest_parent not in self._dir_watches + ): + handler = _ResourceChangeHandler(self) + watch_handle = self._observer.schedule( + handler, + dest_parent, + recursive=False, + ) + self._dir_watches[dest_parent] = watch_handle + + # Use dest_path as the timer key after a move so that + # subsequent events on the new path can cancel/coalesce + # the pending timer correctly. + timer_key = dest_path if dest_path is not None else src_path + existing = self._pending_timers.pop(src_path, None) + if existing is not None: + existing.cancel() + # Also cancel any prior timer for dest_path (e.g. from a + # rapid create-then-move sequence). + if dest_path is not None: + existing_dest = self._pending_timers.pop(dest_path, None) + if existing_dest is not None: + existing_dest.cancel() + timer = threading.Timer( + self._debounce_seconds, + self._fire_change, + args=(timer_key, resource_id, project, change_type, dest_path), + ) + timer.daemon = True + self._pending_timers[timer_key] = timer + timer.start() + + _logger.debug( + "file_watcher.change_detected", + path=_sanitize_log_value(src_path), + resource_id=_sanitize_log_value(resource_id), + change_type=str(change_type), + ) + if dest_path is not None: + _logger.debug( + "file_watcher.path_moved", + old_path=_sanitize_log_value(src_path), + new_path=_sanitize_log_value(dest_path), + resource_id=_sanitize_log_value(resource_id), + ) + + def _fire_change( + self, + path: str, + resource_id: str, + project: str, + change_type: FileChangeType, + dest_path: str | None = None, + ) -> None: + """Fire the change callback and/or EventBus event after debounce.""" + with self._lock: + self._pending_timers.pop(path, None) + if not self._running: + return + + _logger.info( + "file_watcher.change_fired", + path=_sanitize_log_value(path), + resource_id=_sanitize_log_value(resource_id), + project=_sanitize_log_value(project), + change_type=str(change_type), + ) + + if self._on_change is not None: + try: + self._on_change(resource_id, project, change_type) + except Exception: + _logger.exception( + "file_watcher.callback_error", + resource_id=_sanitize_log_value(resource_id), + project=_sanitize_log_value(project), + ) + + if self._event_bus is not None: + try: + # EventType has no RESOURCE_DELETED variant (out of PR + # scope — infrastructure enum). We emit + # RESOURCE_MODIFIED for *all* change types and include + # ``change_type`` in ``details`` so subscribers can + # distinguish created/modified/deleted/moved events. + details: dict[str, str] = { + "resource_id": resource_id, + "path": path, + "change_type": str(change_type), + } + if dest_path is not None: + details["dest_path"] = dest_path + domain_event = DomainEvent( + event_type=EventType.RESOURCE_MODIFIED, + project_name=project, + details=details, + ) + self._event_bus.emit(domain_event) + except Exception: + _logger.exception( + "file_watcher.event_bus_error", + resource_id=_sanitize_log_value(resource_id), + ) + + +class _ResourceChangeHandler(FileSystemEventHandler): + """Watchdog handler that routes filesystem events to the watcher.""" + + def __init__(self, watcher: ResourceFileWatcher) -> None: + super().__init__() + self._watcher = watcher + + def on_created(self, event: FileSystemEvent) -> None: + """Handle file creation.""" + self._watcher._handle_fs_event(event) + + def on_modified(self, event: FileSystemEvent) -> None: + """Handle file modification.""" + self._watcher._handle_fs_event(event) + + def on_deleted(self, event: FileSystemEvent) -> None: + """Handle file deletion.""" + self._watcher._handle_fs_event(event) + + def on_moved(self, event: FileSystemEvent) -> None: + """Handle file move.""" + self._watcher._handle_fs_event(event) + + +__all__ = [ + "DEFAULT_DEBOUNCE_SECONDS", + "FileChangeType", + "ResourceFileWatcher", +] diff --git a/src/cleveragents/application/services/uko_indexer.py b/src/cleveragents/application/services/uko_indexer.py new file mode 100644 index 000000000..8957c78cf --- /dev/null +++ b/src/cleveragents/application/services/uko_indexer.py @@ -0,0 +1,499 @@ +"""UKO Indexer — Real-time Index Synchronization service. + +Based on ``docs/specification.md`` > ACMS > Real-time Index +Synchronization and Forgejo issue #578. +""" + +from __future__ import annotations + +import threading +from datetime import UTC, datetime + +import structlog + +from cleveragents.application.services.uko_indexer_internals import ( + attach_provenance, + fire_on_error, + fire_on_indexed, + fire_on_removed, + index_graph, + index_text, + index_vector, +) +from cleveragents.application.services.uko_indexer_protocols import ( + ContentReader, + DefaultLifecycleHook, + IndexLifecycleHook, + LocationContentReader, +) +from cleveragents.domain.models.acms.analyzers import AnalyzerRegistry +from cleveragents.domain.models.acms.index_backends import ( + GraphIndexBackend, + TextIndexBackend, + VectorIndexBackend, +) +from cleveragents.domain.models.acms.provenance import IndexResult +from cleveragents.domain.models.core.resource import Resource + +logger = structlog.get_logger(__name__) + +#: Default maximum number of triples an analyzer may produce per resource. +DEFAULT_MAX_TRIPLES: int = 50_000 + + +# --------------------------------------------------------------------------- +# UKOIndexer +# --------------------------------------------------------------------------- + + +class UKOIndexer: + """Analyse resources into UKO triples and index into backends. + + Pipeline: analyzer -> content read -> triples -> provenance -> + graph -> text -> vector. Graceful degradation when optional + backends are ``None``. + + All public methods are thread-safe: a global lock guards tracking + dicts while per-resource locks allow concurrent indexing. + """ + + def __init__( + self, + *, + analyzer_registry: AnalyzerRegistry, + graph_backend: GraphIndexBackend, + text_backend: TextIndexBackend | None = None, + vector_backend: VectorIndexBackend | None = None, + content_reader: ContentReader | None = None, + lifecycle_hook: IndexLifecycleHook | None = None, + max_triples: int = DEFAULT_MAX_TRIPLES, + ) -> None: + if not isinstance(analyzer_registry, AnalyzerRegistry): + raise TypeError( + "analyzer_registry must be an AnalyzerRegistry, " + f"got {type(analyzer_registry).__name__}" + ) + if not isinstance(graph_backend, GraphIndexBackend): + raise TypeError( + "graph_backend must satisfy GraphIndexBackend, " + f"got {type(graph_backend).__name__}" + ) + self._analyzer_registry = analyzer_registry + self._graph_backend = graph_backend + self._text_backend = text_backend + self._vector_backend = vector_backend + self._content_reader: ContentReader = content_reader or LocationContentReader() + self._lifecycle_hook: IndexLifecycleHook = ( + lifecycle_hook or DefaultLifecycleHook() + ) + if max_triples < 1: + raise ValueError("max_triples must be positive") + self._max_triples = max_triples + self._lock = threading.Lock() + self._resource_locks: dict[str, threading.Lock] = {} + self._indexed_resources: dict[str, str] = {} + self._resource_subjects: dict[str, set[str]] = {} + self._resource_analyzer: dict[str, str] = {} + + logger.debug( + "indexer.initialized", + text_available=text_backend is not None, + vector_available=vector_backend is not None, + analyzer_count=len(analyzer_registry), + ) + + # ------------------------------------------------------------------ + # Properties + # ------------------------------------------------------------------ + + @property + def analyzer_registry(self) -> AnalyzerRegistry: + """The analyzer registry used for resource analysis.""" + return self._analyzer_registry + + @property + def indexed_resource_count(self) -> int: + """Number of resources currently indexed.""" + with self._lock: + return len(self._indexed_resources) + + @property + def has_text_backend(self) -> bool: + """Whether a text index backend is available.""" + return self._text_backend is not None + + @property + def has_vector_backend(self) -> bool: + """Whether a vector index backend is available.""" + return self._vector_backend is not None + + def _resource_lock(self, resource_id: str) -> threading.Lock: + """Return (or create) the per-resource lock for *resource_id*.""" + with self._lock: + lock = self._resource_locks.get(resource_id) + if lock is None: + lock = threading.Lock() + self._resource_locks[resource_id] = lock + return lock + + def index_resource( + self, + resource: Resource, + *, + project: str, + ) -> IndexResult: + """Index a resource into all available backends. + + Implements the spec's ``UKOIndexer.index_resource()`` pipeline: + analyze -> provenance -> graph -> text -> vector. + + Args: + resource: The resource to index. + project: Namespaced project name (e.g. ``"local/my-app"``). + + Returns: + :class:`IndexResult` summarizing the operation. + + Raises: + ValueError: If *project* is empty or whitespace-only. + """ + if not project or not project.strip(): + raise ValueError("project must be a non-empty string") + + log = logger.bind( + resource_id=resource.resource_id, + project=project, + ) + log.debug("indexer.index_resource.start") + + res_lock = self._resource_lock(resource.resource_id) + prior_errors: list[str] = [] + with res_lock: + # Idempotency: if already indexed, clean up old data first + # so triples don't accumulate (spec: "Indices reflect latest + # state"). Release global lock before calling + # _remove_resource_internal (it acquires the lock itself). + with self._lock: + if resource.resource_id in self._indexed_resources: + old_project = self._indexed_resources[resource.resource_id] + else: + old_project = None + if old_project is not None: + log.debug( + "indexer.removing_stale_data", + old_project=old_project, + ) + prior_errors = self._remove_resource_internal( + resource.resource_id, project=old_project + ) + + return self._index_resource_core( + resource, project, log, prior_errors=prior_errors + ) + + def _index_resource_core( + self, + resource: Resource, + project: str, + log: structlog.stdlib.BoundLogger, + *, + prior_errors: list[str] | None = None, + ) -> IndexResult: + """Core indexing logic — caller MUST hold the per-resource lock.""" + errors: list[str] = list(prior_errors) if prior_errors else [] + + # 1. Determine the best analyzer + analyzer = self._analyzer_registry.get_for_resource(resource) + if analyzer is None: + log.debug("indexer.no_analyzer", location=resource.location) + # No analyzer matched — nothing was indexed, so we do NOT + # fire on_indexed (its contract is "successfully indexed"). + return IndexResult( + resource_id=resource.resource_id, + analyzer_domain="none", + ) + + # 2. Read resource content (I/O — outside global lock) + try: + content = self._content_reader.read_content(resource) + except (OSError, ValueError) as exc: + error_msg = f"Failed to read resource content: {type(exc).__name__}" + log.warning("indexer.read_failed", error=str(exc)) + fire_on_error(self._lifecycle_hook, resource.resource_id, error_msg) + return IndexResult( + resource_id=resource.resource_id, + analyzer_domain=analyzer.domain, + errors=(error_msg,), + ) + + # 3. Produce UKO triples (CPU — outside global lock) + resource_uri = f"uko://resource/{resource.resource_id}" + try: + triples = analyzer.analyze(content, resource_uri) + except Exception as exc: + error_msg = f"Analyzer error ({analyzer.domain}): {type(exc).__name__}" + log.warning("indexer.analyze_failed", error=str(exc), exc_info=True) + errors.append(error_msg) + fire_on_error(self._lifecycle_hook, resource.resource_id, error_msg) + return IndexResult( + resource_id=resource.resource_id, + analyzer_domain=analyzer.domain, + errors=tuple(errors), + ) + + # Cap triple output to prevent OOM from pathological analyzers. + if len(triples) > self._max_triples: + cap_msg = ( + f"Triples capped: analyzer produced {len(triples)}, " + f"limit is {self._max_triples}" + ) + log.warning( + "indexer.triples_capped", + produced=len(triples), + max_triples=self._max_triples, + ) + errors.append(cap_msg) + triples = triples[: self._max_triples] + + log.debug( + "indexer.triples_produced", + triple_count=len(triples), + analyzer=analyzer.domain, + ) + + # 4. Attach provenance + now = datetime.now(UTC) + provenanced = attach_provenance(triples, resource, now) + + # 5. Store in graph backend (backend I/O — outside global lock) + graph_count, subjects = index_graph( + self._graph_backend, + project, + resource, + provenanced, + errors, + resource_uri, + ) + + # 6. Index text content + text_count = index_text( + self._text_backend, + project, + resource, + content, + errors, + resource_uri, + ) + + # 7. Index embeddings + vector_count = index_vector( + self._vector_backend, + project, + resource, + content, + errors, + resource_uri, + ) + + # 8. Update tracking dicts only if at least one backend stored + # data, so indexed_resource_count stays accurate. + if graph_count > 0 or text_count > 0 or vector_count > 0: + with self._lock: + self._indexed_resources[resource.resource_id] = project + self._resource_analyzer[resource.resource_id] = analyzer.domain + if subjects: + self._resource_subjects[resource.resource_id] = subjects + + result = IndexResult( + resource_id=resource.resource_id, + triple_count=graph_count, + text_docs_indexed=text_count, + embeddings_indexed=vector_count, + analyzer_domain=analyzer.domain, + errors=tuple(errors), + ) + fire_on_indexed(self._lifecycle_hook, result) + + log.debug( + "indexer.index_resource.done", + triple_count=graph_count, + text_docs=text_count, + vectors=vector_count, + error_count=len(errors), + ) + return result + + def remove_resource( + self, + resource_id: str, + *, + project: str, + ) -> None: + """Remove a resource from all indices. + + Uses the caller's *project* for backend cleanup. If the + resource was indexed under a different project, the stored + project is used instead (cross-project safety). + + Args: + resource_id: ULID of the resource to remove. + project: Namespaced project name. + + Raises: + ValueError: If *resource_id* or *project* is empty + or whitespace-only. + """ + if not resource_id or not resource_id.strip(): + raise ValueError("resource_id must be a non-empty string") + if not project or not project.strip(): + raise ValueError("project must be a non-empty string") + + log = logger.bind(resource_id=resource_id, project=project) + log.debug("indexer.remove_resource.start") + + res_lock = self._resource_lock(resource_id) + with res_lock: + with self._lock: + if resource_id not in self._indexed_resources: + log.debug("indexer.remove_resource.not_tracked") + return + stored_project = self._indexed_resources[resource_id] + + # Use stored_project for backend cleanup (cross-project + # safety) and rebind the logger to reflect the actual + # project used for removal. + if stored_project != project: + log = log.bind(stored_project=stored_project) + self._remove_resource_internal(resource_id, project=stored_project) + + fire_on_removed(self._lifecycle_hook, resource_id, stored_project) + log.debug("indexer.remove_resource.done") + + def _remove_resource_internal( + self, + resource_id: str, + *, + project: str, + ) -> list[str]: + """Remove resource data from backends (no lifecycle events). + + Caller MUST hold the per-resource lock. Backend calls run + outside the global lock; tracking-dict cleanup acquires the + global lock briefly at the end. + + Returns: + List of error messages from partial backend failures + (empty on full success). + """ + resource_uri = f"uko://resource/{resource_id}" + + with self._lock: + tracked_subjects = self._resource_subjects.get(resource_id, set()).copy() + + # Each backend call is individually guarded so a failure in + # one does not skip cleanup of others. + errors: list[str] = [] + for subj in tracked_subjects: + try: + self._graph_backend.remove_triples( + project, + subject=subj, + predicate="uko:sourceResource", + obj=resource_uri, + ) + except Exception as exc: + errors.append( + f"graph remove provenance {subj}: {type(exc).__name__}: {exc}" + ) + try: + self._graph_backend.remove_triples( + project, + subject=subj, + predicate=None, + obj=None, + ) + except Exception as exc: + errors.append(f"graph remove bulk {subj}: {type(exc).__name__}: {exc}") + + if self._text_backend is not None: + try: + self._text_backend.remove_document(project, resource_uri) + except Exception as exc: + errors.append(f"text remove: {type(exc).__name__}: {exc}") + if self._vector_backend is not None: + try: + self._vector_backend.remove_embedding(project, resource_uri) + except Exception as exc: + errors.append(f"vector remove: {type(exc).__name__}: {exc}") + + if errors: + logger.warning( + "indexer.remove_resource.partial_failure", + resource_id=resource_id, + errors=errors, + ) + + # Clean up tracking dicts. Per-resource locks are intentionally + # retained — popping while held creates a race condition. + with self._lock: + self._resource_subjects.pop(resource_id, None) + self._indexed_resources.pop(resource_id, None) + self._resource_analyzer.pop(resource_id, None) + + return errors + + def reindex_resource( + self, + resource: Resource, + *, + project: str, + ) -> IndexResult: + """Re-index a resource (remove old data, then re-index). + + Note: ``indexed_resource_count`` may transiently dip by one + during the remove-then-reindex window — this is expected. + + Args: + resource: The resource to re-index. + project: Namespaced project name. + + Returns: + :class:`IndexResult` summarizing the operation. + + Raises: + ValueError: If *project* is empty or whitespace-only. + """ + if not project or not project.strip(): + raise ValueError("project must be a non-empty string") + + log = logger.bind( + resource_id=resource.resource_id, + project=project, + ) + log.debug("indexer.reindex_resource.start") + + res_lock = self._resource_lock(resource.resource_id) + with res_lock: + with self._lock: + if resource.resource_id in self._indexed_resources: + old_project = self._indexed_resources[resource.resource_id] + else: + old_project = None + reindex_errors: list[str] = [] + if old_project is not None: + reindex_errors = self._remove_resource_internal( + resource.resource_id, project=old_project + ) + return self._index_resource_core( + resource, project, log, prior_errors=reindex_errors + ) + + +# --------------------------------------------------------------------------- +# Module exports +# --------------------------------------------------------------------------- + +__all__: list[str] = [ + "DEFAULT_MAX_TRIPLES", + "UKOIndexer", +] diff --git a/src/cleveragents/application/services/uko_indexer_internals.py b/src/cleveragents/application/services/uko_indexer_internals.py new file mode 100644 index 000000000..56cd16a9f --- /dev/null +++ b/src/cleveragents/application/services/uko_indexer_internals.py @@ -0,0 +1,293 @@ +"""UKO Indexer internal helpers — backend indexing and lifecycle hooks. + +Extracted from ``uko_indexer.py`` to comply with the 500-line file limit +(CONTRIBUTING.md). All functions in this module are implementation +details; callers should use :class:`UKOIndexer` directly. +""" + +from __future__ import annotations + +from datetime import datetime + +import structlog + +from cleveragents.domain.models.acms.analyzers import UKOTriple +from cleveragents.domain.models.acms.index_backends import ( + GraphIndexBackend, + TextIndexBackend, + VectorIndexBackend, +) +from cleveragents.domain.models.acms.provenance import ( + IndexResult, + ProvenancedTriple, + ProvenanceMetadata, +) +from cleveragents.domain.models.core.resource import Resource + +from .uko_indexer_protocols import IndexLifecycleHook + +logger = structlog.get_logger(__name__) + + +# --------------------------------------------------------------------------- +# Lifecycle hook helpers +# --------------------------------------------------------------------------- + + +def fire_on_indexed( + hook: IndexLifecycleHook, + result: IndexResult, +) -> None: + """Fire *on_indexed* lifecycle hook, guarding against failures.""" + try: + hook.on_indexed(result) + except Exception: + logger.warning( + "indexer.lifecycle_hook_error", + hook="on_indexed", + resource_id=result.resource_id, + exc_info=True, + ) + + +def fire_on_removed( + hook: IndexLifecycleHook, + resource_id: str, + project: str, +) -> None: + """Fire *on_removed* lifecycle hook, guarding against failures.""" + try: + hook.on_removed(resource_id, project) + except Exception: + logger.warning( + "indexer.lifecycle_hook_error", + hook="on_removed", + resource_id=resource_id, + exc_info=True, + ) + + +def fire_on_error( + hook: IndexLifecycleHook, + resource_id: str, + error: str, +) -> None: + """Fire *on_error* lifecycle hook, guarding against failures.""" + try: + hook.on_error(resource_id, error) + except Exception: + logger.warning( + "indexer.lifecycle_hook_error", + hook="on_error", + resource_id=resource_id, + exc_info=True, + ) + + +# --------------------------------------------------------------------------- +# Provenance +# --------------------------------------------------------------------------- + + +def attach_provenance( + triples: list[UKOTriple], + resource: Resource, + timestamp: datetime, +) -> list[ProvenancedTriple]: + """Wrap each triple with provenance metadata.""" + provenance = ProvenanceMetadata( + source_resource=resource.resource_id, + source_path=resource.location or "", + valid_from=timestamp, + is_current=True, + ) + return [ProvenancedTriple(triple=t, provenance=provenance) for t in triples] + + +# --------------------------------------------------------------------------- +# Backend indexing helpers +# --------------------------------------------------------------------------- + + +def index_graph( + graph_backend: GraphIndexBackend, + project: str, + resource: Resource, + provenanced: list[ProvenancedTriple], + errors: list[str], + resource_uri: str, +) -> tuple[int, set[str]]: + """Store provenanced triples in the graph backend. + + Returns: + A ``(stored_count, subjects)`` tuple. + """ + subjects: set[str] = set() + stored = 0 + for pt in provenanced: + t = pt.triple + obj = t.object_uri if t.object_uri else t.object_value + if not obj: + continue + try: + graph_backend.add_triple( + project, + t.subject_uri, + t.predicate, + obj, + ) + except Exception as exc: + errors.append(f"Graph backend error: {type(exc).__name__}") + logger.warning( + "indexer.graph_error", + project=project, + subject=t.subject_uri, + error=str(exc), + ) + continue + + # Data triple succeeded — count it and track subject. + stored += 1 + subjects.add(t.subject_uri) + + # When both object_uri and object_value are set (e.g. a + # URI-identified resource with a literal label), store the + # literal as a separate ``rdfs:label`` triple so no data is + # silently discarded. Track the object_uri as a subject so + # it's cleaned up on resource removal. + if t.object_uri and t.object_value: + try: + graph_backend.add_triple( + project, + t.object_uri, + "rdfs:label", + t.object_value, + ) + subjects.add(t.object_uri) + except Exception as exc: + errors.append(f"Graph label error: {type(exc).__name__}") + + # Best-effort provenance metadata — failures do not affect + # the stored count since the data triple already persisted. + # Per spec Provenance Contract: sourceResource, sourcePath, + # sourceRange, validFrom, isCurrent, and confidence. + prov = pt.provenance + prov_triples: list[tuple[str, str]] = [ + ("uko:sourceResource", resource_uri), + ] + if prov.source_path: + prov_triples.append(("uko:sourcePath", prov.source_path)) + if prov.source_range: + prov_triples.append(("uko:sourceRange", prov.source_range)) + prov_triples.append(("uko:validFrom", prov.valid_from.isoformat())) + prov_triples.append(("uko:isCurrent", str(prov.is_current).lower())) + if t.confidence < 1.0: + prov_triples.append(("uko:confidence", str(t.confidence))) + for pred, obj_val in prov_triples: + try: + graph_backend.add_triple( + project, + t.subject_uri, + pred, + obj_val, + ) + except Exception as exc: + errors.append(f"Graph provenance error ({pred}): {type(exc).__name__}") + return stored, subjects + + +def index_text( + text_backend: TextIndexBackend | None, + project: str, + resource: Resource, + content: str, + errors: list[str], + resource_uri: str, +) -> int: + """Index resource content in the text backend. + + Returns: + Number of documents indexed (0 or 1). + """ + if text_backend is None: + return 0 + try: + text_backend.index_document( + project, + resource_uri, + content, + { + "resource_id": resource.resource_id, + "location": resource.location or "", + "resource_type": resource.resource_type_name, + }, + ) + return 1 + except Exception as exc: + error_msg = f"Text backend error: {type(exc).__name__}" + errors.append(error_msg) + logger.warning( + "indexer.text_error", + project=project, + resource_id=resource.resource_id, + error=str(exc), + ) + return 0 + + +def index_vector( + vector_backend: VectorIndexBackend | None, + project: str, + resource: Resource, + content: str, + errors: list[str], + resource_uri: str, +) -> int: + """Index resource embedding in the vector backend. + + Returns: + Number of embeddings indexed (0 or 1). + """ + if vector_backend is None: + return 0 + # TODO(#578): integrate real embedding model — placeholder vector + # avoids leaking content size metadata by using a constant. + # See docs/reference/uko_indexer.md § Known Limitations. + placeholder_embedding = [1.0] + try: + vector_backend.index_embedding( + project, + resource_uri, + placeholder_embedding, + { + "resource_id": resource.resource_id, + "location": resource.location or "", + "resource_type": resource.resource_type_name, + }, + ) + return 1 + except Exception as exc: + error_msg = f"Vector backend error: {type(exc).__name__}" + errors.append(error_msg) + logger.warning( + "indexer.vector_error", + project=project, + resource_id=resource.resource_id, + error=str(exc), + ) + return 0 + + +# --------------------------------------------------------------------------- +# Module exports +# --------------------------------------------------------------------------- + +__all__: list[str] = [ + "attach_provenance", + "fire_on_error", + "fire_on_indexed", + "fire_on_removed", + "index_graph", + "index_text", + "index_vector", +] diff --git a/src/cleveragents/application/services/uko_indexer_protocols.py b/src/cleveragents/application/services/uko_indexer_protocols.py new file mode 100644 index 000000000..bacf93ac3 --- /dev/null +++ b/src/cleveragents/application/services/uko_indexer_protocols.py @@ -0,0 +1,264 @@ +"""Protocols and default implementations for UKO Indexer. + +Defines the dependency-injection protocols used by :class:`UKOIndexer`: + +- :class:`ContentReader` — reads resource content (decouples from FS) +- :class:`IndexLifecycleHook` — callbacks for index lifecycle events +- :class:`DefaultLifecycleHook` — structlog-based default hook +- :class:`LocationContentReader` — filesystem-based default reader + +Based on ``docs/specification.md`` > ACMS > Real-time Index +Synchronization (issue #578). +""" + +from __future__ import annotations + +import os +import stat +from pathlib import Path +from typing import Protocol, runtime_checkable + +import structlog + +from cleveragents.domain.models.acms.provenance import IndexResult +from cleveragents.domain.models.core.resource import Resource + +logger = structlog.get_logger(__name__) + + +# --------------------------------------------------------------------------- +# Content reader protocol (DI — decouples UKOIndexer from filesystem) +# --------------------------------------------------------------------------- + + +@runtime_checkable +class ContentReader(Protocol): + """Protocol for reading resource content. + + Implementations may read from the local filesystem, a sandbox, + a VCS checkout, or an in-memory buffer. + """ + + def read_content(self, resource: Resource) -> str: + """Read the textual content of *resource*. + + Args: + resource: The resource to read. + + Returns: + The resource content as a string. + + Raises: + OSError: If the resource cannot be read. + ValueError: If the resource location is invalid, the + resolved path escapes the base directory, or content + exceeds the maximum size. + """ + raise NotImplementedError + + +# --------------------------------------------------------------------------- +# Index lifecycle hook protocol +# --------------------------------------------------------------------------- + + +@runtime_checkable +class IndexLifecycleHook(Protocol): + """Callback protocol for index lifecycle events. + + Implementations can react to indexing events for logging, + metrics, or triggering downstream processing. + """ + + def on_indexed(self, result: IndexResult) -> None: + """Called after a resource is successfully indexed. + + Args: + result: Summary of the indexing operation. + """ + raise NotImplementedError + + def on_removed(self, resource_id: str, project: str) -> None: + """Called after a resource is removed from indices. + + Args: + resource_id: ULID of the removed resource. + project: Namespaced project name. + """ + raise NotImplementedError + + def on_error( + self, + resource_id: str, + error: str, + ) -> None: + """Called when an indexing operation encounters an error. + + .. note:: + + **Design choice**: ``on_error`` is only fired for errors + that abort the *pipeline* (e.g. content-read or analyzer + failures). Individual backend failures (graph, text, + vector) are recorded in :attr:`IndexResult.errors` but do + **not** trigger ``on_error``, because the pipeline + continues with graceful degradation. + + Args: + resource_id: ULID of the resource. + error: Error description. + """ + raise NotImplementedError + + +class DefaultLifecycleHook: + """Default lifecycle hook that logs events via structlog.""" + + def on_indexed(self, result: IndexResult) -> None: + """Log successful indexing.""" + logger.info( + "indexer.resource_indexed", + resource_id=result.resource_id, + triple_count=result.triple_count, + analyzer=result.analyzer_domain, + ) + + def on_removed(self, resource_id: str, project: str) -> None: + """Log resource removal.""" + logger.info( + "indexer.resource_removed", + resource_id=resource_id, + project=project, + ) + + def on_error(self, resource_id: str, error: str) -> None: + """Log indexing error.""" + logger.warning( + "indexer.error", + resource_id=resource_id, + error=error, + ) + + +# --------------------------------------------------------------------------- +# LocationContentReader — default filesystem reader +# --------------------------------------------------------------------------- + + +#: Default maximum content size in **characters** (≈10M characters). +#: For multi-byte UTF-8 content the actual RAM footprint may exceed +#: 10 MB. Use a smaller value for CJK / emoji-heavy corpora. +DEFAULT_MAX_CONTENT_SIZE: int = 10 * 1024 * 1024 + + +class LocationContentReader: + """Reads resource content from the filesystem via ``resource.location``. + + This is the default :class:`ContentReader` implementation. It reads + the file at ``resource.location`` and returns its text content. + + Security: validates the resolved (symlink-followed) path against + a configurable *base_dir*. Rejects paths that escape the base + directory via symlinks, ``..`` components, or absolute paths. + Enforces a maximum content size (measured in **characters**, not + bytes) to prevent OOM on unexpectedly large files. Rejects + non-regular files (named pipes, device files). + """ + + def __init__( + self, + *, + max_content_size: int = DEFAULT_MAX_CONTENT_SIZE, + base_dir: Path | None = None, + ) -> None: + if max_content_size < 1: + raise ValueError("max_content_size must be positive") + self._max_content_size = max_content_size + self._base_dir = base_dir.resolve() if base_dir is not None else None + if self._base_dir is None: + logger.warning( + "content_reader.no_base_dir", + hint="No base_dir set — reader can access any file the " + "process can read. Set base_dir in production.", + ) + + def read_content(self, resource: Resource) -> str: + """Read file content from ``resource.location``. + + Args: + resource: Must have a non-``None`` ``location``. + + Returns: + File content as a UTF-8 string. + + Raises: + ValueError: If ``resource.location`` is ``None``, the + resolved path escapes the base directory, or content + exceeds max size. + OSError: If the file cannot be read. + """ + if resource.location is None: + raise ValueError(f"Resource {resource.resource_id} has no location") + + raw = Path(resource.location) + # Reject path traversal attempts (pre-resolution check) + if ".." in raw.parts: + raise ValueError(f"Path traversal rejected: {resource.location}") + + resolved = raw.resolve() + + # Validate resolved path is under base_dir (post-resolution check + # catches symlink escapes and absolute paths). + if self._base_dir is not None: + try: + resolved.relative_to(self._base_dir) + except ValueError: + raise ValueError( + f"Path escapes base directory: {resource.location}" + ) from None + + # Open with O_NONBLOCK then fstat — eliminates the TOCTOU + # window between a pre-open is_file() check and the actual + # open(), and avoids blocking on FIFOs/named pipes. + fd = -1 + try: + fd = os.open(str(resolved), os.O_RDONLY | os.O_NONBLOCK) + mode = os.fstat(fd).st_mode + if not stat.S_ISREG(mode): + raise ValueError(f"Not a regular file: {resource.location}") + finally: + if fd >= 0: + os.close(fd) + fd = -1 + + # Re-open in text mode for proper UTF-8 decoding (the raw fd + # was opened non-blocking only for the fstat check). + with open(resolved, encoding="utf-8") as fh: + content = fh.read(self._max_content_size + 1) + + if len(content) > self._max_content_size: + raise ValueError( + f"Resource {resource.resource_id} exceeds max content " + f"size ({self._max_content_size} characters)" + ) + return content + + +# --------------------------------------------------------------------------- +# Protocol compliance assertions (Pyright static verification) +# --------------------------------------------------------------------------- + +_assert_hook: type[IndexLifecycleHook] = DefaultLifecycleHook +_assert_reader: type[ContentReader] = LocationContentReader + + +# --------------------------------------------------------------------------- +# Module exports +# --------------------------------------------------------------------------- + +__all__: list[str] = [ + "DEFAULT_MAX_CONTENT_SIZE", + "ContentReader", + "DefaultLifecycleHook", + "IndexLifecycleHook", + "LocationContentReader", +] diff --git a/src/cleveragents/domain/models/acms/__init__.py b/src/cleveragents/domain/models/acms/__init__.py index d7dfb3ab5..88b8b8285 100644 --- a/src/cleveragents/domain/models/acms/__init__.py +++ b/src/cleveragents/domain/models/acms/__init__.py @@ -71,6 +71,23 @@ Ontology registry - ``OntologyRegistry`` functions -- Domain lookup, Layer 1 listing, DetailLevelMap chain building, Turtle validation +Index backend types (from :mod:`~cleveragents.domain.models.acms.index_backends`): +- ``TextIndexBackend`` -- Write-side full-text indexing protocol +- ``VectorIndexBackend`` -- Write-side vector embedding indexing protocol +- ``GraphIndexBackend`` -- Write-side UKO triple storage protocol +- ``IndexedDocument`` -- Record of an indexed document +- ``SearchResult`` -- Result from index backend search + +Index stubs (from :mod:`~cleveragents.domain.models.acms.index_stubs`): +- ``InMemoryTextIndexBackend`` -- In-memory text indexing stub +- ``InMemoryVectorIndexBackend`` -- In-memory vector indexing stub +- ``InMemoryGraphIndexBackend`` -- In-memory graph indexing stub + +Provenance types (from :mod:`~cleveragents.domain.models.acms.provenance`): +- ``ProvenanceMetadata`` -- Source-tracking metadata for triples +- ``ProvenancedTriple`` -- UKOTriple + provenance composite +- ``IndexResult`` -- Summary of an indexing operation + Based on ``docs/specification.md`` ACMS / CRP sections and ADR-014. """ @@ -100,6 +117,18 @@ from cleveragents.domain.models.acms.crp import ( from cleveragents.domain.models.acms.docker_compose_analyzer import ( DockerComposeAnalyzer, ) +from cleveragents.domain.models.acms.index_backends import ( + GraphIndexBackend, + IndexedDocument, + SearchResult, + TextIndexBackend, + VectorIndexBackend, +) +from cleveragents.domain.models.acms.index_stubs import ( + InMemoryGraphIndexBackend, + InMemoryTextIndexBackend, + InMemoryVectorIndexBackend, +) from cleveragents.domain.models.acms.markdown_analyzer import MarkdownAnalyzer from cleveragents.domain.models.acms.ontology_registry import ( DomainDescriptor, @@ -112,6 +141,11 @@ from cleveragents.domain.models.acms.ontology_registry import ( validate_turtle_file, ) from cleveragents.domain.models.acms.postgresql_analyzer import PostgreSQLAnalyzer +from cleveragents.domain.models.acms.provenance import ( + IndexResult, + ProvenancedTriple, + ProvenanceMetadata, +) from cleveragents.domain.models.acms.python_analyzer import PythonAnalyzer from cleveragents.domain.models.acms.scope_resolution import ( ResourceAliasResolver, @@ -177,20 +211,29 @@ __all__: list[str] = [ "DomainDescriptor", "FragmentProvenance", "GraphBackend", + "GraphIndexBackend", "GraphResult", "InMemoryGraphBackend", + "InMemoryGraphIndexBackend", "InMemoryTextBackend", + "InMemoryTextIndexBackend", "InMemoryVectorBackend", + "InMemoryVectorIndexBackend", + "IndexResult", + "IndexedDocument", "MarkdownAnalyzer", "PlanContext", "PlanDecisionContextStrategy", "PostgreSQLAnalyzer", + "ProvenanceMetadata", + "ProvenancedTriple", "PythonAnalyzer", "ResourceAliasResolver", "ResourceScope", "ScopeViolationError", "ScopedBackendSet", "ScopedBackendView", + "SearchResult", "SemanticEmbeddingStrategy", "SimpleKeywordStrategy", "StrategyCapabilities", @@ -198,6 +241,7 @@ __all__: list[str] = [ "StrategyRegistryEntry", "TemporalArchaeologyStrategy", "TextBackend", + "TextIndexBackend", "TextResult", "TierBudget", "TierMetrics", @@ -205,6 +249,7 @@ __all__: list[str] = [ "TurtleValidationError", "UKOTriple", "VectorBackend", + "VectorIndexBackend", "VectorResult", "build_detail_map_chain", "create_scoped_backend_set", diff --git a/src/cleveragents/domain/models/acms/analyzers.py b/src/cleveragents/domain/models/acms/analyzers.py index 1a12df2df..daca09684 100644 --- a/src/cleveragents/domain/models/acms/analyzers.py +++ b/src/cleveragents/domain/models/acms/analyzers.py @@ -22,13 +22,16 @@ Language-Specific Analyzers, and configuration key from __future__ import annotations -import logging import re +from pathlib import PurePosixPath from typing import Protocol, runtime_checkable -from pydantic import BaseModel, ConfigDict, Field, field_validator +import structlog +from pydantic import BaseModel, ConfigDict, Field, model_validator -logger = logging.getLogger(__name__) +from cleveragents.domain.models.core.resource import Resource + +logger = structlog.get_logger(__name__) # --------------------------------------------------------------------------- # UKOTriple @@ -72,20 +75,18 @@ class UKOTriple(BaseModel): model_config = ConfigDict(frozen=True, str_strip_whitespace=True) # -- Validators ----------------------------------------------------------- + # NOTE: subject_uri and predicate are validated by ``min_length=1`` + # on their ``Field()`` definitions, so explicit ``@field_validator`` + # methods are unnecessary. - @field_validator("subject_uri") - @classmethod - def _validate_subject_uri(cls, value: str) -> str: - if not value: - raise ValueError("subject_uri must not be empty.") - return value - - @field_validator("predicate") - @classmethod - def _validate_predicate(cls, value: str) -> str: - if not value: - raise ValueError("predicate must not be empty.") - return value + @model_validator(mode="after") + def _validate_object(self) -> UKOTriple: + """Enforce that at least one of object_uri/object_value is set.""" + if not self.object_uri and not self.object_value: + raise ValueError( + "At least one of object_uri or object_value must be non-empty." + ) + return self # --------------------------------------------------------------------------- @@ -99,11 +100,11 @@ def safe_uri_segment(text: str) -> str: """Sanitise *text* for use in a UKO URI path segment. Replaces non-alphanumeric characters (except ``_``, ``.``, ``-``) - with underscores, strips leading/trailing underscores, and - truncates to 120 characters. Returns ``"_unknown_"`` if the - result is empty. + with underscores, truncates to 120 characters, then strips + leading/trailing underscores from the truncated result. Returns + ``"_unknown_"`` if the result is empty. """ - result = _SAFE_URI_RE.sub("_", text).strip("_")[:120] + result = _SAFE_URI_RE.sub("_", text)[:120].strip("_") return result if result else "_unknown_" @@ -128,12 +129,12 @@ class AnalyzerProtocol(Protocol): @property def supported_extensions(self) -> frozenset[str]: """File extensions this analyzer handles (e.g. ``{".py"}``).""" - ... # pragma: no cover + raise NotImplementedError @property def domain(self) -> str: """Human-readable domain label (e.g. ``"python"``).""" - ... # pragma: no cover + raise NotImplementedError def analyze(self, content: str, resource_uri: str) -> list[UKOTriple]: """Parse *content* and return UKO triples. @@ -145,7 +146,7 @@ class AnalyzerProtocol(Protocol): Returns: List of ``UKOTriple`` instances extracted from *content*. """ - ... # pragma: no cover + raise NotImplementedError # --------------------------------------------------------------------------- @@ -157,27 +158,34 @@ class AnalyzerRegistry: """In-memory registry mapping file extensions to analyzer instances. Analyzers are registered via :meth:`register` and looked up by file - extension via :meth:`get_for_extension`. Multiple analyzers may - handle the same extension; the *first* registered wins. + extension via :meth:`get_for_extension`. When multiple analyzers + handle the same extension, the one with the highest *priority* wins + (higher numeric value = higher priority). Equal-priority ties are + resolved by registration order (first registered wins). Example:: registry = AnalyzerRegistry() - registry.register(PythonAnalyzer()) + registry.register(PythonAnalyzer(), priority=10) analyzer = registry.get_for_extension(".py") """ def __init__(self) -> None: - self._by_extension: dict[str, AnalyzerProtocol] = {} + # Maps extension -> (priority, analyzer); highest priority wins. + self._by_extension: dict[str, tuple[int, AnalyzerProtocol]] = {} self._all: list[AnalyzerProtocol] = [] # -- Registration --------------------------------------------------------- - def register(self, analyzer: AnalyzerProtocol) -> None: + def register(self, analyzer: AnalyzerProtocol, *, priority: int = 0) -> None: """Register an analyzer for all its supported extensions. Args: analyzer: An object satisfying ``AnalyzerProtocol``. + priority: Numeric priority (higher wins). When two + analyzers compete for the same extension, the one + with the greater *priority* is kept. Equal-priority + ties preserve the first-registered analyzer. Raises: TypeError: If *analyzer* does not satisfy the protocol. @@ -192,19 +200,38 @@ class AnalyzerRegistry: "supported extensions." ) for ext in extensions: - if ext not in self._by_extension: - self._by_extension[ext] = analyzer - logger.debug( - "Registered analyzer %s for extension '%s'", - type(analyzer).__name__, - ext, + if not ext.startswith("."): + raise ValueError( + f"Extension '{ext}' must start with a leading dot " + f"(e.g. '.py'), got from {type(analyzer).__name__}" ) + existing = self._by_extension.get(ext) + if existing is None: + self._by_extension[ext] = (priority, analyzer) + logger.debug( + "analyzer.registered", + analyzer=type(analyzer).__name__, + extension=ext, + priority=priority, + ) + elif priority > existing[0]: + logger.debug( + "analyzer.superseded", + new_analyzer=type(analyzer).__name__, + new_priority=priority, + old_analyzer=type(existing[1]).__name__, + old_priority=existing[0], + extension=ext, + ) + self._by_extension[ext] = (priority, analyzer) else: logger.debug( - "Extension '%s' already handled by %s; skipping %s", - ext, - type(self._by_extension[ext]).__name__, - type(analyzer).__name__, + "analyzer.skipped", + extension=ext, + existing_analyzer=type(existing[1]).__name__, + existing_priority=existing[0], + skipped_analyzer=type(analyzer).__name__, + skipped_priority=priority, ) self._all.append(analyzer) @@ -213,6 +240,9 @@ class AnalyzerRegistry: def get_for_extension(self, extension: str) -> AnalyzerProtocol | None: """Return the analyzer registered for *extension*, or ``None``. + When multiple analyzers were registered for the same extension, + the one with the highest priority is returned. + Args: extension: File extension including leading dot (e.g. ``".py"``). @@ -222,7 +252,34 @@ class AnalyzerRegistry: """ if not extension: return None - return self._by_extension.get(extension) + entry = self._by_extension.get(extension) + return entry[1] if entry is not None else None + + def get_for_resource(self, resource: Resource) -> AnalyzerProtocol | None: + """Return the best analyzer for *resource*, or ``None``. + + Determines the file extension from ``resource.location`` and + delegates to :meth:`get_for_extension`. If the resource has no + location, returns ``None``. + + This method implements the spec's + ``analyzers.get_for_resource(resource)`` pattern + (``specification.md`` ~line 43214). + + Args: + resource: A :class:`Resource` domain model instance. + + Returns: + The registered ``AnalyzerProtocol`` for the resource's + file extension, or ``None`` if no analyzer matches. + """ + location = resource.location + if not location: + return None + ext = PurePosixPath(location).suffix + if not ext: + return None + return self.get_for_extension(ext) # -- Listing -------------------------------------------------------------- @@ -237,3 +294,15 @@ class AnalyzerRegistry: def __len__(self) -> int: """Return the number of registered analyzers.""" return len(self._all) + + +# --------------------------------------------------------------------------- +# Module exports +# --------------------------------------------------------------------------- + +__all__: list[str] = [ + "AnalyzerProtocol", + "AnalyzerRegistry", + "UKOTriple", + "safe_uri_segment", +] diff --git a/src/cleveragents/domain/models/acms/index_backends.py b/src/cleveragents/domain/models/acms/index_backends.py new file mode 100644 index 000000000..c26b8a734 --- /dev/null +++ b/src/cleveragents/domain/models/acms/index_backends.py @@ -0,0 +1,351 @@ +"""Index backend protocols for the ACMS indexing layer. + +Defines the **write-side** backend protocols used by :class:`UKOIndexer` +to persist analysed content into text, vector, and graph stores. These +are distinct from the **read-side** query protocols in ``backends.py`` +(``TextBackend``, ``VectorBackend``, ``GraphBackend``) which are used by +context strategies at query time. + +Each protocol is ``@runtime_checkable`` so that DI containers can verify +implementations satisfy the contract at registration time. + ++-------------------------+-----------------------------+ +| Protocol | Purpose | ++=========================+=============================+ +| ``TextIndexBackend`` | Full-text document indexing | +| ``VectorIndexBackend`` | Embedding vector indexing | +| ``GraphIndexBackend`` | UKO triple storage | ++-------------------------+-----------------------------+ + +Based on ``docs/specification.md`` > ACMS > Custom Index Backends +and Real-time Index Synchronization. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Protocol, runtime_checkable + +# --------------------------------------------------------------------------- +# Result types +# --------------------------------------------------------------------------- + + +@dataclass(frozen=True) +class IndexedDocument: + """Record of a document successfully indexed in a text backend. + + Attributes: + project: Namespaced project name that owns the document. + doc_id: Unique document identifier (typically a UKO URI). + char_count: Number of characters indexed. + """ + + project: str + doc_id: str + char_count: int = 0 + + def __post_init__(self) -> None: + if not self.project or not self.project.strip(): + raise ValueError("project must be a non-empty string") + if not self.doc_id or not self.doc_id.strip(): + raise ValueError("doc_id must be a non-empty string") + if self.char_count < 0: + raise ValueError(f"char_count must be non-negative, got {self.char_count}") + + +@dataclass(frozen=True) +class SearchResult: + """Result returned by :meth:`TextIndexBackend.search`. + + Attributes: + doc_id: Document identifier. + content: Matched text snippet. + score: Relevance score in ``[0.0, 1.0]``. + metadata: Backend-specific metadata. + """ + + doc_id: str + content: str + score: float + metadata: dict[str, str] = field(default_factory=dict) + + def __post_init__(self) -> None: + if not self.doc_id or not self.doc_id.strip(): + raise ValueError("doc_id must be a non-empty string") + if not (0.0 <= self.score <= 1.0): + raise ValueError(f"score must be between 0.0 and 1.0, got {self.score}") + + +# --------------------------------------------------------------------------- +# TextIndexBackend +# --------------------------------------------------------------------------- + + +@runtime_checkable +class TextIndexBackend(Protocol): + """Protocol for full-text document indexing backends. + + Write-side counterpart to :class:`TextBackend` (query-side). + Implementations wrap stores such as Tantivy, SQLite FTS, or + Elasticsearch for indexing documents produced by analyzers. + + Based on ``specification.md`` > ACMS > Custom Index Backends. + """ + + def index_document( + self, + project: str, + doc_id: str, + content: str, + metadata: dict[str, str], + ) -> IndexedDocument: + """Index a document for full-text search. + + Args: + project: Namespaced project name (e.g. ``"local/my-app"``). + doc_id: Unique document identifier (typically a UKO URI). + content: Full text content to index. + metadata: Key-value metadata to store alongside the document. + + Returns: + Record of the indexed document. + + Raises: + ValueError: If *project*, *doc_id*, or *content* is empty + or whitespace-only. + """ + ... # pragma: no cover + + def search( + self, + project: str, + query: str, + limit: int = 20, + ) -> list[SearchResult]: + """Search indexed documents for *query*. + + Args: + project: Namespaced project name to search within. + query: Search query string. + limit: Maximum number of results. Must be positive. + + Returns: + List of :class:`SearchResult` ordered by descending score. + + Raises: + ValueError: If *project* or *query* is empty or + whitespace-only, or *limit* < 1. + """ + ... # pragma: no cover + + def remove_document( + self, + project: str, + doc_id: str, + ) -> None: + """Remove a document from the index. + + Args: + project: Namespaced project name. + doc_id: Document identifier to remove. + + Raises: + ValueError: If *project* or *doc_id* is empty or + whitespace-only. + """ + ... # pragma: no cover + + def rebuild_index(self, project: str) -> None: + """Rebuild the full-text index for *project*. + + Drops and recreates the index. Used during maintenance + reindex operations. + + Args: + project: Namespaced project name. + + Raises: + ValueError: If *project* is empty or whitespace-only. + """ + ... # pragma: no cover + + +# --------------------------------------------------------------------------- +# VectorIndexBackend +# --------------------------------------------------------------------------- + + +@runtime_checkable +class VectorIndexBackend(Protocol): + """Protocol for vector embedding indexing backends. + + Write-side counterpart to :class:`VectorBackend` (query-side). + Implementations wrap stores such as FAISS, Qdrant, or Weaviate. + + Based on ``specification.md`` > ACMS > Custom Index Backends. + """ + + def index_embedding( + self, + project: str, + doc_id: str, + embedding: list[float], + metadata: dict[str, str], + ) -> None: + """Index a vector embedding. + + Args: + project: Namespaced project name. + doc_id: Unique document identifier. + embedding: Embedding vector. Must be non-empty. + metadata: Key-value metadata. + + Raises: + ValueError: If *project* or *doc_id* is empty or + whitespace-only, or *embedding* is empty. + """ + ... # pragma: no cover + + def search_similar( + self, + project: str, + query_embedding: list[float], + limit: int = 20, + min_relevance: float = 0.0, + ) -> list[SearchResult]: + """Find vectors most similar to *query_embedding*. + + Args: + project: Namespaced project name. + query_embedding: Query embedding vector. + limit: Maximum number of results. Must be positive. + min_relevance: Minimum similarity threshold in + ``[0.0, 1.0]``. + + Returns: + List of :class:`SearchResult` ordered by descending score. + + Raises: + ValueError: If *project* is empty or whitespace-only, + *query_embedding* is empty, or *limit* < 1. + """ + ... # pragma: no cover + + def remove_embedding( + self, + project: str, + doc_id: str, + ) -> None: + """Remove an embedding from the index. + + Args: + project: Namespaced project name. + doc_id: Document identifier to remove. + + Raises: + ValueError: If *project* or *doc_id* is empty or + whitespace-only. + """ + ... # pragma: no cover + + +# --------------------------------------------------------------------------- +# GraphIndexBackend +# --------------------------------------------------------------------------- + + +@runtime_checkable +class GraphIndexBackend(Protocol): + """Protocol for knowledge-graph indexing backends. + + Write-side counterpart to :class:`GraphBackend` (query-side). + Implementations wrap stores such as Blazegraph, Neo4j, or RDFLib. + + Based on ``specification.md`` > ACMS > Custom Index Backends. + """ + + def add_triple( + self, + project: str, + subject: str, + predicate: str, + obj: str, + ) -> None: + """Add a single triple to the graph. + + Args: + project: Namespaced project name. + subject: Subject URI. + predicate: Predicate URI. + obj: Object URI or literal value. + + Raises: + ValueError: If *project*, *subject*, *predicate*, or *obj* + is empty or whitespace-only. + """ + ... # pragma: no cover + + def query( + self, + project: str, + sparql: str, + ) -> list[dict[str, str]]: + """Execute a SPARQL query against the graph. + + .. note:: + + Implementations **must** parameterise or sanitise the + *sparql* argument before passing it to the underlying store. + Never interpolate untrusted user input directly into the + query string. + + Args: + project: Namespaced project name. + sparql: SPARQL query string. + + Returns: + List of binding dictionaries. + + Raises: + ValueError: If *project* or *sparql* is empty or + whitespace-only. + """ + ... # pragma: no cover + + def remove_triples( + self, + project: str, + subject: str | None, + predicate: str | None, + obj: str | None, + ) -> None: + """Remove triples matching the given pattern. + + Pass ``None`` for any component to match all values for that + position (wildcard). + + Args: + project: Namespaced project name. + subject: Subject URI filter, or ``None`` for wildcard. + predicate: Predicate URI filter, or ``None`` for wildcard. + obj: Object filter, or ``None`` for wildcard. + + Raises: + ValueError: If *project* is empty or whitespace-only, or + all three filters are ``None`` (would delete everything). + """ + ... # pragma: no cover + + +# --------------------------------------------------------------------------- +# Module exports +# --------------------------------------------------------------------------- + +__all__: list[str] = [ + "GraphIndexBackend", + "IndexedDocument", + "SearchResult", + "TextIndexBackend", + "VectorIndexBackend", +] diff --git a/src/cleveragents/domain/models/acms/index_stubs.py b/src/cleveragents/domain/models/acms/index_stubs.py new file mode 100644 index 000000000..576e88b66 --- /dev/null +++ b/src/cleveragents/domain/models/acms/index_stubs.py @@ -0,0 +1,343 @@ +"""In-memory stub implementations for index backend protocols. + +These stubs satisfy :class:`TextIndexBackend`, +:class:`VectorIndexBackend`, and :class:`GraphIndexBackend` with +minimal in-memory implementations. They serve as: + +1. **Development placeholders** while physical store integrations + (Tantivy, FAISS, Blazegraph, etc.) are built — selected at + startup via DI container provider configuration. +2. **Reference implementations** documenting the expected behaviour + and argument validation of each protocol method. +3. **Test doubles** for indexer and pipeline tests without external + infrastructure (imported by ``features/mocks/`` and Robot helpers). + +Based on ``docs/specification.md`` > ACMS > Custom Index Backends. +""" + +from __future__ import annotations + +import structlog + +from cleveragents.domain.models.acms.index_backends import ( + GraphIndexBackend, + IndexedDocument, + SearchResult, + TextIndexBackend, + VectorIndexBackend, +) + +logger = structlog.get_logger(__name__) + +#: Default maximum entries per stub to prevent unbounded memory growth. +DEFAULT_MAX_STUB_ENTRIES: int = 100_000 + + +# --------------------------------------------------------------------------- +# Validation helpers +# --------------------------------------------------------------------------- + + +def _require_non_empty(value: str, name: str) -> None: + """Raise ``ValueError`` if *value* is empty or whitespace-only.""" + if not value or not value.strip(): + raise ValueError(f"{name} must be a non-empty string") + + +# --------------------------------------------------------------------------- +# InMemoryTextIndexBackend +# --------------------------------------------------------------------------- + + +class InMemoryTextIndexBackend: + """Stub :class:`TextIndexBackend` — dict-based, with max_entries cap.""" + + def __init__(self, *, max_entries: int = DEFAULT_MAX_STUB_ENTRIES) -> None: + self._docs: dict[tuple[str, str], tuple[str, dict[str, str]]] = {} + self._max_entries = max_entries + + def index_document( + self, + project: str, + doc_id: str, + content: str, + metadata: dict[str, str], + ) -> IndexedDocument: + """Index a document in the in-memory store.""" + _require_non_empty(project, "project") + _require_non_empty(doc_id, "doc_id") + _require_non_empty(content, "content") + if (project, doc_id) not in self._docs and len(self._docs) >= self._max_entries: + raise RuntimeError( + f"InMemoryTextIndexBackend: max_entries ({self._max_entries}) reached" + ) + self._docs[(project, doc_id)] = (content, dict(metadata)) + logger.debug( + "index.text.document_indexed", + project=project, + doc_id=doc_id, + char_count=len(content), + ) + return IndexedDocument( + project=project, + doc_id=doc_id, + char_count=len(content), + ) + + def search( + self, + project: str, + query: str, + limit: int = 20, + ) -> list[SearchResult]: + """Search documents via naive substring matching.""" + _require_non_empty(project, "project") + _require_non_empty(query, "query") + if limit < 1: + raise ValueError(f"limit must be positive, got {limit}") + results: list[SearchResult] = [] + query_lower = query.lower() + for (proj, doc_id), (content, meta) in self._docs.items(): + if proj != project: + continue + if query_lower in content.lower(): + results.append( + SearchResult( + doc_id=doc_id, + content=content[:200], + score=1.0, + metadata=dict(meta), + ) + ) + if len(results) >= limit: + break + return results + + def remove_document(self, project: str, doc_id: str) -> None: + """Remove a document from the in-memory store.""" + _require_non_empty(project, "project") + _require_non_empty(doc_id, "doc_id") + self._docs.pop((project, doc_id), None) + logger.debug("index.text.document_removed", project=project, doc_id=doc_id) + + def rebuild_index(self, project: str) -> None: + """Remove all documents for *project* (simulates rebuild).""" + _require_non_empty(project, "project") + keys_to_remove = [k for k in self._docs if k[0] == project] + for k in keys_to_remove: + del self._docs[k] + logger.debug( + "index.text.index_rebuilt", + project=project, + removed_count=len(keys_to_remove), + ) + + @property + def document_count(self) -> int: + """Total number of indexed documents (test helper).""" + return len(self._docs) + + +# --------------------------------------------------------------------------- +# InMemoryVectorIndexBackend +# --------------------------------------------------------------------------- + + +class InMemoryVectorIndexBackend: + """Stub :class:`VectorIndexBackend` — dict-based, with max_entries cap.""" + + def __init__(self, *, max_entries: int = DEFAULT_MAX_STUB_ENTRIES) -> None: + self._embeddings: dict[tuple[str, str], tuple[list[float], dict[str, str]]] = {} + self._max_entries = max_entries + + def index_embedding( + self, + project: str, + doc_id: str, + embedding: list[float], + metadata: dict[str, str], + ) -> None: + """Store an embedding in the in-memory store.""" + _require_non_empty(project, "project") + _require_non_empty(doc_id, "doc_id") + if not embedding: + raise ValueError("embedding must be a non-empty list") + if (project, doc_id) not in self._embeddings and len( + self._embeddings + ) >= self._max_entries: + raise RuntimeError( + f"InMemoryVectorIndexBackend: max_entries ({self._max_entries}) reached" + ) + self._embeddings[(project, doc_id)] = (list(embedding), dict(metadata)) + logger.debug( + "index.vector.embedding_indexed", + project=project, + doc_id=doc_id, + dimensions=len(embedding), + ) + + def search_similar( + self, + project: str, + query_embedding: list[float], + limit: int = 20, + min_relevance: float = 0.0, + ) -> list[SearchResult]: + """Return stored embeddings for *project* (no real similarity).""" + _require_non_empty(project, "project") + if not query_embedding: + raise ValueError("query_embedding must be a non-empty list") + if limit < 1: + raise ValueError(f"limit must be positive, got {limit}") + if not (0.0 <= min_relevance <= 1.0): + raise ValueError( + f"min_relevance must be in [0.0, 1.0], got {min_relevance}" + ) + results: list[SearchResult] = [] + for (proj, doc_id), (_emb, meta) in self._embeddings.items(): + if proj != project: + continue + score = 1.0 # stub always returns perfect score + if score < min_relevance: + continue + results.append( + SearchResult( + doc_id=doc_id, + content=f"embedding:{doc_id}", + score=score, + metadata=dict(meta), + ) + ) + if len(results) >= limit: + break + return results + + def remove_embedding(self, project: str, doc_id: str) -> None: + """Remove an embedding from the in-memory store.""" + _require_non_empty(project, "project") + _require_non_empty(doc_id, "doc_id") + self._embeddings.pop((project, doc_id), None) + logger.debug("index.vector.embedding_removed", project=project, doc_id=doc_id) + + @property + def embedding_count(self) -> int: + """Total number of stored embeddings (test helper).""" + return len(self._embeddings) + + +# --------------------------------------------------------------------------- +# InMemoryGraphIndexBackend +# --------------------------------------------------------------------------- + + +class InMemoryGraphIndexBackend: + """Stub :class:`GraphIndexBackend` — list-based, with max_entries cap.""" + + def __init__(self, *, max_entries: int = DEFAULT_MAX_STUB_ENTRIES) -> None: + self._triples: dict[str, list[tuple[str, str, str]]] = {} + self._triple_sets: dict[str, set[tuple[str, str, str]]] = {} + self._max_entries = max_entries + + def add_triple(self, project: str, subject: str, predicate: str, obj: str) -> None: + """Add a triple to the in-memory graph.""" + _require_non_empty(project, "project") + _require_non_empty(subject, "subject") + _require_non_empty(predicate, "predicate") + _require_non_empty(obj, "obj") + triple = (subject, predicate, obj) + if project not in self._triples: + self._triples[project] = [] + self._triple_sets[project] = set() + if triple not in self._triple_sets[project]: + if self.triple_count() >= self._max_entries: + raise RuntimeError( + f"InMemoryGraphIndexBackend: max_entries " + f"({self._max_entries}) reached" + ) + self._triple_sets[project].add(triple) + self._triples[project].append(triple) + logger.debug( + "index.graph.triple_added", + project=project, + subject=subject, + predicate=predicate, + ) + + def query(self, project: str, sparql: str) -> list[dict[str, str]]: + """Return all triples for *project* as binding dicts (stub).""" + _require_non_empty(project, "project") + _require_non_empty(sparql, "sparql") + triples = self._triples.get(project, []) + return [{"s": s, "p": p, "o": o} for s, p, o in triples] + + def remove_triples( + self, + project: str, + subject: str | None, + predicate: str | None, + obj: str | None, + ) -> None: + """Remove triples matching the given pattern.""" + _require_non_empty(project, "project") + if subject is None and predicate is None and obj is None: + raise ValueError( + "At least one of subject, predicate, obj must be " + "non-None to avoid deleting all triples" + ) + # Treat empty strings as programming errors — callers should + # pass None for wildcard, not "". + if subject is not None and not subject.strip(): + raise ValueError("subject filter must be non-empty if provided") + if predicate is not None and not predicate.strip(): + raise ValueError("predicate filter must be non-empty if provided") + if obj is not None and not obj.strip(): + raise ValueError("obj filter must be non-empty if provided") + if project not in self._triples: + return + original = self._triples[project] + filtered = [ + (s, p, o) + for s, p, o in original + if not ( + (subject is None or s == subject) + and (predicate is None or p == predicate) + and (obj is None or o == obj) + ) + ] + removed = len(original) - len(filtered) + self._triples[project] = filtered + self._triple_sets[project] = set(filtered) + logger.debug( + "index.graph.triples_removed", + project=project, + subject=subject, + predicate=predicate, + obj=obj, + removed_count=removed, + ) + + def triple_count(self, project: str | None = None) -> int: + """Count stored triples, optionally filtered by *project*.""" + if project is not None: + return len(self._triples.get(project, [])) + return sum(len(ts) for ts in self._triples.values()) + + +# --------------------------------------------------------------------------- +# Protocol compliance assertions (static verification by Pyright) +# --------------------------------------------------------------------------- + +_assert_text: type[TextIndexBackend] = InMemoryTextIndexBackend +_assert_vector: type[VectorIndexBackend] = InMemoryVectorIndexBackend +_assert_graph: type[GraphIndexBackend] = InMemoryGraphIndexBackend + +# --------------------------------------------------------------------------- +# Module exports +# --------------------------------------------------------------------------- + +__all__: list[str] = [ + "DEFAULT_MAX_STUB_ENTRIES", + "InMemoryGraphIndexBackend", + "InMemoryTextIndexBackend", + "InMemoryVectorIndexBackend", +] diff --git a/src/cleveragents/domain/models/acms/provenance.py b/src/cleveragents/domain/models/acms/provenance.py new file mode 100644 index 000000000..ee0d4abb7 --- /dev/null +++ b/src/cleveragents/domain/models/acms/provenance.py @@ -0,0 +1,172 @@ +"""Provenance metadata for UKO triples produced by analyzers. + +When the :class:`UKOIndexer` indexes a resource, it attaches provenance +metadata to every triple so that downstream consumers can trace each +fact back to its source. + ++-------------------------+--------------------------------------------+ +| Model | Purpose | ++=========================+============================================+ +| ``ProvenanceMetadata`` | Source-tracking metadata for a triple | +| ``ProvenancedTriple`` | UKOTriple + ProvenanceMetadata composite | +| ``IndexResult`` | Summary of an indexing operation | ++-------------------------+--------------------------------------------+ + +Based on ``docs/specification.md`` > ACMS > Real-time Index +Synchronization: provenance includes ``sourceResource``, +``sourcePath``, ``sourceRange``, ``validFrom``, ``isCurrent``. +""" + +from __future__ import annotations + +import re +from datetime import UTC, datetime + +from pydantic import BaseModel, ConfigDict, Field, field_validator + +from cleveragents.domain.models.acms.analyzers import UKOTriple + +# --------------------------------------------------------------------------- +# ProvenanceMetadata +# --------------------------------------------------------------------------- + + +class ProvenanceMetadata(BaseModel): + """Source-tracking metadata attached to indexed UKO triples. + + Attributes: + source_resource: ULID of the resource that produced the triple. + source_path: File path within the resource (e.g. + ``"src/main.py"``). Empty string if not applicable. + source_range: Line range within the file (e.g. ``"10-25"``). + Empty string if not applicable (whole-file scope). + valid_from: UTC timestamp when this triple was produced. + is_current: Whether this triple reflects the latest state + of the source. Set to ``False`` when a resource is + removed or superseded. + """ + + source_resource: str = Field( + ..., + min_length=1, + description="ULID of the source resource.", + ) + source_path: str = Field( + default="", + description="File path within the resource.", + ) + source_range: str = Field( + default="", + description="Line range (e.g. '10-25').", + ) + + @field_validator("source_range") + @classmethod + def _validate_source_range(cls, value: str) -> str: + """Validate source_range format: empty or ``'-'``.""" + if not value: + return value + if not re.fullmatch(r"\d+-\d+", value): + raise ValueError( + f"source_range must be empty or match '-' " + f"(e.g. '10-25'), got '{value}'" + ) + return value + + valid_from: datetime = Field( + default_factory=lambda: datetime.now(UTC), + description="UTC timestamp when this triple was produced.", + ) + is_current: bool = Field( + default=True, + description="Whether this triple reflects the latest state.", + ) + + model_config = ConfigDict(frozen=True) + + +# --------------------------------------------------------------------------- +# ProvenancedTriple +# --------------------------------------------------------------------------- + + +class ProvenancedTriple(BaseModel): + """A UKO triple paired with its provenance metadata. + + This composite is the primary output of the :class:`UKOIndexer` + pipeline: every triple produced by an analyzer is wrapped with + provenance before being stored in the graph backend. + + Attributes: + triple: The UKO triple (subject, predicate, object). + provenance: Source-tracking metadata. + """ + + triple: UKOTriple + provenance: ProvenanceMetadata + + model_config = ConfigDict(frozen=True) + + +# --------------------------------------------------------------------------- +# IndexResult +# --------------------------------------------------------------------------- + + +class IndexResult(BaseModel): + """Summary of an indexing operation on a single resource. + + Returned by :meth:`UKOIndexer.index_resource` and + :meth:`UKOIndexer.reindex_resource` for observability. + + Attributes: + resource_id: ULID of the indexed resource. + triple_count: Number of UKO triples produced. + text_docs_indexed: Number of text documents indexed. + embeddings_indexed: Number of vector embeddings indexed. + analyzer_domain: Domain of the analyzer used (e.g. + ``"python"``), or ``"none"`` if no analyzer matched. + errors: Error messages for any backends that failed. + """ + + resource_id: str = Field( + ..., + min_length=1, + description="ULID of the indexed resource.", + ) + triple_count: int = Field( + default=0, + ge=0, + description="Number of UKO triples produced.", + ) + text_docs_indexed: int = Field( + default=0, + ge=0, + description="Number of text documents indexed.", + ) + embeddings_indexed: int = Field( + default=0, + ge=0, + description="Number of vector embeddings indexed.", + ) + analyzer_domain: str = Field( + default="none", + description="Domain of the analyzer used.", + ) + errors: tuple[str, ...] = Field( + default=(), + description="Error messages from failed backends.", + ) + + model_config = ConfigDict(frozen=True) + + +# --------------------------------------------------------------------------- +# Module exports +# --------------------------------------------------------------------------- + +__all__: list[str] = [ + "IndexResult", + "ProvenanceMetadata", + "ProvenancedTriple", +] diff --git a/vulture_whitelist.py b/vulture_whitelist.py index ac0fe47b5..fcf36f4b8 100644 --- a/vulture_whitelist.py +++ b/vulture_whitelist.py @@ -1001,3 +1001,44 @@ depth_resolved_count # noqa: B018, F821 dropped_by_overage_guard # noqa: B018, F821 budget_utilization # noqa: B018, F821 fusion_input_count # noqa: B018, F821 + +# VectorIndexBackend protocol parameter — used by implementations +min_relevance # noqa: B018, F821 + +# Index backend stubs — protocol compliance assertions +_assert_text # noqa: B018, F821 +_assert_vector # noqa: B018, F821 +_assert_graph # noqa: B018, F821 + +# UKO Indexer — public API, DI wiring, and protocol compliance assertions +DEFAULT_MAX_TRIPLES # noqa: B018, F821 +UKOIndexer # noqa: B018, F821 +uko_indexer # noqa: B018, F821 +analyzer_registry # noqa: B018, F821 +_assert_hook # noqa: B018, F821 +_assert_reader # noqa: B018, F821 + +# ResourceFileWatcher — public API (issue #578, file-watching) +FileChangeType # noqa: B018, F821 +ResourceFileWatcher # noqa: B018, F821 +DEFAULT_DEBOUNCE_SECONDS # noqa: B018, F821 +auto_reindex # noqa: B018, F821 +resource_file_watcher # noqa: B018, F821 +_sanitize_log_value # noqa: B018, F821 + +# Index stubs — size limit constant +DEFAULT_MAX_STUB_ENTRIES # noqa: B018, F821 + +# UKO Indexer protocols — public API (issue #578) +DEFAULT_MAX_CONTENT_SIZE # noqa: B018, F821 +DefaultLifecycleHook # noqa: B018, F821 +LocationContentReader # noqa: B018, F821 + +# Analyzer __all__ exports — public API (issue #578) +safe_uri_segment # noqa: B018, F821 + +# Container DI factory — wired via dependency_injector (issue #578) +_build_analyzer_registry # noqa: B018, F821 + +# Test doubles — public API for BDD/Robot test infrastructure +TrackingEventBus # noqa: B018, F821