diff --git a/features/acms_semantic_chunking.feature b/features/acms_semantic_chunking.feature new file mode 100644 index 000000000..9c2d2bfa5 --- /dev/null +++ b/features/acms_semantic_chunking.feature @@ -0,0 +1,44 @@ +Feature: Semantic chunking context strategy for ACMS + As a context assembly system + I want to split files into semantic chunks + So that I can select individual chunks with relevance scoring + + Background: + Given a semantic chunking strategy is initialized + + Scenario: Chunk Python file into functions and classes + Given a Python file with functions and classes + When I chunk the Python file + Then I should get chunks for each function + And I should get chunks for each class + And I should get chunks for each method + + Scenario: Chunk Markdown file into sections + Given a Markdown file with multiple sections + When I chunk the Markdown file + Then I should get chunks for each section + And each chunk should have the section title as name + + Scenario: Handle Python file with syntax errors + Given a Python file with syntax errors + When I chunk the Python file + Then I should get a single chunk for the whole file + + Scenario: Score chunk relevance + Given a list of semantic chunks + When I score the chunks for relevance + Then smaller chunks should have higher relevance scores + And chunks matching a query should have higher scores + + Scenario: Select chunks within budget + Given a list of semantic chunks + And a context budget of 4096 characters + When I select chunks within the budget + Then only chunks that fit should be selected + And chunks should be sorted by relevance + + Scenario: Convert chunks to context fragments + Given a list of semantic chunks + When I convert them to context fragments + Then each fragment should have the chunk content + And each fragment should have metadata about the chunk type diff --git a/features/steps/acms_semantic_chunking_steps.py b/features/steps/acms_semantic_chunking_steps.py new file mode 100644 index 000000000..6f4fff0a0 --- /dev/null +++ b/features/steps/acms_semantic_chunking_steps.py @@ -0,0 +1,237 @@ +"""Step definitions for semantic chunking BDD tests.""" + +from behave import given, then, when + +from cleveragents.domain.models.acms.semantic_chunking import ( + ChunkRelevanceScorer, + SemanticChunk, + SemanticChunkingStrategy, +) + + +@given("a semantic chunking strategy is initialized") +def step_init_strategy(context): + """Initialize a semantic chunking strategy.""" + context.strategy = SemanticChunkingStrategy() + + +@given("a Python file with functions and classes") +def step_python_file_with_functions_and_classes(context): + """Create a Python file with functions and classes.""" + context.python_content = ''' +def hello_world(): + """A simple function.""" + return "Hello, World!" + +class MyClass: + """A simple class.""" + + def __init__(self): + """Initialize the class.""" + self.value = 42 + + def get_value(self): + """Get the value.""" + return self.value +''' + context.python_file = "test.py" + + +@when("I chunk the Python file") +def step_chunk_python_file(context): + """Chunk the Python file.""" + context.chunks = context.strategy.chunk_file( + context.python_file, context.python_content + ) + + +@then("I should get chunks for each function") +def step_check_function_chunks(context): + """Check that function chunks were created.""" + function_chunks = [c for c in context.chunks if c.chunk_type == "function"] + assert len(function_chunks) > 0, "No function chunks found" + assert any(c.name == "hello_world" for c in function_chunks) + + +@then("I should get chunks for each class") +def step_check_class_chunks(context): + """Check that class chunks were created.""" + class_chunks = [c for c in context.chunks if c.chunk_type == "class"] + assert len(class_chunks) > 0, "No class chunks found" + assert any(c.name == "MyClass" for c in class_chunks) + + +@then("I should get chunks for each method") +def step_check_method_chunks(context): + """Check that method chunks were created.""" + method_chunks = [c for c in context.chunks if c.chunk_type == "method"] + assert len(method_chunks) > 0, "No method chunks found" + assert any("__init__" in c.name for c in method_chunks) + assert any("get_value" in c.name for c in method_chunks) + + +@given("a Markdown file with multiple sections") +def step_markdown_file_with_sections(context): + """Create a Markdown file with multiple sections.""" + context.markdown_content = '''# Introduction +This is the introduction section. + +## Getting Started +This section explains how to get started. + +### Installation +Install the package using pip. + +## Usage +This section explains how to use the package. + +### Examples +Here are some examples. +''' + context.markdown_file = "test.md" + + +@when("I chunk the Markdown file") +def step_chunk_markdown_file(context): + """Chunk the Markdown file.""" + context.chunks = context.strategy.chunk_file( + context.markdown_file, context.markdown_content + ) + + +@then("I should get chunks for each section") +def step_check_section_chunks(context): + """Check that section chunks were created.""" + section_chunks = [c for c in context.chunks if c.chunk_type == "section"] + assert len(section_chunks) > 0, "No section chunks found" + + +@then("each chunk should have the section title as name") +def step_check_section_names(context): + """Check that section chunks have proper names.""" + section_chunks = [c for c in context.chunks if c.chunk_type == "section"] + names = [c.name for c in section_chunks] + assert "Introduction" in names or any("Introduction" in n for n in names) + + +@given("a Python file with syntax errors") +def step_python_file_with_syntax_errors(context): + """Create a Python file with syntax errors.""" + context.python_content = ''' +def broken_function( + # Missing closing parenthesis + return "This won't parse" +''' + context.python_file = "broken.py" + + +@then("I should get a single chunk for the whole file") +def step_check_single_chunk(context): + """Check that a single chunk was created for the whole file.""" + assert len(context.chunks) == 1 + assert context.chunks[0].chunk_type == "file" + + +@given("a list of semantic chunks") +def step_create_semantic_chunks(context): + """Create a list of semantic chunks.""" + context.chunks = [ + SemanticChunk( + file_path="test.py", + chunk_type="function", + name="small_func", + start_line=1, + end_line=3, + content="def small_func():\n return 42\n", + ), + SemanticChunk( + file_path="test.py", + chunk_type="function", + name="large_func", + start_line=5, + end_line=50, + content="def large_func():\n" + " x = 1\n" * 45, + ), + ] + + +@when("I score the chunks for relevance") +def step_score_chunks(context): + """Score the chunks for relevance.""" + context.scorer = ChunkRelevanceScorer() + context.scored_chunks = [ + (chunk, context.scorer.score(chunk, query=None, context_budget=4096)) + for chunk in context.chunks + ] + + +@then("smaller chunks should have higher relevance scores") +def step_check_smaller_chunks_higher_score(context): + """Check that smaller chunks have higher scores.""" + small_chunk_score = next( + score for chunk, score in context.scored_chunks if chunk.name == "small_func" + ) + large_chunk_score = next( + score for chunk, score in context.scored_chunks if chunk.name == "large_func" + ) + assert small_chunk_score > large_chunk_score + + +@then("chunks matching a query should have higher scores") +def step_check_query_matching_score(context): + """Check that chunks matching a query have higher scores.""" + scorer = ChunkRelevanceScorer() + score_without_query = scorer.score(context.chunks[0], query=None) + score_with_query = scorer.score(context.chunks[0], query="small_func") + assert score_with_query > score_without_query + + +@given("a context budget of {budget:d} characters") +def step_set_context_budget(context, budget): + """Set the context budget.""" + context.context_budget = budget + + +@when("I select chunks within the budget") +def step_select_chunks_within_budget(context): + """Select chunks within the budget.""" + context.selected_chunks = context.strategy.select_chunks( + context.chunks, context_budget=context.context_budget + ) + + +@then("only chunks that fit should be selected") +def step_check_chunks_fit_budget(context): + """Check that selected chunks fit within the budget.""" + total_size = sum(len(chunk.content) for chunk in context.selected_chunks) + assert total_size <= context.context_budget + + +@then("chunks should be sorted by relevance") +def step_check_chunks_sorted_by_relevance(context): + """Check that chunks are sorted by relevance.""" + scores = [chunk.relevance_score for chunk in context.selected_chunks] + assert scores == sorted(scores, reverse=True) + + +@when("I convert them to context fragments") +def step_convert_to_fragments(context): + """Convert chunks to context fragments.""" + context.fragments = context.strategy.to_context_fragments(context.chunks) + + +@then("each fragment should have the chunk content") +def step_check_fragment_content(context): + """Check that fragments have the chunk content.""" + for fragment, chunk in zip(context.fragments, context.chunks, strict=True): + assert fragment.content == chunk.content + + +@then("each fragment should have metadata about the chunk type") +def step_check_fragment_metadata(context): + """Check that fragments have metadata about the chunk type.""" + for fragment, chunk in zip(context.fragments, context.chunks, strict=True): + assert "chunk_type" in fragment.metadata + assert fragment.metadata["chunk_type"] == chunk.chunk_type + assert "chunk_name" in fragment.metadata + assert fragment.metadata["chunk_name"] == chunk.name diff --git a/src/cleveragents/domain/models/acms/semantic_chunking.py b/src/cleveragents/domain/models/acms/semantic_chunking.py new file mode 100644 index 000000000..508fb7d59 --- /dev/null +++ b/src/cleveragents/domain/models/acms/semantic_chunking.py @@ -0,0 +1,305 @@ +"""Semantic chunking strategy for ACMS context assembly.""" + +import ast +import re +from dataclasses import dataclass +from pathlib import Path + +from cleveragents.domain.models.core.context_fragment import ( + ContextFragment, + FragmentProvenance, +) + + +@dataclass +class SemanticChunk: + """Represents a semantically meaningful chunk of code or text.""" + + file_path: str + chunk_type: str + name: str + start_line: int + end_line: int + content: str + relevance_score: float = 0.5 + + +class PythonSemanticChunker: + """Chunks Python files into semantic units using AST analysis.""" + + def __init__(self) -> None: + """Initialize the Python semantic chunker.""" + self.chunks: list[SemanticChunk] = [] + + def chunk(self, file_path: str, content: str) -> list[SemanticChunk]: + """Chunk a Python file into semantic units.""" + self.chunks = [] + + try: + tree = ast.parse(content) + except SyntaxError: + return [ + SemanticChunk( + file_path=file_path, + chunk_type="file", + name=Path(file_path).stem, + start_line=1, + end_line=len(content.splitlines()), + content=content, + relevance_score=0.5, + ) + ] + + lines = content.splitlines(keepends=True) + + for node in ast.walk(tree): + if isinstance(node, ast.FunctionDef): + self._extract_function_chunk(node, lines, file_path) + elif isinstance(node, ast.ClassDef): + self._extract_class_chunk(node, lines, file_path) + + if not self.chunks: + return [ + SemanticChunk( + file_path=file_path, + chunk_type="file", + name=Path(file_path).stem, + start_line=1, + end_line=len(lines), + content=content, + relevance_score=0.5, + ) + ] + + return self.chunks + + def _extract_function_chunk( + self, node: ast.FunctionDef, lines: list[str], file_path: str + ) -> None: + """Extract a function as a semantic chunk.""" + start_line = node.lineno + end_line = node.end_lineno or node.lineno + + chunk_content = "".join(lines[start_line - 1 : end_line]) + + self.chunks.append( + SemanticChunk( + file_path=file_path, + chunk_type="function", + name=node.name, + start_line=start_line, + end_line=end_line, + content=chunk_content, + relevance_score=0.5, + ) + ) + + def _extract_class_chunk( + self, node: ast.ClassDef, lines: list[str], file_path: str + ) -> None: + """Extract a class and its methods as semantic chunks.""" + start_line = node.lineno + end_line = node.end_lineno or node.lineno + + chunk_content = "".join(lines[start_line - 1 : end_line]) + + self.chunks.append( + SemanticChunk( + file_path=file_path, + chunk_type="class", + name=node.name, + start_line=start_line, + end_line=end_line, + content=chunk_content, + relevance_score=0.5, + ) + ) + + for item in node.body: + if isinstance(item, ast.FunctionDef): + method_start = item.lineno + method_end = item.end_lineno or item.lineno + method_content = "".join(lines[method_start - 1 : method_end]) + + self.chunks.append( + SemanticChunk( + file_path=file_path, + chunk_type="method", + name=f"{node.name}.{item.name}", + start_line=method_start, + end_line=method_end, + content=method_content, + relevance_score=0.5, + ) + ) + + +class MarkdownSemanticChunker: + """Chunks Markdown files into semantic units based on sections.""" + + def __init__(self) -> None: + """Initialize the Markdown semantic chunker.""" + self.chunks: list[SemanticChunk] = [] + + def chunk(self, file_path: str, content: str) -> list[SemanticChunk]: + """Chunk a Markdown file into semantic units.""" + self.chunks = [] + lines = content.splitlines(keepends=True) + + header_pattern = re.compile(r"^(#{1,6})\s+(.+)$") + headers: list[tuple[int, int, str]] = [] + + for i, line in enumerate(lines): + match = header_pattern.match(line) + if match: + _level = len(match.group(1)) + title = match.group(2).strip() + headers.append((i, _level, title)) + + if not headers: + return [ + SemanticChunk( + file_path=file_path, + chunk_type="file", + name=Path(file_path).stem, + start_line=1, + end_line=len(lines), + content=content, + relevance_score=0.5, + ) + ] + + for i, (header_line, _level, title) in enumerate(headers): + end_line = headers[i + 1][0] if i + 1 < len(headers) else len(lines) + + chunk_content = "".join(lines[header_line:end_line]) + + self.chunks.append( + SemanticChunk( + file_path=file_path, + chunk_type="section", + name=title, + start_line=header_line + 1, + end_line=end_line, + content=chunk_content, + relevance_score=0.5, + ) + ) + + return self.chunks + + +class ChunkRelevanceScorer: + """Scores the relevance of chunks for context selection.""" + + def __init__(self) -> None: + """Initialize the chunk relevance scorer.""" + + def score( + self, + chunk: SemanticChunk, + query: str | None = None, + context_budget: int = 4096, + ) -> float: + """Score the relevance of a chunk.""" + score = 0.5 + + chunk_size = len(chunk.content) + if chunk_size < context_budget * 0.1: + score += 0.2 + elif chunk_size < context_budget * 0.5: + score += 0.1 + + if chunk.chunk_type in ("class", "function", "method"): + score += 0.1 + + if query and query.lower() in chunk.name.lower(): + score += 0.3 + + return min(1.0, max(0.0, score)) + + +class SemanticChunkingStrategy: + """Context strategy that uses semantic chunking for context assembly.""" + + def __init__(self) -> None: + """Initialize the semantic chunking strategy.""" + self.python_chunker = PythonSemanticChunker() + self.markdown_chunker = MarkdownSemanticChunker() + self.scorer = ChunkRelevanceScorer() + + def chunk_file(self, file_path: str, content: str) -> list[SemanticChunk]: + """Chunk a file based on its type.""" + path = Path(file_path) + + if path.suffix == ".py": + return self.python_chunker.chunk(file_path, content) + elif path.suffix in (".md", ".markdown"): + return self.markdown_chunker.chunk(file_path, content) + else: + return [ + SemanticChunk( + file_path=file_path, + chunk_type="file", + name=path.stem, + start_line=1, + end_line=len(content.splitlines()), + content=content, + relevance_score=0.5, + ) + ] + + def select_chunks( + self, + chunks: list[SemanticChunk], + query: str | None = None, + context_budget: int = 4096, + ) -> list[SemanticChunk]: + """Select chunks based on relevance and budget constraints.""" + scored_chunks = [ + (chunk, self.scorer.score(chunk, query, context_budget)) + for chunk in chunks + ] + + scored_chunks.sort(key=lambda x: x[1], reverse=True) + + selected: list[SemanticChunk] = [] + total_size = 0 + + for chunk, score in scored_chunks: + chunk_size = len(chunk.content) + if total_size + chunk_size <= context_budget: + chunk.relevance_score = score + selected.append(chunk) + total_size += chunk_size + + return selected + + def to_context_fragments( + self, chunks: list[SemanticChunk] + ) -> list[ContextFragment]: + """Convert semantic chunks to context fragments.""" + fragments: list[ContextFragment] = [] + + for chunk in chunks: + provenance = FragmentProvenance( + resource_uri=chunk.file_path, + location=f"{chunk.start_line}-{chunk.end_line}", + strategy="semantic_chunking", + ) + fragment = ContextFragment( + uko_node=f"file://{chunk.file_path}#{chunk.name}", + content=chunk.content, + detail_depth=0, + token_count=len(chunk.content.split()), + relevance_score=chunk.relevance_score, + provenance=provenance, + metadata={ + "chunk_type": chunk.chunk_type, + "chunk_name": chunk.name, + "relevance_score": str(chunk.relevance_score), + }, + ) + fragments.append(fragment) + + return fragments