f808abff86
Fix JSON syntax errors in .devcontainer/devcontainer.json (removed
invalid JS-style // comments) and .devcontainer/opencode.json (removed
90+ trailing commas). Apply auto-fixes for end-of-file and trailing
whitespace issues across 100+ files. Fix SIM105 ruff violations in
benchmarks/core_circuit_breaker_bench.py (use contextlib.suppress).
Note: The security fix from issue #7478 (validate_path startswith bypass)
was already delivered to master in commit e18ac5f2. This PR as currently
structured is non-atomic (35 commits across 10+ issues) and needs
significant restructure before merge. This commit only addresses the
CI/pre-commit failures.
ISSUES CLOSED: #7478
1917 lines
68 KiB
Python
1917 lines
68 KiB
Python
"""Behave step definitions for container tool execution (issue #515)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
import json
|
|
import os
|
|
import shutil
|
|
import tempfile
|
|
from pathlib import Path
|
|
from typing import Any
|
|
from unittest.mock import MagicMock, patch
|
|
|
|
from behave import given, then, when # type: ignore[import-untyped]
|
|
|
|
from cleveragents.application.services.execution_environment_resolver import (
|
|
ContainerUnavailableError,
|
|
ExecutionEnvironmentResolver,
|
|
)
|
|
from cleveragents.domain.models.core.change import ToolInvocation
|
|
from cleveragents.domain.models.core.plan import ExecutionEnvironment
|
|
from cleveragents.tool.container_executor import (
|
|
_MAX_OUTPUT_BYTES,
|
|
_MAX_RECURSION_DEPTH,
|
|
ContainerConfig,
|
|
ContainerExecutionError,
|
|
ContainerMetadata,
|
|
ContainerTimeoutError,
|
|
ContainerToolExecutor,
|
|
_ExecResult,
|
|
_looks_like_path,
|
|
)
|
|
from cleveragents.tool.path_mapper import PathMapper
|
|
from cleveragents.tool.registry import ToolRegistry as _TR
|
|
from cleveragents.tool.runner import ToolRunner
|
|
from cleveragents.tool.runtime import ToolError, ToolResult, ToolSpec
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_FAKE_DEVCONTAINER_BIN = "/usr/local/bin/devcontainer"
|
|
|
|
|
|
def _set_fake_devcontainer_bin(executor: ContainerToolExecutor) -> None:
|
|
"""Inject a fake devcontainer binary path for tests.
|
|
|
|
In CI/test environments where the real ``devcontainer`` CLI is not
|
|
installed, the executor stores ``None`` and would fail fast. Test
|
|
scenarios that mock ``_run_command`` or ``subprocess.Popen`` never
|
|
invoke the real binary, so a dummy path is sufficient.
|
|
"""
|
|
executor._devcontainer_bin = _FAKE_DEVCONTAINER_BIN # type: ignore[assignment]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Background
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given("I have the container tool execution module imported")
|
|
def step_have_container_exec_module(context: Any) -> None:
|
|
# Import check — will fail at parse time if modules are broken
|
|
from cleveragents.tool import container_executor, path_mapper # noqa: F401
|
|
|
|
context.imported = True
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# PathMapper
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given('a temporary sandbox directory "{path}"')
|
|
def step_given_named_sandbox(context: Any, path: str) -> None:
|
|
"""Create the named directory as a temporary sandbox root for path tests."""
|
|
os.makedirs(path, exist_ok=True)
|
|
if not hasattr(context, "_cleanup_handlers"):
|
|
context._cleanup_handlers = []
|
|
context._cleanup_handlers.append(
|
|
lambda p=path: shutil.rmtree(p, ignore_errors=True)
|
|
)
|
|
context.sandbox_dir = path
|
|
|
|
|
|
@given(
|
|
'I have a PathMapper with host_root "{host_root}" and container_root "{container_root}"'
|
|
)
|
|
def step_create_path_mapper(context: Any, host_root: str, container_root: str) -> None:
|
|
context.path_mapper = PathMapper(host_root=host_root, container_root=container_root)
|
|
|
|
|
|
@when('I map host path "{path}" to container')
|
|
def step_map_host_to_container(context: Any, path: str) -> None:
|
|
context.mapped_path = context.path_mapper.host_to_container(path)
|
|
|
|
|
|
@when('I map the host path "{path}" to container')
|
|
def step_map_the_host_to_container(context: Any, path: str) -> None:
|
|
"""Alias for step_map_host_to_container (used in path containment security tests)."""
|
|
context.mapped_path = context.path_mapper.host_to_container(path)
|
|
|
|
|
|
@when('I map container path "{path}" to host')
|
|
def step_map_container_to_host(context: Any, path: str) -> None:
|
|
context.mapped_path = context.path_mapper.container_to_host(path)
|
|
|
|
|
|
@then('the container path should be "{expected}"')
|
|
def step_check_container_path(context: Any, expected: str) -> None:
|
|
assert context.mapped_path == expected, (
|
|
f"Expected {expected!r}, got {context.mapped_path!r}"
|
|
)
|
|
|
|
|
|
@then('the mapped path should be "{expected}"')
|
|
def step_check_mapped_path(context: Any, expected: str) -> None:
|
|
"""Assert that the most recently mapped path equals expected."""
|
|
assert context.mapped_path == expected, (
|
|
f"Expected {expected!r}, got {context.mapped_path!r}"
|
|
)
|
|
|
|
|
|
@then('the host path should be "{expected}"')
|
|
def step_check_host_path(context: Any, expected: str) -> None:
|
|
assert context.mapped_path == expected, (
|
|
f"Expected {expected!r}, got {context.mapped_path!r}"
|
|
)
|
|
|
|
|
|
@then('"{path}" should be a host path')
|
|
def step_is_host_path(context: Any, path: str) -> None:
|
|
assert context.path_mapper.is_host_path(path), f"{path} should be a host path"
|
|
|
|
|
|
@then('"{path}" should not be a host path')
|
|
def step_is_not_host_path(context: Any, path: str) -> None:
|
|
assert not context.path_mapper.is_host_path(path), (
|
|
f"{path} should NOT be a host path"
|
|
)
|
|
|
|
|
|
@then('"{path}" should be a container path')
|
|
def step_is_container_path(context: Any, path: str) -> None:
|
|
assert context.path_mapper.is_container_path(path), (
|
|
f"{path} should be a container path"
|
|
)
|
|
|
|
|
|
@then('"{path}" should not be a container path')
|
|
def step_is_not_container_path(context: Any, path: str) -> None:
|
|
assert not context.path_mapper.is_container_path(path), (
|
|
f"{path} should NOT be a container path"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# PathMapper prefix-collision security tests (#7478)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given("a sibling directory with a name that is a prefix of the sandbox root")
|
|
def step_sibling_prefix_dir(context: Any) -> None:
|
|
"""Create a sibling directory whose name starts with the sandbox root name.
|
|
|
|
For example, if the sandbox host_root is ``/tmp/sandbox``, this creates
|
|
``/tmp/sandbox-escape``. This exercises the startswith prefix-collision
|
|
path traversal bypass (issue #7478).
|
|
"""
|
|
path_mapper = getattr(context, "path_mapper", None)
|
|
assert path_mapper is not None, "PathMapper not initialized in context"
|
|
host_root: str = path_mapper.host_root
|
|
root_parent = str(Path(host_root).parent)
|
|
sibling_name = Path(host_root).name + "-escape"
|
|
sibling_path = Path(root_parent) / sibling_name
|
|
sibling_path.mkdir(parents=True, exist_ok=True)
|
|
context._cleanup_handlers.append(
|
|
lambda: shutil.rmtree(str(sibling_path), ignore_errors=True)
|
|
)
|
|
|
|
|
|
@when("I check whether host path is safe from prefix collision")
|
|
def step_check_prefix_collision(context: Any) -> None:
|
|
"""Check whether a sibling-prefix path incorrectly passes containment."""
|
|
path_mapper = getattr(context, "path_mapper", None)
|
|
assert path_mapper is not None, "PathMapper not initialized in context"
|
|
host_root: str = path_mapper.host_root
|
|
root_parent = str(Path(host_root).parent)
|
|
root_name = Path(host_root).name
|
|
escape_root = f"{root_name}-escape"
|
|
escape_path = os.path.join(root_parent, escape_root, "evil.txt")
|
|
context.prefix_collision_result = path_mapper.is_host_path(escape_path)
|
|
# Store the escape path for use in the error message
|
|
context.escape_path = escape_path
|
|
|
|
|
|
@then("the prefix collision check should return ``False``")
|
|
def step_prefix_collision_rejected(context: Any) -> None:
|
|
result: bool = getattr(context, "prefix_collision_result", None)
|
|
assert result is False, (
|
|
f"Expected is_host_path({getattr(context, 'escape_path', 'unknown')!r}) to be False "
|
|
f"but got True — prefix collision bypass not prevented!"
|
|
)
|
|
|
|
|
|
@then('"{path}" should result be false for host path containment')
|
|
def step_host_path_result_false(context: Any, path: str) -> None:
|
|
"""Generalized step to assert is_host_path returns False for a given path."""
|
|
result = context.path_mapper.is_host_path(path)
|
|
assert result is False, (
|
|
f"Expected is_host_path({path!r}) to be False but got {result}"
|
|
)
|
|
|
|
|
|
@then('"{path}" should result be true for host path containment')
|
|
def step_host_path_result_true(context: Any, path: str) -> None:
|
|
"""Generalized step to assert is_host_path returns True for a given path."""
|
|
result = context.path_mapper.is_host_path(path)
|
|
assert result is True, (
|
|
f"Expected is_host_path({path!r}) to be True but got {result}"
|
|
)
|
|
|
|
|
|
@given('a file "{name}" with content "{content}" at "{absolute_path}"')
|
|
def step_create_file_at_absolute_path(
|
|
context: Any, name: str, content: str, absolute_path: str
|
|
) -> None:
|
|
"""Create a file at an arbitrary absolute path (used for sibling-prefix tests)."""
|
|
abs_path = Path(absolute_path)
|
|
abs_path.parent.mkdir(parents=True, exist_ok=True)
|
|
abs_path.write_text(content)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Generic result assertions (for prefix collision test scenarios)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when('I check whether "{path}" is a host path')
|
|
def step_check_host_path(context: Any, path: str) -> None:
|
|
"""Store the result of is_host_path for later assertion."""
|
|
path_mapper = getattr(context, "path_mapper", None)
|
|
assert path_mapper is not None, "PathMapper not initialized in context"
|
|
context._host_path_result = path_mapper.is_host_path(path)
|
|
|
|
|
|
@then("the host containment result should be true")
|
|
def step_result_is_true(context: Any) -> None:
|
|
"""Assert that the most recent host path containment check returned True."""
|
|
result: bool = getattr(context, "_host_path_result", False)
|
|
assert result is True, f"Expected host path check to return True but got {result}"
|
|
|
|
|
|
@then("the host containment result should be false")
|
|
def step_result_is_false(context: Any) -> None:
|
|
"""Assert that the most recent host path containment check returned False."""
|
|
result: bool = getattr(context, "_host_path_result", True)
|
|
assert result is False, f"Expected host path check to return False but got {result}"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# ContainerConfig
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when("I create a default ContainerConfig")
|
|
def step_create_default_config(context: Any) -> None:
|
|
context.container_config = ContainerConfig()
|
|
|
|
|
|
@then('the workspace_folder should be "{expected}"')
|
|
def step_check_workspace_folder(context: Any, expected: str) -> None:
|
|
assert context.container_config.workspace_folder == expected
|
|
|
|
|
|
@then("the container timeout_seconds should be {expected:d}")
|
|
def step_check_timeout_seconds(context: Any, expected: int) -> None:
|
|
assert context.container_config.timeout_seconds == expected
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# ContainerMetadata
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when('I create a ContainerMetadata with container_id "{cid}" and image "{img}"')
|
|
def step_create_metadata(context: Any, cid: str, img: str) -> None:
|
|
context.container_metadata = ContainerMetadata(container_id=cid, image=img)
|
|
|
|
|
|
@then('the container_metadata container_id should be "{expected}"')
|
|
def step_check_metadata_cid(context: Any, expected: str) -> None:
|
|
assert context.container_metadata.container_id == expected
|
|
|
|
|
|
@then('the container_metadata image should be "{expected}"')
|
|
def step_check_metadata_image(context: Any, expected: str) -> None:
|
|
assert context.container_metadata.image == expected
|
|
|
|
|
|
@then("the container_metadata timed_out should be false")
|
|
def step_check_metadata_timed_out_false(context: Any) -> None:
|
|
assert context.container_metadata.timed_out is False
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# ContainerToolExecutor instantiation
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given('I have a ContainerConfig with workspace "{ws}" and container_id "{cid}"')
|
|
def step_create_config_with_params(context: Any, ws: str, cid: str) -> None:
|
|
context.container_config = ContainerConfig(workspace_folder=ws, container_id=cid)
|
|
|
|
|
|
@when("I create a ContainerToolExecutor with that config")
|
|
def step_create_executor(context: Any) -> None:
|
|
context.executor = ContainerToolExecutor(context.container_config)
|
|
|
|
|
|
@then('the executor config container_id should be "{expected}"')
|
|
def step_check_executor_cid(context: Any, expected: str) -> None:
|
|
assert context.executor.config.container_id == expected
|
|
|
|
|
|
@then("the executor should have a path mapper")
|
|
def step_executor_has_path_mapper(context: Any) -> None:
|
|
assert context.executor.path_mapper is not None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Input / output path mapping
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given('I have a ContainerToolExecutor with host_root "{hr}" and container_root "{cr}"')
|
|
def step_create_executor_with_roots(context: Any, hr: str, cr: str) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder=cr,
|
|
host_sandbox_path=hr,
|
|
container_id="test",
|
|
)
|
|
context.executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(context.executor)
|
|
|
|
|
|
@when('I map tool inputs with path "{path}"')
|
|
def step_map_tool_inputs(context: Any, path: str) -> None:
|
|
inputs = {"path": path}
|
|
context.mapped_inputs = context.executor._map_input_paths(inputs)
|
|
|
|
|
|
@then('the mapped input path should be "{expected}"')
|
|
def step_check_mapped_input(context: Any, expected: str) -> None:
|
|
assert context.mapped_inputs["path"] == expected, (
|
|
f"Expected {expected!r}, got {context.mapped_inputs['path']!r}"
|
|
)
|
|
|
|
|
|
@when('I map tool output with path "{path}"')
|
|
def step_map_tool_output(context: Any, path: str) -> None:
|
|
output = {"path": path}
|
|
context.mapped_output = context.executor._map_output_paths(output)
|
|
|
|
|
|
@then('the mapped output path should be "{expected}"')
|
|
def step_check_mapped_output(context: Any, expected: str) -> None:
|
|
assert context.mapped_output["path"] == expected, (
|
|
f"Expected {expected!r}, got {context.mapped_output['path']!r}"
|
|
)
|
|
|
|
|
|
@when('I map tool inputs with nested paths under "{root}"')
|
|
def step_map_nested_inputs(context: Any, root: str) -> None:
|
|
inputs = {
|
|
"file": f"{root}/a.py",
|
|
"config": {"paths": [f"{root}/b.py", f"{root}/c.py"]},
|
|
"other": 42,
|
|
}
|
|
context.mapped_inputs = context.executor._map_input_paths(inputs)
|
|
|
|
|
|
@then("all nested paths should be mapped to container paths")
|
|
def step_check_nested_mapped(context: Any) -> None:
|
|
assert context.mapped_inputs["file"].startswith("/workspace"), (
|
|
f"Expected /workspace, got {context.mapped_inputs['file']}"
|
|
)
|
|
# Nested dicts are also mapped
|
|
inner = context.mapped_inputs["config"]
|
|
for p in inner["paths"]:
|
|
assert p.startswith("/workspace"), f"Expected /workspace, got {p}"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Command building (S12 — flag/value adjacency)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when('I build an exec command for tool "{tool_name}" with timeout {timeout:d}')
|
|
def step_build_exec_command(context: Any, tool_name: str, timeout: int) -> None:
|
|
cmd, stdin_data = context.executor._build_exec_command(
|
|
tool_name, {"key": "value"}, timeout
|
|
)
|
|
context.built_command = cmd
|
|
context.built_stdin = stdin_data
|
|
|
|
|
|
@then('the command should have "--container-id" adjacent to its value')
|
|
def step_check_flag_adjacency(context: Any) -> None:
|
|
cmd = context.built_command
|
|
cid = context.executor.config.container_id
|
|
# Verify --container-id and its value are adjacent in the command list
|
|
assert "--container-id" in cmd, f"Expected '--container-id' in command: {cmd}"
|
|
idx = cmd.index("--container-id")
|
|
assert idx + 1 < len(cmd), "'--container-id' is the last element, no value follows"
|
|
assert cmd[idx + 1] == cid, (
|
|
f"Expected '{cid}' immediately after '--container-id', got '{cmd[idx + 1]}'"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Timeout handling (mock)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that times out")
|
|
def step_executor_mock_timeout(context: Any) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="timeout-ctr",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
timeout_seconds=5,
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
# Patch _run_command to simulate timeout
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout="",
|
|
stderr="timed out after 5s",
|
|
exit_code=-1,
|
|
duration_ms=5000.0,
|
|
timed_out=True,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@when('I execute tool "{tool_name}" with inputs in the container')
|
|
def step_execute_tool_in_container(context: Any, tool_name: str) -> None:
|
|
context.tool_result = context.executor.execute_tool(tool_name, {"key": "value"})
|
|
|
|
|
|
@then("the container tool result should indicate failure")
|
|
def step_check_container_tool_result_failure(context: Any) -> None:
|
|
assert context.tool_result.success is False
|
|
|
|
|
|
@then('the container tool result error should mention "{text}"')
|
|
def step_check_container_tool_result_error_text(context: Any, text: str) -> None:
|
|
assert text in context.tool_result.error, (
|
|
f"Expected {text!r} in {context.tool_result.error!r}"
|
|
)
|
|
|
|
|
|
@then("the tool result metadata should contain container info with timed_out true")
|
|
def step_check_metadata_timed_out_true(context: Any) -> None:
|
|
meta = context.tool_result.metadata.get("container", {})
|
|
assert meta.get("timed_out") is True, (
|
|
f"Expected timed_out=True in metadata.container, got {meta}"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Error reporting (mock)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that returns exit code {code:d}")
|
|
def step_executor_mock_exit_code(context: Any, code: int) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="fail-ctr",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout="",
|
|
stderr="something went wrong",
|
|
exit_code=code,
|
|
duration_ms=100.0,
|
|
timed_out=False,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with subprocess raising OSError")
|
|
def step_executor_mock_oserror(context: Any) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="os-err-ctr",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
# Patch subprocess.Popen to raise a real OSError so the except OSError
|
|
# branch in _run_command is actually exercised (S10).
|
|
patcher = patch(
|
|
"cleveragents.tool.container_executor.subprocess.Popen",
|
|
side_effect=OSError("No such file or directory: 'devcontainer'"),
|
|
)
|
|
patcher.start()
|
|
context.executor = executor
|
|
context.oserror_patcher = patcher
|
|
# Register cleanup so the patcher is stopped even if the scenario
|
|
# aborts before the "When" step runs (P2-18).
|
|
context.add_cleanup(patcher.stop)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Successful execution (mock)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that returns JSON output")
|
|
def step_executor_mock_json(context: Any) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="json-ctr",
|
|
image="node:20",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout=json.dumps({"lint_errors": 0, "files_checked": 5}),
|
|
stderr="",
|
|
exit_code=0,
|
|
duration_ms=250.0,
|
|
timed_out=False,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@then("the container tool result should indicate success")
|
|
def step_check_container_tool_result_success(context: Any) -> None:
|
|
assert context.tool_result.success is True
|
|
|
|
|
|
@then("the tool result output should contain the parsed JSON")
|
|
def step_check_tool_result_json(context: Any) -> None:
|
|
assert "lint_errors" in context.tool_result.output
|
|
|
|
|
|
@then("the tool result metadata should contain container info")
|
|
def step_check_tool_result_has_metadata(context: Any) -> None:
|
|
assert "container" in context.tool_result.metadata, (
|
|
f"Expected 'container' in metadata, got keys: "
|
|
f"{list(context.tool_result.metadata.keys())}"
|
|
)
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that returns plain text")
|
|
def step_executor_mock_plaintext(context: Any) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="text-ctr",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout="Hello from container!",
|
|
stderr="",
|
|
exit_code=0,
|
|
duration_ms=50.0,
|
|
timed_out=False,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@then("the tool result output should contain raw_output")
|
|
def step_check_raw_output(context: Any) -> None:
|
|
assert "raw_output" in context.tool_result.output
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# ToolRunner container routing
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given("I have a ToolRunner with a ContainerToolExecutor mock")
|
|
def step_runner_with_container_executor(context: Any) -> None:
|
|
registry = MagicMock(spec=_TR)
|
|
spec = ToolSpec(
|
|
name="test_tool",
|
|
description="A test tool",
|
|
input_schema={},
|
|
handler=lambda _: {"ok": True},
|
|
)
|
|
registry.get.return_value = spec
|
|
registry.list_tools.return_value = [spec]
|
|
|
|
# Mock resolver to return CONTAINER
|
|
resolver = MagicMock(spec=ExecutionEnvironmentResolver)
|
|
resolver.resolve_and_validate.return_value = ExecutionEnvironment.CONTAINER
|
|
|
|
# Mock container executor
|
|
mock_executor = MagicMock(spec=ContainerToolExecutor)
|
|
mock_executor.execute_tool.return_value = ToolResult(
|
|
success=True,
|
|
output={"result": "container_ok"},
|
|
duration_ms=100.0,
|
|
)
|
|
|
|
context.runner = ToolRunner(
|
|
registry=registry,
|
|
execution_environment_resolver=resolver,
|
|
container_executor=mock_executor,
|
|
)
|
|
context.mock_container_executor = mock_executor
|
|
|
|
|
|
@when("I execute a tool with container execution environment")
|
|
def step_execute_tool_container_env(context: Any) -> None:
|
|
context.tool_result = context.runner.execute(
|
|
"test_tool",
|
|
{"arg": "val"},
|
|
tool_env="container",
|
|
linked_resource_types=["devcontainer-instance"],
|
|
)
|
|
|
|
|
|
@then("the ContainerToolExecutor should have been called")
|
|
def step_check_container_executor_called(context: Any) -> None:
|
|
context.mock_container_executor.execute_tool.assert_called_once()
|
|
|
|
|
|
@then("the resolver should have been called with the correct arguments")
|
|
def step_check_resolver_args(context: Any) -> None:
|
|
resolver = context.runner._env_resolver
|
|
resolver.resolve_and_validate.assert_called_once()
|
|
call_kwargs = resolver.resolve_and_validate.call_args
|
|
# Verify the resolver received the expected arguments (S13)
|
|
assert call_kwargs.kwargs.get("tool_env") == "container" or (
|
|
call_kwargs[1].get("tool_env") == "container"
|
|
), f"Expected tool_env='container' in resolver call, got {call_kwargs}"
|
|
resource_types = call_kwargs.kwargs.get("linked_resource_types") or call_kwargs[
|
|
1
|
|
].get("linked_resource_types")
|
|
assert "devcontainer-instance" in resource_types, (
|
|
f"Expected 'devcontainer-instance' in linked_resource_types, "
|
|
f"got {resource_types}"
|
|
)
|
|
|
|
|
|
@given("I have a ToolRunner without a ContainerToolExecutor")
|
|
def step_runner_without_container_executor(context: Any) -> None:
|
|
registry = MagicMock(spec=_TR)
|
|
spec = ToolSpec(
|
|
name="test_tool",
|
|
description="A test tool",
|
|
input_schema={},
|
|
handler=lambda _: {"ok": True},
|
|
)
|
|
registry.get.return_value = spec
|
|
|
|
resolver = MagicMock(spec=ExecutionEnvironmentResolver)
|
|
resolver.resolve_and_validate.return_value = ExecutionEnvironment.CONTAINER
|
|
|
|
context.runner = ToolRunner(
|
|
registry=registry,
|
|
execution_environment_resolver=resolver,
|
|
# No container_executor
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# ToolInvocation audit trail
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when("I create a ToolInvocation with container_metadata")
|
|
def step_create_invocation_with_metadata(context: Any) -> None:
|
|
context.invocation = ToolInvocation(
|
|
plan_id="01TESTPLANID000000000000000",
|
|
tool_name="ns/tool",
|
|
container_metadata={
|
|
"container_id": "abc",
|
|
"image": "node:20",
|
|
"exec_time_ms": 150.0,
|
|
},
|
|
)
|
|
|
|
|
|
@then('the ToolInvocation container_metadata should contain "{key}"')
|
|
def step_check_invocation_metadata_key(context: Any, key: str) -> None:
|
|
assert key in context.invocation.container_metadata
|
|
|
|
|
|
@when("I create a ToolInvocation without container_metadata")
|
|
def step_create_invocation_without_metadata(context: Any) -> None:
|
|
context.invocation = ToolInvocation(
|
|
plan_id="01TESTPLANID000000000000000",
|
|
tool_name="ns/tool",
|
|
)
|
|
|
|
|
|
@then("the ToolInvocation container_metadata should be None")
|
|
def step_check_invocation_metadata_none(context: Any) -> None:
|
|
assert context.invocation.container_metadata is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# ContainerExecutionError / ContainerTimeoutError
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when('I raise a ContainerExecutionError with exit_code {code:d} and stderr "{stderr}"')
|
|
def step_raise_container_error(context: Any, code: int, stderr: str) -> None:
|
|
context.error = ContainerExecutionError("test error", exit_code=code, stderr=stderr)
|
|
|
|
|
|
@then("the error exit_code should be {code:d}")
|
|
def step_check_error_exit_code(context: Any, code: int) -> None:
|
|
assert context.error.exit_code == code
|
|
|
|
|
|
@then('the error stderr should be "{expected}"')
|
|
def step_check_error_stderr(context: Any, expected: str) -> None:
|
|
assert context.error.stderr == expected
|
|
|
|
|
|
@when("I raise a ContainerTimeoutError with timeout {timeout:d}")
|
|
def step_raise_timeout_error(context: Any, timeout: int) -> None:
|
|
context.error = ContainerTimeoutError(timeout_seconds=timeout)
|
|
|
|
|
|
@then('the timeout error message should mention "{text}"')
|
|
def step_check_timeout_msg(context: Any, text: str) -> None:
|
|
assert text in str(context.error)
|
|
|
|
|
|
@then("the timeout error timed_out should be true")
|
|
def step_check_timeout_timed_out(context: Any) -> None:
|
|
assert context.error.timed_out is True
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# execute_tool input validation (P2-19)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when("I execute tool with empty tool_name")
|
|
def step_execute_empty_tool_name(context: Any) -> None:
|
|
try:
|
|
context.executor.execute_tool("", {"key": "value"})
|
|
context.error = None
|
|
except (ValueError, TypeError) as exc:
|
|
context.error = exc
|
|
|
|
|
|
@when("I execute tool with non-dict inputs")
|
|
def step_execute_non_dict_inputs(context: Any) -> None:
|
|
try:
|
|
context.executor.execute_tool("test_tool", "not a dict") # type: ignore[arg-type]
|
|
context.error = None
|
|
except (ValueError, TypeError) as exc:
|
|
context.error = exc
|
|
|
|
|
|
@when("I execute tool with negative timeout")
|
|
def step_execute_negative_timeout(context: Any) -> None:
|
|
try:
|
|
context.executor.execute_tool("test_tool", {}, timeout_seconds=-1)
|
|
context.error = None
|
|
except (ValueError, TypeError) as exc:
|
|
context.error = exc
|
|
|
|
|
|
@then('a container TypeError should be raised with message "{text}"')
|
|
def step_check_type_error(context: Any, text: str) -> None:
|
|
assert context.error is not None, "Expected a TypeError but none was raised"
|
|
assert isinstance(context.error, TypeError), (
|
|
f"Expected TypeError, got {type(context.error).__name__}"
|
|
)
|
|
assert text in str(context.error), (
|
|
f"Expected '{text}' in error message, got: {context.error}"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# extract_container_metadata helper (P1-9)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when("I create a ToolResult with container metadata in metadata dict")
|
|
def step_create_tool_result_with_container_meta(context: Any) -> None:
|
|
context.tool_result_with_meta = ToolResult(
|
|
success=True,
|
|
output={"result": "ok"},
|
|
duration_ms=100.0,
|
|
metadata={
|
|
"container": {
|
|
"container_id": "abc123",
|
|
"image": "node:20",
|
|
"exec_time_ms": 150.0,
|
|
}
|
|
},
|
|
)
|
|
|
|
|
|
@then("extract_container_metadata should return the container dict")
|
|
def step_check_extract_container_metadata(context: Any) -> None:
|
|
extracted = ContainerToolExecutor.extract_container_metadata(
|
|
context.tool_result_with_meta
|
|
)
|
|
assert extracted is not None, "Expected container metadata, got None"
|
|
assert extracted["container_id"] == "abc123"
|
|
assert extracted["image"] == "node:20"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# sync_results_to_host
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given(
|
|
"I have a ContainerToolExecutor with a mock that succeeds on sync using a temp dir"
|
|
)
|
|
def step_executor_mock_sync_success(context: Any) -> None:
|
|
# Use a real temp directory so the file-write in sync_results_to_host
|
|
# succeeds without polluting a fixed path. Explicitly use /tmp so
|
|
# the path never falls under the container_root "/workspace" — CI
|
|
# workspaces live at /workspace/… and PathMapper rejects overlapping
|
|
# host_root / container_root.
|
|
context.sync_tmp_dir = tempfile.mkdtemp(prefix="ca_sync_test_", dir="/tmp")
|
|
context.add_cleanup(
|
|
lambda d=context.sync_tmp_dir: shutil.rmtree(d, ignore_errors=True)
|
|
)
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="sync-ctr",
|
|
host_sandbox_path=context.sync_tmp_dir,
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout="file content",
|
|
stderr="",
|
|
raw_stdout=b"file content",
|
|
exit_code=0,
|
|
duration_ms=20.0,
|
|
timed_out=False,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@when('I sync container path "{path}" to host')
|
|
def step_sync_to_host(context: Any, path: str) -> None:
|
|
context.synced_path = context.executor.sync_results_to_host(path)
|
|
|
|
|
|
@then("the synced host path should be under the host sandbox root")
|
|
def step_check_synced_path(context: Any) -> None:
|
|
# Compare resolved paths so the assertion holds when /tmp is a
|
|
# symlink or bind-mount redirect (common in CI Docker containers).
|
|
resolved_sandbox = str(Path(context.sync_tmp_dir).resolve())
|
|
resolved_synced = str(Path(context.synced_path).resolve())
|
|
assert resolved_synced.startswith(resolved_sandbox), (
|
|
f"Expected path under {resolved_sandbox}, got {resolved_synced}"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Signal exit code preservation (H5 fix)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@then("the tool result metadata should have exit_code {code:d}")
|
|
def step_check_metadata_exit_code(context: Any, code: int) -> None:
|
|
meta = context.tool_result.metadata.get("container", {})
|
|
assert meta.get("exit_code") == code, (
|
|
f"Expected exit_code={code} in metadata.container, got {meta.get('exit_code')}"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Output truncation (C2 fix)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that returns oversized output")
|
|
def step_executor_mock_oversized_output(context: Any) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="big-ctr",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
context.max_output_bytes = _MAX_OUTPUT_BYTES
|
|
|
|
# Mock subprocess.Popen to return oversized output, so that
|
|
# _run_command's bounded-read + truncation logic is exercised.
|
|
oversized_bytes = b"x" * (_MAX_OUTPUT_BYTES + 1024)
|
|
mock_proc = MagicMock()
|
|
mock_proc.stdout = io.BytesIO(oversized_bytes)
|
|
mock_proc.stderr = io.BytesIO(b"")
|
|
mock_proc.stdin = MagicMock()
|
|
mock_proc.returncode = 0
|
|
mock_proc.wait = MagicMock(return_value=0)
|
|
patcher = patch(
|
|
"cleveragents.tool.container_executor.subprocess.Popen",
|
|
return_value=mock_proc,
|
|
)
|
|
patcher.start()
|
|
context.oversized_patcher = patcher
|
|
context.executor = executor
|
|
# Register cleanup so the patcher is stopped even if the scenario
|
|
# aborts before the "When" step runs (P2-18).
|
|
context.add_cleanup(patcher.stop)
|
|
|
|
|
|
@then("the container tool result stdout should be truncated")
|
|
def step_check_output_truncated(context: Any) -> None:
|
|
# The output was parsed from stdout; verify the raw_output is within limit.
|
|
# Since the oversized output is not valid JSON, it ends up as raw_output.
|
|
raw = context.tool_result.output.get("raw_output", "")
|
|
max_bytes = context.max_output_bytes
|
|
assert len(raw.encode("utf-8")) <= max_bytes, (
|
|
f"Expected output <= {max_bytes} bytes, got {len(raw.encode('utf-8'))}"
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Path traversal protection (C3 fix)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@when("I attempt to sync a path that traverses outside the sandbox")
|
|
def step_sync_traversal_attempt(context: Any) -> None:
|
|
try:
|
|
# This path maps to a host path outside the sandbox due to ".."
|
|
context.executor.sync_results_to_host("/workspace/../../../etc/passwd")
|
|
context.traversal_error = None
|
|
except ContainerExecutionError as exc:
|
|
context.traversal_error = exc
|
|
|
|
|
|
@then("a ContainerExecutionError should be raised for path traversal")
|
|
def step_check_traversal_error(context: Any) -> None:
|
|
assert context.traversal_error is not None, (
|
|
"Expected ContainerExecutionError for path traversal, but none was raised"
|
|
)
|
|
assert "outside" in str(context.traversal_error).lower(), (
|
|
f"Expected 'outside' in error message, got: {context.traversal_error}"
|
|
)
|
|
|
|
|
|
# =========================================================================
|
|
# PathMapper safe initialisation guards
|
|
# =========================================================================
|
|
|
|
|
|
@when("I try to create a PathMapper with null bytes in host_root")
|
|
def step_pathmapper_null_host(context: Any) -> None:
|
|
try:
|
|
PathMapper(host_root="/tmp/sand\x00box", container_root="/workspace")
|
|
context.pm_error = None
|
|
except ValueError as exc:
|
|
context.pm_error = exc
|
|
|
|
|
|
@when("I try to create a PathMapper with null bytes in container_root")
|
|
def step_pathmapper_null_container(context: Any) -> None:
|
|
try:
|
|
PathMapper(host_root="/tmp/sandbox", container_root="/work\x00space")
|
|
context.pm_error = None
|
|
except ValueError as exc:
|
|
context.pm_error = exc
|
|
|
|
|
|
@when("I try to create a PathMapper with empty host_root")
|
|
def step_pathmapper_empty_host(context: Any) -> None:
|
|
try:
|
|
PathMapper(host_root="", container_root="/workspace")
|
|
context.pm_error = None
|
|
except ValueError as exc:
|
|
context.pm_error = exc
|
|
|
|
|
|
@when('I try to create a PathMapper with host_root "/"')
|
|
def step_pathmapper_root_host(context: Any) -> None:
|
|
try:
|
|
PathMapper(host_root="/", container_root="/workspace")
|
|
context.pm_error = None
|
|
except ValueError as exc:
|
|
context.pm_error = exc
|
|
|
|
|
|
@when('I try to create a PathMapper with container_root "/"')
|
|
def step_pathmapper_root_container(context: Any) -> None:
|
|
try:
|
|
PathMapper(host_root="/tmp/sandbox", container_root="/")
|
|
context.pm_error = None
|
|
except ValueError as exc:
|
|
context.pm_error = exc
|
|
|
|
|
|
@when("I try to create a PathMapper with overlapping roots")
|
|
def step_pathmapper_overlapping(context: Any) -> None:
|
|
try:
|
|
PathMapper(host_root="/workspace/sub", container_root="/workspace")
|
|
context.pm_error = None
|
|
except ValueError as exc:
|
|
context.pm_error = exc
|
|
|
|
|
|
@then('a PathMapper ValueError should be raised mentioning "{text}"')
|
|
def step_check_pathmapper_error(context: Any, text: str) -> None:
|
|
assert context.pm_error is not None, (
|
|
"Expected a ValueError from PathMapper but none was raised"
|
|
)
|
|
assert text.lower() in str(context.pm_error).lower(), (
|
|
f"Expected '{text}' in error: {context.pm_error}"
|
|
)
|
|
|
|
|
|
@then("a path with null bytes should not be a host path")
|
|
def step_null_bytes_not_host(context: Any) -> None:
|
|
assert not context.path_mapper.is_host_path("/tmp/sandbox/\x00evil")
|
|
|
|
|
|
@then("a path with null bytes should not be a container path")
|
|
def step_null_bytes_not_container(context: Any) -> None:
|
|
assert not context.path_mapper.is_container_path("/workspace/\x00evil")
|
|
|
|
|
|
# =========================================================================
|
|
# ContainerConfig safe initialisation guards
|
|
# =========================================================================
|
|
|
|
|
|
@when('I try to create a ContainerConfig with workspace_folder "{folder}"')
|
|
def step_config_bad_workspace(context: Any, folder: str) -> None:
|
|
try:
|
|
ContainerConfig(workspace_folder=folder)
|
|
context.cfg_error = None
|
|
except ValueError as exc:
|
|
context.cfg_error = exc
|
|
|
|
|
|
@then('a ContainerConfig ValueError should be raised mentioning "{text}"')
|
|
def step_check_config_error(context: Any, text: str) -> None:
|
|
assert context.cfg_error is not None, (
|
|
"Expected a ValueError from ContainerConfig but none was raised"
|
|
)
|
|
assert text.lower() in str(context.cfg_error).lower(), (
|
|
f"Expected '{text}' in error: {context.cfg_error}"
|
|
)
|
|
|
|
|
|
# =========================================================================
|
|
# ContainerToolExecutor binary resolution
|
|
# =========================================================================
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with no container_id configured")
|
|
def step_executor_no_cid(context: Any) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
context.executor = executor
|
|
|
|
|
|
@when("I request the target CLI arguments")
|
|
def step_request_target_args(context: Any) -> None:
|
|
context.target_args = context.executor._devcontainer_target_args()
|
|
|
|
|
|
@then('the target args should use "--workspace-folder"')
|
|
def step_check_workspace_folder_args(context: Any) -> None:
|
|
assert "--workspace-folder" in context.target_args, (
|
|
f"Expected '--workspace-folder' in {context.target_args}"
|
|
)
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with no devcontainer binary")
|
|
def step_executor_no_binary(context: Any) -> None:
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="test",
|
|
host_sandbox_path="/tmp/sandbox",
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
# Ensure the binary is None
|
|
executor._devcontainer_bin = None
|
|
context.executor = executor
|
|
|
|
|
|
@when("I try to require the devcontainer binary")
|
|
def step_require_binary(context: Any) -> None:
|
|
try:
|
|
context.executor._require_devcontainer_bin()
|
|
context.binary_error = None
|
|
except ContainerExecutionError as exc:
|
|
context.binary_error = exc
|
|
|
|
|
|
@then("a ContainerExecutionError should be raised about missing binary")
|
|
def step_check_missing_binary_error(context: Any) -> None:
|
|
assert context.binary_error is not None, (
|
|
"Expected ContainerExecutionError for missing binary"
|
|
)
|
|
assert "not found" in str(context.binary_error).lower(), (
|
|
f"Expected 'not found' in error: {context.binary_error}"
|
|
)
|
|
|
|
|
|
# =========================================================================
|
|
# sync_results_to_host cleanup error paths
|
|
# =========================================================================
|
|
|
|
|
|
@when("I try to sync a path containing null bytes")
|
|
def step_sync_null_bytes(context: Any) -> None:
|
|
try:
|
|
context.executor.sync_results_to_host("/workspace/\x00evil.txt")
|
|
context.sync_error = None
|
|
except ValueError as exc:
|
|
context.sync_error = exc
|
|
|
|
|
|
@then('a sync ValueError should be raised mentioning "{text}"')
|
|
def step_check_sync_value_error(context: Any, text: str) -> None:
|
|
assert context.sync_error is not None, (
|
|
"Expected a ValueError from sync but none was raised"
|
|
)
|
|
assert text.lower() in str(context.sync_error).lower(), (
|
|
f"Expected '{text}' in error: {context.sync_error}"
|
|
)
|
|
|
|
|
|
@when("I try to sync a relative container path")
|
|
def step_sync_relative_path(context: Any) -> None:
|
|
try:
|
|
context.executor.sync_results_to_host("relative/path.txt")
|
|
context.sync_error = None
|
|
except ValueError as exc:
|
|
context.sync_error = exc
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that times out on sync")
|
|
def step_executor_sync_timeout(context: Any) -> None:
|
|
context.sync_tmp_dir = tempfile.mkdtemp(prefix="ca_sync_timeout_", dir="/tmp")
|
|
context.add_cleanup(
|
|
lambda d=context.sync_tmp_dir: shutil.rmtree(d, ignore_errors=True)
|
|
)
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="sync-timeout-ctr",
|
|
host_sandbox_path=context.sync_tmp_dir,
|
|
timeout_seconds=5,
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout="",
|
|
stderr="timed out",
|
|
raw_stdout=b"",
|
|
exit_code=-1,
|
|
duration_ms=5000.0,
|
|
timed_out=True,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@when("I try to sync container path to host with timeout")
|
|
def step_sync_with_timeout(context: Any) -> None:
|
|
try:
|
|
context.executor.sync_results_to_host("/workspace/output.txt")
|
|
context.sync_timeout_error = None
|
|
except ContainerTimeoutError as exc:
|
|
context.sync_timeout_error = exc
|
|
|
|
|
|
@then("a ContainerTimeoutError should be raised from sync")
|
|
def step_check_sync_timeout(context: Any) -> None:
|
|
assert context.sync_timeout_error is not None, (
|
|
"Expected ContainerTimeoutError from sync"
|
|
)
|
|
assert context.sync_timeout_error.timed_out is True
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that returns non-zero exit on sync")
|
|
def step_executor_sync_fail(context: Any) -> None:
|
|
context.sync_tmp_dir = tempfile.mkdtemp(prefix="ca_sync_fail_", dir="/tmp")
|
|
context.add_cleanup(
|
|
lambda d=context.sync_tmp_dir: shutil.rmtree(d, ignore_errors=True)
|
|
)
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="sync-fail-ctr",
|
|
host_sandbox_path=context.sync_tmp_dir,
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout="",
|
|
stderr="cat: no such file",
|
|
raw_stdout=b"",
|
|
exit_code=1,
|
|
duration_ms=10.0,
|
|
timed_out=False,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@when("I try to sync container path to host with failure")
|
|
def step_sync_with_failure(context: Any) -> None:
|
|
try:
|
|
context.executor.sync_results_to_host("/workspace/missing.txt")
|
|
context.sync_exec_error = None
|
|
except ContainerExecutionError as exc:
|
|
context.sync_exec_error = exc
|
|
|
|
|
|
@then("a ContainerExecutionError should be raised from sync failure")
|
|
def step_check_sync_exec_error(context: Any) -> None:
|
|
assert context.sync_exec_error is not None, (
|
|
"Expected ContainerExecutionError from sync failure"
|
|
)
|
|
assert context.sync_exec_error.exit_code == 1
|
|
|
|
|
|
@given("I have a ContainerToolExecutor with a mock that triggers OSError on write")
|
|
def step_executor_sync_oserror(context: Any) -> None:
|
|
context.sync_tmp_dir = tempfile.mkdtemp(prefix="ca_sync_oserr_", dir="/tmp")
|
|
context.add_cleanup(
|
|
lambda d=context.sync_tmp_dir: shutil.rmtree(d, ignore_errors=True)
|
|
)
|
|
config = ContainerConfig(
|
|
workspace_folder="/workspace",
|
|
container_id="sync-oserr-ctr",
|
|
host_sandbox_path=context.sync_tmp_dir,
|
|
)
|
|
executor = ContainerToolExecutor(config)
|
|
_set_fake_devcontainer_bin(executor)
|
|
executor._run_command = MagicMock(
|
|
return_value=_ExecResult(
|
|
stdout="content",
|
|
stderr="",
|
|
raw_stdout=b"content",
|
|
exit_code=0,
|
|
duration_ms=10.0,
|
|
timed_out=False,
|
|
)
|
|
)
|
|
context.executor = executor
|
|
|
|
|
|
@when("I try to sync container path to host triggering write error")
|
|
def step_sync_trigger_write_error(context: Any) -> None:
|
|
# Patch os.open to simulate a permission error during file write
|
|
patcher = patch(
|
|
"cleveragents.tool.container_executor.os.open",
|
|
side_effect=OSError("Permission denied"),
|
|
)
|
|
patcher.start()
|
|
context.add_cleanup(patcher.stop)
|
|
try:
|
|
context.executor.sync_results_to_host("/workspace/output.txt")
|
|
context.sync_write_error = None
|
|
except ContainerExecutionError as exc:
|
|
context.sync_write_error = exc
|
|
|
|
|
|
@then("a ContainerExecutionError should be raised from write failure")
|
|
def step_check_sync_write_error(context: Any) -> None:
|
|
assert context.sync_write_error is not None, (
|
|
"Expected ContainerExecutionError from write failure"
|
|
)
|
|
assert (
|
|
"Permission denied" in str(context.sync_write_error)
|
|
or "write" in str(context.sync_write_error).lower()
|
|
), f"Unexpected error: {context.sync_write_error}"
|
|
|
|
|
|
# =========================================================================
|
|
# _looks_like_path heuristic boundary checks
|
|
# =========================================================================
|
|
|
|
|
|
@when("I check whether a string with newlines looks like a path")
|
|
def step_looks_like_path_newline(context: Any) -> None:
|
|
context.looks_like = _looks_like_path("/some/path\nwith newline")
|
|
|
|
|
|
@when("I check whether a string with null bytes looks like a path")
|
|
def step_looks_like_path_null(context: Any) -> None:
|
|
context.looks_like = _looks_like_path("/some/\x00path")
|
|
|
|
|
|
@when("I check whether a protocol-relative URI looks like a path")
|
|
def step_looks_like_path_proto_relative(context: Any) -> None:
|
|
context.looks_like = _looks_like_path("//cdn.example.com/resource")
|
|
|
|
|
|
@when("I check whether a path with query string looks like a path")
|
|
def step_looks_like_path_query(context: Any) -> None:
|
|
context.looks_like = _looks_like_path("/api/v1/resource?key=value")
|
|
|
|
|
|
@when("I check whether a path with fragment identifier looks like a path")
|
|
def step_looks_like_path_fragment(context: Any) -> None:
|
|
context.looks_like = _looks_like_path("/docs/page#section")
|
|
|
|
|
|
@when("I check whether a URL-like string looks like a path")
|
|
def step_looks_like_path_url(context: Any) -> None:
|
|
context.looks_like = _looks_like_path("/prefix://host/path")
|
|
|
|
|
|
@then("it should not look like a path")
|
|
def step_check_not_looks_like_path(context: Any) -> None:
|
|
assert context.looks_like is False, "Expected False from _looks_like_path"
|
|
|
|
|
|
# =========================================================================
|
|
# Recursion depth guards on path mapping
|
|
# =========================================================================
|
|
|
|
|
|
@when("I map deeply nested input paths exceeding the recursion limit")
|
|
def step_deep_input_mapping(context: Any) -> None:
|
|
# Build a dict nested beyond _MAX_RECURSION_DEPTH
|
|
nested: dict[str, Any] = {"path": "/tmp/sandbox/file.py"}
|
|
current = nested
|
|
for i in range(_MAX_RECURSION_DEPTH + 2):
|
|
current["inner"] = {"path": f"/tmp/sandbox/f{i}.py"}
|
|
current = current["inner"]
|
|
context.deep_mapped = context.executor._map_input_paths(nested)
|
|
|
|
|
|
@then("the input mapping should return without error")
|
|
def step_check_deep_input_ok(context: Any) -> None:
|
|
assert context.deep_mapped is not None
|
|
|
|
|
|
@when("I map deeply nested output paths exceeding the recursion limit")
|
|
def step_deep_output_mapping(context: Any) -> None:
|
|
nested: dict[str, Any] = {"path": "/workspace/file.py"}
|
|
current = nested
|
|
for i in range(_MAX_RECURSION_DEPTH + 2):
|
|
current["inner"] = {"path": f"/workspace/f{i}.py"}
|
|
current = current["inner"]
|
|
context.deep_mapped = context.executor._map_output_paths(nested)
|
|
|
|
|
|
@then("the output mapping should return without error")
|
|
def step_check_deep_output_ok(context: Any) -> None:
|
|
assert context.deep_mapped is not None
|
|
|
|
|
|
# =========================================================================
|
|
# _parse_output edge cases
|
|
# =========================================================================
|
|
|
|
|
|
@when("I parse stdout containing a JSON array")
|
|
def step_parse_json_array(context: Any) -> None:
|
|
context.parsed_output = ContainerToolExecutor._parse_output("[1, 2, 3]")
|
|
|
|
|
|
@then('the parsed output should wrap it under "result" key')
|
|
def step_check_result_key(context: Any) -> None:
|
|
assert "result" in context.parsed_output, (
|
|
f"Expected 'result' key, got {context.parsed_output}"
|
|
)
|
|
assert context.parsed_output["result"] == [1, 2, 3]
|
|
|
|
|
|
@when("I parse empty stdout")
|
|
def step_parse_empty(context: Any) -> None:
|
|
context.parsed_output = ContainerToolExecutor._parse_output(" ")
|
|
|
|
|
|
@then("the parsed output should be an empty dict")
|
|
def step_check_empty_parsed(context: Any) -> None:
|
|
assert context.parsed_output == {}
|
|
|
|
|
|
# =========================================================================
|
|
# _map_output_paths raw_output preservation
|
|
# =========================================================================
|
|
|
|
|
|
@when("I map output containing a raw_output key with container path")
|
|
def step_map_raw_output(context: Any) -> None:
|
|
output = {
|
|
"raw_output": "/workspace/data/results.txt\nmore lines here",
|
|
"path": "/workspace/file.py",
|
|
}
|
|
context.mapped_output = context.executor._map_output_paths(output)
|
|
|
|
|
|
@then("the raw_output value should remain unmapped")
|
|
def step_check_raw_output_unmapped(context: Any) -> None:
|
|
raw = context.mapped_output["raw_output"]
|
|
assert raw.startswith("/workspace"), f"raw_output should stay unmapped, got: {raw}"
|
|
# But the path key should be mapped to host
|
|
assert context.mapped_output["path"].startswith("/tmp/sandbox"), (
|
|
f"path key should be mapped, got: {context.mapped_output['path']}"
|
|
)
|
|
|
|
|
|
# =========================================================================
|
|
# execute_tool with non-serialisable inputs
|
|
# =========================================================================
|
|
|
|
|
|
# =========================================================================
|
|
# List-type value mapping branches
|
|
# =========================================================================
|
|
|
|
|
|
@when("I map tool inputs containing a list of host paths")
|
|
def step_map_list_inputs(context: Any) -> None:
|
|
inputs = {"paths": ["/tmp/sandbox/a.py", "/tmp/sandbox/b.py", "/usr/lib/x.so"]}
|
|
context.mapped_inputs = context.executor._map_input_paths(inputs)
|
|
|
|
|
|
@then("the list values should be mapped to container paths")
|
|
def step_check_list_inputs_mapped(context: Any) -> None:
|
|
mapped = context.mapped_inputs["paths"]
|
|
assert mapped[0] == "/workspace/a.py", f"Expected /workspace/a.py, got {mapped[0]}"
|
|
assert mapped[1] == "/workspace/b.py", f"Expected /workspace/b.py, got {mapped[1]}"
|
|
# Path outside host_root should remain unchanged
|
|
assert mapped[2] == "/usr/lib/x.so", f"Expected /usr/lib/x.so, got {mapped[2]}"
|
|
|
|
|
|
@when("I map tool output containing a list of container paths")
|
|
def step_map_list_outputs(context: Any) -> None:
|
|
output = {"paths": ["/workspace/out1.txt", "/workspace/out2.txt"]}
|
|
context.mapped_output = context.executor._map_output_paths(output)
|
|
|
|
|
|
@then("the list values should be mapped to host paths")
|
|
def step_check_list_outputs_mapped(context: Any) -> None:
|
|
mapped = context.mapped_output["paths"]
|
|
assert mapped[0] == "/tmp/sandbox/out1.txt", (
|
|
f"Expected /tmp/sandbox/out1.txt, got {mapped[0]}"
|
|
)
|
|
assert mapped[1] == "/tmp/sandbox/out2.txt", (
|
|
f"Expected /tmp/sandbox/out2.txt, got {mapped[1]}"
|
|
)
|
|
|
|
|
|
# =========================================================================
|
|
# PathMapper null byte rejection in mapping methods
|
|
# =========================================================================
|
|
|
|
|
|
@when("I try to map a host path with null bytes to container")
|
|
def step_map_host_null(context: Any) -> None:
|
|
try:
|
|
context.path_mapper.host_to_container("/tmp/sandbox/\x00evil")
|
|
context.mapping_error = None
|
|
except ValueError as exc:
|
|
context.mapping_error = exc
|
|
|
|
|
|
@when("I try to map a container path with null bytes to host")
|
|
def step_map_container_null(context: Any) -> None:
|
|
try:
|
|
context.path_mapper.container_to_host("/workspace/\x00evil")
|
|
context.mapping_error = None
|
|
except ValueError as exc:
|
|
context.mapping_error = exc
|
|
|
|
|
|
@then("a mapping ValueError should be raised about null bytes")
|
|
def step_check_mapping_null_error(context: Any) -> None:
|
|
assert context.mapping_error is not None, (
|
|
"Expected ValueError for null bytes in path mapping"
|
|
)
|
|
assert "null" in str(context.mapping_error).lower(), (
|
|
f"Expected 'null' in error: {context.mapping_error}"
|
|
)
|
|
|
|
|
|
@when("I execute a tool with non-serialisable inputs")
|
|
def step_execute_non_serialisable(context: Any) -> None:
|
|
context.tool_result = context.executor.execute_tool("some_tool", {"bad": object()})
|
|
|
|
|
|
# =========================================================================
|
|
# ToolRunner full lifecycle
|
|
# =========================================================================
|
|
|
|
|
|
def _make_dummy_spec(name: str = "ns/test-tool") -> ToolSpec:
|
|
"""Create a minimal ToolSpec for lifecycle tests."""
|
|
return ToolSpec(
|
|
name=name,
|
|
description="A test tool for lifecycle validation",
|
|
input_schema={},
|
|
handler=lambda inputs: {"status": "ok"},
|
|
)
|
|
|
|
|
|
@given("I have a ToolRunner with a populated registry")
|
|
def step_runner_populated_registry(context: Any) -> None:
|
|
registry = _TR()
|
|
spec = _make_dummy_spec()
|
|
registry.register(spec)
|
|
context.runner = ToolRunner(registry=registry)
|
|
context.test_tool_name = spec.name
|
|
|
|
|
|
@when("I call discover on the runner")
|
|
def step_call_discover(context: Any) -> None:
|
|
context.discovered = context.runner.discover()
|
|
|
|
|
|
@then("the discovered tools list should not be empty")
|
|
def step_check_discovered(context: Any) -> None:
|
|
assert len(context.discovered) > 0
|
|
|
|
|
|
@when("I activate a tool on the runner")
|
|
def step_activate_tool(context: Any) -> None:
|
|
context.activated_spec = context.runner.activate(context.test_tool_name)
|
|
|
|
|
|
@then("the activated tool spec should be returned")
|
|
def step_check_activated(context: Any) -> None:
|
|
assert context.activated_spec is not None
|
|
assert context.activated_spec.name == context.test_tool_name
|
|
|
|
|
|
@when("I try to activate an unknown tool")
|
|
def step_activate_unknown(context: Any) -> None:
|
|
try:
|
|
context.runner.activate("ns/nonexistent")
|
|
context.activation_error = None
|
|
except ToolError as exc:
|
|
context.activation_error = exc
|
|
|
|
|
|
@then("a ToolError should be raised for activation")
|
|
def step_check_activation_error(context: Any) -> None:
|
|
assert context.activation_error is not None, (
|
|
"Expected ToolError for unknown tool activation"
|
|
)
|
|
assert "not found" in context.activation_error.details.lower()
|
|
|
|
|
|
@when("I activate and then deactivate a tool")
|
|
def step_activate_deactivate(context: Any) -> None:
|
|
context.runner.activate(context.test_tool_name)
|
|
context.deactivate_result = context.runner.deactivate(context.test_tool_name)
|
|
|
|
|
|
@then("deactivate should return true")
|
|
def step_check_deactivate_true(context: Any) -> None:
|
|
assert context.deactivate_result is True
|
|
|
|
|
|
@then("a second deactivate should return false")
|
|
def step_check_second_deactivate(context: Any) -> None:
|
|
result = context.runner.deactivate(context.test_tool_name)
|
|
assert result is False
|
|
|
|
|
|
# =========================================================================
|
|
# ToolRunner host execution path
|
|
# =========================================================================
|
|
|
|
|
|
@given("I have a ToolRunner with a host-routed tool")
|
|
def step_runner_host_tool(context: Any) -> None:
|
|
registry = _TR()
|
|
spec = ToolSpec(
|
|
name="ns/host-tool",
|
|
description="Host-routed tool for testing",
|
|
input_schema={},
|
|
handler=lambda inputs: {"data": "from_handler", "echo": inputs.get("key")},
|
|
)
|
|
registry.register(spec)
|
|
context.runner = ToolRunner(registry=registry)
|
|
context.host_tool_name = "ns/host-tool"
|
|
|
|
|
|
@when("I execute the host tool with valid inputs")
|
|
def step_execute_host_tool(context: Any) -> None:
|
|
context.tool_result = context.runner.execute(
|
|
context.host_tool_name, {"key": "value"}
|
|
)
|
|
|
|
|
|
@then("the host tool result should indicate success")
|
|
def step_check_host_success(context: Any) -> None:
|
|
assert context.tool_result.success is True
|
|
|
|
|
|
@then("the host tool result output should contain handler data")
|
|
def step_check_host_output(context: Any) -> None:
|
|
assert context.tool_result.output.get("data") == "from_handler"
|
|
|
|
|
|
@given("I have a ToolRunner with a tool returning non-dict output")
|
|
def step_runner_nondict_output(context: Any) -> None:
|
|
registry = _TR()
|
|
spec = ToolSpec(
|
|
name="ns/nondict-tool",
|
|
description="Returns a string",
|
|
input_schema={},
|
|
handler=lambda inputs: "plain_string_result",
|
|
)
|
|
registry.register(spec)
|
|
context.runner = ToolRunner(registry=registry)
|
|
|
|
|
|
@when("I execute the non-dict-output tool")
|
|
def step_execute_nondict(context: Any) -> None:
|
|
context.tool_result = context.runner.execute("ns/nondict-tool", {})
|
|
|
|
|
|
@then('the result output should wrap the value under "result"')
|
|
def step_check_wrapped_result(context: Any) -> None:
|
|
assert context.tool_result.success is True
|
|
assert context.tool_result.output.get("result") == "plain_string_result"
|
|
|
|
|
|
@given("I have a ToolRunner with a tool that raises an exception")
|
|
def step_runner_raising_tool(context: Any) -> None:
|
|
registry = _TR()
|
|
|
|
def _bomb(inputs: dict) -> dict:
|
|
raise RuntimeError("handler exploded")
|
|
|
|
spec = ToolSpec(
|
|
name="ns/bomb-tool",
|
|
description="Always raises",
|
|
input_schema={},
|
|
handler=_bomb,
|
|
)
|
|
registry.register(spec)
|
|
context.runner = ToolRunner(registry=registry)
|
|
|
|
|
|
@when("I execute the raising tool")
|
|
def step_execute_raising(context: Any) -> None:
|
|
context.tool_result = context.runner.execute("ns/bomb-tool", {})
|
|
|
|
|
|
@then("the tool result should indicate failure with exception info")
|
|
def step_check_failure_with_exc(context: Any) -> None:
|
|
assert context.tool_result.success is False
|
|
assert "RuntimeError" in context.tool_result.error
|
|
assert "handler exploded" in context.tool_result.error
|
|
|
|
|
|
@given("I have a ToolRunner with a tool returning non-serialisable output")
|
|
def step_runner_nonserial_output(context: Any) -> None:
|
|
registry = _TR()
|
|
spec = ToolSpec(
|
|
name="ns/nonserial-tool",
|
|
description="Returns non-serialisable data",
|
|
input_schema={},
|
|
handler=lambda inputs: {"bad_value": object()},
|
|
)
|
|
registry.register(spec)
|
|
context.runner = ToolRunner(registry=registry)
|
|
|
|
|
|
@when("I execute the non-serialisable-output tool")
|
|
def step_execute_nonserial(context: Any) -> None:
|
|
context.tool_result = context.runner.execute("ns/nonserial-tool", {})
|
|
|
|
|
|
@then('the tool result should indicate failure mentioning "{text}"')
|
|
def step_check_failure_mentioning(context: Any, text: str) -> None:
|
|
assert context.tool_result.success is False
|
|
assert text in context.tool_result.error, (
|
|
f"Expected '{text}' in error: {context.tool_result.error}"
|
|
)
|
|
|
|
|
|
@when("I execute the host tool with non-serialisable inputs")
|
|
def step_execute_host_nonserial_inputs(context: Any) -> None:
|
|
context.tool_result = context.runner.execute(
|
|
context.host_tool_name, {"bad": object()}
|
|
)
|
|
|
|
|
|
@given("I have a ToolRunner with an empty registry")
|
|
def step_runner_empty_registry(context: Any) -> None:
|
|
registry = _TR()
|
|
context.runner = ToolRunner(registry=registry)
|
|
|
|
|
|
@when("I try to execute an unregistered tool")
|
|
def step_execute_unregistered(context: Any) -> None:
|
|
try:
|
|
context.runner.execute("ns/ghost", {})
|
|
context.exec_error = None
|
|
except ToolError as exc:
|
|
context.exec_error = exc
|
|
|
|
|
|
@then("a ToolError should be raised for execution")
|
|
def step_check_exec_tool_error(context: Any) -> None:
|
|
assert context.exec_error is not None, (
|
|
"Expected ToolError for unregistered tool execution"
|
|
)
|
|
assert "not found" in context.exec_error.details.lower()
|
|
|
|
|
|
# =========================================================================
|
|
# ToolRunner container routing — additional edge cases
|
|
# =========================================================================
|
|
|
|
|
|
@given("I have a ToolRunner with an unavailable-container resolver")
|
|
def step_runner_unavailable_container(context: Any) -> None:
|
|
registry = MagicMock(spec=_TR)
|
|
spec = _make_dummy_spec("ns/ctr-tool")
|
|
registry.get.return_value = spec
|
|
|
|
resolver = MagicMock(spec=ExecutionEnvironmentResolver)
|
|
resolver.resolve_and_validate.side_effect = ContainerUnavailableError(
|
|
"No container resource linked"
|
|
)
|
|
context.runner = ToolRunner(
|
|
registry=registry,
|
|
execution_environment_resolver=resolver,
|
|
)
|
|
|
|
|
|
@when("I execute a tool through the unavailable-container runner")
|
|
def step_execute_unavailable_ctr(context: Any) -> None:
|
|
context.tool_result = context.runner.execute("ns/ctr-tool", {})
|
|
|
|
|
|
@then("the tool result should indicate container unavailable")
|
|
def step_check_container_unavailable(context: Any) -> None:
|
|
assert context.tool_result.success is False
|
|
assert "No container resource linked" in context.tool_result.error
|
|
|
|
|
|
@given("I have a ToolRunner with a container executor and required-field tool")
|
|
def step_runner_required_fields(context: Any) -> None:
|
|
registry = MagicMock(spec=_TR)
|
|
spec = ToolSpec(
|
|
name="ns/strict-tool",
|
|
description="Tool with required fields",
|
|
input_schema={
|
|
"type": "object",
|
|
"required": ["file_path", "mode"],
|
|
"properties": {
|
|
"file_path": {"type": "string"},
|
|
"mode": {"type": "string"},
|
|
},
|
|
},
|
|
handler=lambda inputs: {"ok": True},
|
|
)
|
|
registry.get.return_value = spec
|
|
|
|
resolver = MagicMock(spec=ExecutionEnvironmentResolver)
|
|
resolver.resolve_and_validate.return_value = ExecutionEnvironment.CONTAINER
|
|
|
|
mock_executor = MagicMock(spec=ContainerToolExecutor)
|
|
context.runner = ToolRunner(
|
|
registry=registry,
|
|
execution_environment_resolver=resolver,
|
|
container_executor=mock_executor,
|
|
)
|
|
|
|
|
|
@when("I execute the container tool with missing required fields")
|
|
def step_execute_missing_fields(context: Any) -> None:
|
|
# Only provide file_path but not mode
|
|
context.tool_result = context.runner.execute(
|
|
"ns/strict-tool", {"file_path": "/tmp/file.py"}
|
|
)
|
|
|
|
|
|
@given("I have a ToolRunner with a container executor for JSON validation")
|
|
def step_runner_ctr_json_validation(context: Any) -> None:
|
|
registry = MagicMock(spec=_TR)
|
|
spec = ToolSpec(
|
|
name="ns/json-tool",
|
|
description="Tool for JSON validation test",
|
|
input_schema={},
|
|
handler=lambda inputs: {"ok": True},
|
|
)
|
|
registry.get.return_value = spec
|
|
|
|
resolver = MagicMock(spec=ExecutionEnvironmentResolver)
|
|
resolver.resolve_and_validate.return_value = ExecutionEnvironment.CONTAINER
|
|
|
|
mock_executor = MagicMock(spec=ContainerToolExecutor)
|
|
context.runner = ToolRunner(
|
|
registry=registry,
|
|
execution_environment_resolver=resolver,
|
|
container_executor=mock_executor,
|
|
)
|
|
|
|
|
|
@when("I execute a container tool with non-serialisable inputs")
|
|
def step_execute_ctr_nonserial(context: Any) -> None:
|
|
context.tool_result = context.runner.execute("ns/json-tool", {"bad": object()})
|
|
|
|
|
|
@given("I have a ToolRunner with a failing container executor")
|
|
def step_runner_failing_ctr_executor(context: Any) -> None:
|
|
registry = MagicMock(spec=_TR)
|
|
spec = ToolSpec(
|
|
name="ns/fail-tool",
|
|
description="Tool that triggers executor failure",
|
|
input_schema={},
|
|
handler=lambda inputs: {"ok": True},
|
|
)
|
|
registry.get.return_value = spec
|
|
|
|
resolver = MagicMock(spec=ExecutionEnvironmentResolver)
|
|
resolver.resolve_and_validate.return_value = ExecutionEnvironment.CONTAINER
|
|
|
|
mock_executor = MagicMock(spec=ContainerToolExecutor)
|
|
mock_executor.execute_tool.side_effect = ContainerExecutionError(
|
|
"Container crashed", exit_code=-1
|
|
)
|
|
context.runner = ToolRunner(
|
|
registry=registry,
|
|
execution_environment_resolver=resolver,
|
|
container_executor=mock_executor,
|
|
)
|
|
|
|
|
|
@when("I execute a tool through the failing container runner")
|
|
def step_execute_failing_ctr(context: Any) -> None:
|
|
context.tool_result = context.runner.execute("ns/fail-tool", {})
|
|
|
|
|
|
# =========================================================================
|
|
# ToolResult validation — success/error consistency
|
|
# =========================================================================
|
|
|
|
|
|
@when("I try to create a ToolResult with success true and an error")
|
|
def step_toolresult_success_with_error(context: Any) -> None:
|
|
try:
|
|
ToolResult(success=True, output={}, error="should not be here")
|
|
context.tr_error = None
|
|
except ValueError as exc:
|
|
context.tr_error = exc
|
|
|
|
|
|
@when("I try to create a ToolResult with success false and no error")
|
|
def step_toolresult_fail_without_error(context: Any) -> None:
|
|
try:
|
|
ToolResult(success=False, output={})
|
|
context.tr_error = None
|
|
except ValueError as exc:
|
|
context.tr_error = exc
|
|
|
|
|
|
@then("a ToolResult validation error should be raised")
|
|
def step_check_toolresult_error(context: Any) -> None:
|
|
assert context.tr_error is not None, (
|
|
"Expected ValueError from ToolResult validation"
|
|
)
|
|
|
|
|
|
# =========================================================================
|
|
# ToolError structured attributes
|
|
# =========================================================================
|
|
|
|
|
|
@when("I create a ToolError with known attributes")
|
|
def step_create_tool_error(context: Any) -> None:
|
|
context.tool_error = ToolError(
|
|
tool_name="ns/broken",
|
|
error_type="ExecutionError",
|
|
details="Something went wrong",
|
|
)
|
|
|
|
|
|
@then("the ToolError should expose tool_name, error_type, and details")
|
|
def step_check_tool_error_attrs(context: Any) -> None:
|
|
err = context.tool_error
|
|
assert err.tool_name == "ns/broken"
|
|
assert err.error_type == "ExecutionError"
|
|
assert err.details == "Something went wrong"
|
|
assert "ExecutionError" in str(err)
|
|
assert "ns/broken" in str(err)
|