"""Behave environment setup for feature tests.""" import contextlib import fcntl import logging import os import re import shutil import sys import tempfile from collections.abc import Callable from pathlib import Path from typing import Any, cast from behave.model import Scenario, Status LANGSMITH_ENV_VARS = [ "CLEVERAGENTS_LANGSMITH_ENABLED", "CLEVERAGENTS_LANGSMITH_PROJECT", "CLEVERAGENTS_LANGSMITH_ENDPOINT", "CLEVERAGENTS_LANGSMITH_API_KEY", "CLEVERAGENTS_LANGSMITH_TRACING_V2", "CLEVERAGENTS_LANGSMITH_USER_ID", "LANGCHAIN_TRACING_V2", "LANGCHAIN_PROJECT", "LANGCHAIN_ENDPOINT", "LANGCHAIN_API_KEY", "LANGSMITH_PROJECT", "LANGSMITH_ENDPOINT", "LANGSMITH_API_KEY", "LANGSMITH_TRACING_V2", "LANGSMITH_USER_ID", ] # --------------------------------------------------------------------------- # TDD Issue Test Tags — Three-Tag System # --------------------------------------------------------------------------- # TDD issue-capture tests use a three-tag system documented in # CONTRIBUTING.md > Bug Fix Workflow > TDD Issue Test Tags: # # @tdd_issue — Generic filter tag. Present on ALL TDD issue tests. # @tdd_issue_ — Issue reference (e.g. @tdd_issue_123). Links the # test to the specific Type/Bug issue it captures. # @tdd_expected_fail — Behavioral switch. When present, the test result # is inverted: a failure means the bug still exists # (reported as passed), and a pass means the bug was # fixed without removing the tag (reported as failed). # # The ``validate_tdd_tags`` and ``should_invert_result`` helpers below are # called from the ``before_scenario`` hook and the ``Scenario.run()`` # wrapper (installed in ``before_all``) respectively. They are extracted # as standalone functions so they can be unit-tested directly from Behave # step definitions. ``apply_tdd_inversion`` encapsulates the full # inversion logic and is likewise directly testable. # --------------------------------------------------------------------------- _TDD_ISSUE_N_RE = re.compile(r"tdd_issue_\d+") # --------------------------------------------------------------------------- # Process-global set of already-initialized DB paths # --------------------------------------------------------------------------- # After ``_fast_init_or_upgrade`` has processed a given database URL (either # by copying the template or confirming the file is non-empty), the URL is # added to this set. Subsequent calls with the same URL short-circuit at # the very top of the function, avoiding all URL parsing, path extraction, # prefix matching, and ``stat()`` syscalls. The set is cleared in # ``before_scenario`` to prevent cross-scenario state leaks. # # See issue #735 — eliminates ~65,000 unnecessary function executions per # full test run. # --------------------------------------------------------------------------- _INITIALIZED_DBS: set[str] = set() _tdd_logger = logging.getLogger("cleveragents.testing.tdd_tags") def _warning_with_stderr(message: str) -> None: """Emit a TDD warning to both logger output and stderr. Behave's standard console output captures stderr reliably, while project logging configuration may route warning logs to structured sinks that are not always visible in CI snippets. Emitting to both channels makes non-inversion guard firings obvious during flaky-test diagnosis. """ _tdd_logger.warning(message) print( message, file=sys.stderr ) # Intentional duplication: logger may route to structured sinks not visible in CI; stderr guarantees visibility in Behave output. def validate_tdd_tags(tags: set[str]) -> None: """Validate TDD issue-capture tag combinations. Raises ``ValueError`` with a descriptive message when the tag set is inconsistent according to the rules in CONTRIBUTING.md: * ``@tdd_issue_`` requires ``@tdd_issue`` to also be present. * ``@tdd_expected_fail`` requires both ``@tdd_issue`` and at least one ``@tdd_issue_``. Args: tags: The *effective* tags of a scenario (own tags + feature tags). Raises: ValueError: If the tag combination violates the TDD tag rules. """ has_tdd_issue = "tdd_issue" in tags has_tdd_issue_n = any(_TDD_ISSUE_N_RE.fullmatch(t) for t in tags) has_expected_fail = "tdd_expected_fail" in tags if has_tdd_issue_n and not has_tdd_issue: issue_n_tags = sorted(t for t in tags if _TDD_ISSUE_N_RE.fullmatch(t)) raise ValueError( f"Scenario has {', '.join('@' + t for t in issue_n_tags)} but is " f"missing the required @tdd_issue tag. All TDD issue tests must " f"include @tdd_issue. See CONTRIBUTING.md > TDD Issue Test Tags." ) if has_expected_fail: missing: list[str] = [] if not has_tdd_issue: missing.append("@tdd_issue") if not has_tdd_issue_n: missing.append("@tdd_issue_") if missing: raise ValueError( f"Scenario has @tdd_expected_fail but is missing required " f"tag(s): {', '.join(missing)}. @tdd_expected_fail requires " f"both @tdd_issue and at least one @tdd_issue_. " f"See CONTRIBUTING.md > TDD Issue Test Tags." ) def should_invert_result(tags: set[str]) -> bool: """Return ``True`` if the scenario result should be inverted. A scenario result is inverted when ``@tdd_expected_fail`` is present in the effective tags, indicating the test captures a known bug that has not yet been fixed. Args: tags: The *effective* tags of a scenario (own tags + feature tags). """ return "tdd_expected_fail" in tags _UNEXPECTED_PASS_MSG = ( "Bug appears to be fixed. Remove the @tdd_expected_fail tag " "from this scenario and verify the fix through the bug fix " "workflow. See CONTRIBUTING.md > Bug Fix Workflow." ) def apply_tdd_inversion(scenario: Any, failed: bool) -> bool: """Apply TDD expected-fail result inversion to a completed scenario. Encapsulates the full inversion logic so it can be unit-tested directly from Behave step definitions (using mock scenario objects). Guards against infrastructure errors and dry-run mode before inverting: * **Hook/cleanup errors** (``scenario.hook_failed``): never inverted — infrastructure failures must propagate regardless of tags. * **Dry-run mode** (``scenario.was_dry_run``): never inverted — no test actually executed, so the result is meaningless. * **Non-assertion exceptions**: a step that failed with an exception other than ``AssertionError`` likely indicates an infrastructure problem, not the captured bug — a warning is logged and the result is *not* inverted. When inverting expected failures, ``step.error_message`` is also cleared alongside ``step.exception`` and ``step.exc_traceback`` to prevent stale failure text from leaking to JUnit XML reporters and custom formatters. Args: scenario: A Behave ``Scenario`` (or compatible mock) with at least ``effective_tags``, ``all_steps``, ``name``, ``hook_failed``, ``was_dry_run``, ``clear_status()``, and ``set_status()`` attributes. failed: The boolean return value of the original ``Scenario.run()``. Returns: The (possibly inverted) failure status to be returned to the runner. """ if not should_invert_result(set(scenario.effective_tags)): return failed # Guard: never invert infrastructure / hook errors. if getattr(scenario, "hook_failed", False): return failed # Guard: no-op during dry-run — no test actually executed. if getattr(scenario, "was_dry_run", False): return failed # Materialise all_steps once — some Behave versions return an # iterator (list_iterator / itertools.chain) instead of a list. all_steps = list(scenario.all_steps) if failed: # Guard: do not invert non-assertion exceptions — they likely # indicate an infrastructure problem, not the captured bug. for step in all_steps: if ( step.status == Status.failed and step.exception is not None and not isinstance(step.exception, AssertionError) ): # Intentional f-string (eager evaluation) rather than # lazy %-style: the message is always emitted to stderr via # print(), so deferred formatting buys nothing here. exc_text = str( step.exception )[ :500 ] # Defensive truncation — avoids runaway output for exceptions with very long str() representations. _warning_with_stderr( "Non-assertion exception in expected-fail scenario " f"'{scenario.name}' step '{step.name}': " f"{exc_text} — not inverting." ) return failed # Expected failure — the bug still exists. Reset the failed # and skipped steps so the scenario is reported as passed. # (When a step fails, subsequent steps are marked as skipped; # both must be set to ``Status.passed`` for accurate summary # counts and consistent ``compute_status()`` behaviour.) for step in all_steps: if step.status in (Status.failed, Status.skipped): _tdd_logger.debug( "Clearing expected-fail exception for step '%s': %s", step.name, step.exception, ) step.status = Status.passed step.exception = None step.exc_traceback = None step.error_message = None scenario.clear_status() scenario.set_status(Status.passed) return False # Not a failure for the runner # Unexpected pass — the bug appears to be fixed but the # @tdd_expected_fail tag has not been removed. Force a failure # so CI blocks the PR until the tag is cleaned up. _tdd_logger.warning( "Bug appears to be fixed. Remove the @tdd_expected_fail " "tag from scenario '%s' and verify the fix through the " "bug fix workflow. See CONTRIBUTING.md > Bug Fix Workflow.", scenario.name, ) # Attach a synthetic error to the last step so the failure reason # appears in standard Behave output (formatters show the failed # step's exception text). if all_steps: last_step = all_steps[-1] last_step.status = Status.failed last_step.exception = AssertionError(_UNEXPECTED_PASS_MSG) last_step.error_message = "Assertion Failed: " + _UNEXPECTED_PASS_MSG scenario.set_status(Status.failed) return True # Force failure for the runner def handle_tdd_expected_fail(scenario: Any) -> None: """Process a scenario through TDD expected-fail tag validation and inversion. Public, testable entry point that encapsulates the full TDD expected-fail logic operating directly on a ``Scenario`` object: 1. **Tag validation** — rejects invalid TDD tag combinations by forcing the scenario to ``Status.failed``. 2. **Result inversion** — delegates to :func:`apply_tdd_inversion` for valid ``@tdd_expected_fail`` scenarios, which inverts the result: a failure becomes passed (expected), a pass becomes failed (unexpected — the bug appears fixed). All guard logic (hook errors, dry-run, non-assertion exceptions) and step-level status manipulation are handled by :func:`apply_tdd_inversion`. Args: scenario: A Behave ``Scenario`` (or compatible mock) with at least ``effective_tags`` (or ``tags``), ``all_steps``, ``status``, ``hook_failed``, ``was_dry_run``, ``clear_status()``, and ``set_status()``. """ tags = set(getattr(scenario, "effective_tags", getattr(scenario, "tags", []))) # Validate tags — force failure on invalid combinations. try: validate_tdd_tags(tags) except ValueError as exc: _tdd_logger.warning( "Invalid TDD tag combination on scenario '%s': %s", scenario.name, exc, ) scenario.set_status(Status.failed) return # Delegate inversion logic to the shared implementation. apply_tdd_inversion(scenario, scenario.status == Status.failed) def _install_tdd_expected_fail_patch() -> None: """Monkey-patch ``Scenario.run`` to invert results for ``@tdd_expected_fail``. Behave's ``Scenario.run()`` returns a local ``failed`` boolean that ``after_scenario`` hooks cannot modify. To correctly invert the result (so that an expected-fail scenario is reported as passed to the runner), we wrap ``Scenario.run()`` with a thin post-processing layer that delegates to :func:`apply_tdd_inversion`. The patch is installed once in ``before_all`` and is idempotent. """ if getattr(Scenario, "_tdd_run_patched", False): return # Already patched (e.g. forked worker reloading hooks) _original_run = Scenario.run def _tdd_aware_run(self: Any, runner: Any) -> bool: failed: bool = _original_run(self, runner) return apply_tdd_inversion(self, failed) Scenario.run = _tdd_aware_run Scenario._tdd_run_patched = True def before_all(context): """Set up test environment before all tests.""" # Add src to path sys.path.insert(0, str(Path(__file__).parent.parent / "src")) # Ensure tests never block on migration prompts or real providers os.environ.setdefault("CLEVERAGENTS_AUTO_APPLY_MIGRATIONS", "true") os.environ.setdefault("CLEVERAGENTS_TESTING_USE_MOCK_AI", "true") # Use per-process unique database paths so parallel test subprocesses # (behave-parallel) never contend on the same SQLite file. if "CLEVERAGENTS_DATABASE_URL" not in os.environ: _fd, _db_path = tempfile.mkstemp(suffix=".db", prefix="cleveragents_") os.close(_fd) os.environ["CLEVERAGENTS_DATABASE_URL"] = f"sqlite:///{_db_path}" if "CLEVERAGENTS_TEST_DATABASE_URL" not in os.environ: _fd, _db_path = tempfile.mkstemp(suffix=".db", prefix="cleveragents_test_") os.close(_fd) os.environ["CLEVERAGENTS_TEST_DATABASE_URL"] = f"sqlite:///{_db_path}" os.environ.setdefault("BEHAVE_TESTING", "true") # Set up mock AI provider for all tests try: from cleveragents.application.container import override_providers from features.mocks.mock_ai_provider import MockAIProvider # Override the AI provider with mock for all tests mock_provider = MockAIProvider() override_providers(ai_provider=mock_provider) except ImportError: pass # Container not needed for all tests # --- Eliminate retry waits --- # Tenacity retry decorators (database_retry, network_retry, etc.) use # real time.sleep() waits during retries. In tests, mocked operations # fail deterministically so waiting is pure overhead. Patch # time.sleep() globally so all tenacity waits (and any other sleeps) # complete instantly. The small handful of sleep() calls that exist # in step definitions already use sub-100ms waits and are unaffected # by this optimisation in practice. _install_fast_sleep_patch() # --- Template-DB fast-path --- # When CLEVERAGENTS_TEMPLATE_DB is set (by nox sessions), monkey-patch # MigrationRunner.init_or_upgrade so that fresh file-based SQLite # databases are created by copying the pre-migrated template (~1ms) # instead of running 25 Alembic migrations (~0.5-3s each). # # If running outside of nox (i.e. direct `behave` invocation), auto- # create the template DB so the fast-path is always active. _ensure_template_db() _install_template_db_patch() # --- TDD Expected-Fail Patch --- # Wrap Scenario.run() so @tdd_expected_fail inverts the result for # both the scenario status AND the runner's pass/fail return value. _install_tdd_expected_fail_patch() def _install_fast_sleep_patch() -> None: """Cap ``time.sleep`` and ``asyncio.sleep`` at 10 ms for fast test execution. Tenacity retry decorators (``@database_retry``, ``@retry_network_operation``, etc.) ultimately call ``time.sleep()`` with waits of 0.5-30 s between retry attempts. Async retry helpers (``retry_auto_debug``, ``async_retry_with_exponential_backoff``) call ``asyncio.sleep()`` with exponential waits of 1-4 s per attempt. In the test suite, mocked operations fail deterministically, so the long sleeps are pure overhead (~1 s per retry cycle x hundreds of scenarios = minutes of wasted time). Both functions are replaced with capped versions (<=10 ms). The originals are stored on the modules as ``_original_sleep`` and can be retrieved with ``getattr(time, "_original_sleep", time.sleep)`` by any test step that needs a genuine delay (e.g. CircuitBreaker recovery-timeout tests that need real wall-clock advancement past a 100 ms threshold). ``cast(Any, module)`` is used to assign dynamic attributes without ``# type: ignore`` suppressions: the ``features/`` directory is excluded from Pyright's ``include`` list, so the cast is a documentation aid rather than a runtime necessity. """ import asyncio import time _MAX_SLEEP = 0.01 # 10 ms cap # --- synchronous time.sleep --- # Store the original in a typed local variable so the inner closure can # call it directly. cast(Any, time) lets us assign _original_sleep and # replace sleep without attr-defined / assignment type errors. if not callable(getattr(time, "_original_sleep", None)): _original_time_sleep: Callable[[float], None] = time.sleep _time_mod: Any = cast(Any, time) _time_mod._original_sleep = _original_time_sleep def _capped_sleep(seconds: float) -> None: _original_time_sleep(min(seconds, _MAX_SLEEP)) _time_mod.sleep = _capped_sleep # --- asynchronous asyncio.sleep --- # Same pattern: capture the original in a typed local variable, then use # cast(Any, asyncio) to assign _original_sleep and replace sleep. if not callable(getattr(asyncio, "_original_sleep", None)): _original_asyncio_sleep = asyncio.sleep _asyncio_mod: Any = cast(Any, asyncio) _asyncio_mod._original_sleep = _original_asyncio_sleep async def _capped_async_sleep(seconds: float, result: object = None) -> object: return await _original_asyncio_sleep(min(seconds, _MAX_SLEEP), result) _asyncio_mod.sleep = _capped_async_sleep def _ensure_template_db() -> None: """Auto-create the template DB when CLEVERAGENTS_TEMPLATE_DB is not set. When running tests directly via ``behave`` (without nox), the env var is missing. This function creates the template on the fly using ``scripts/create_template_db.py`` so the fast-path is always active. """ if os.environ.get("CLEVERAGENTS_TEMPLATE_DB"): return # Already set by nox or CI template_path = Path(__file__).parent.parent / "build" / ".template-migrated.db" if template_path.is_file(): # Template already exists from a prior run — reuse it. os.environ["CLEVERAGENTS_TEMPLATE_DB"] = str(template_path.resolve()) return lock_path = template_path.with_suffix(".db.lock") _lock_fd = -1 try: _lock_fd = os.open(str(lock_path), os.O_CREAT | os.O_RDWR) fcntl.flock(_lock_fd, fcntl.LOCK_EX) if template_path.is_file(): os.environ["CLEVERAGENTS_TEMPLATE_DB"] = str(template_path.resolve()) return scripts_dir = Path(__file__).parent.parent / "scripts" sys.path.insert(0, str(scripts_dir)) from create_template_db import create_template create_template(str(template_path)) os.environ["CLEVERAGENTS_TEMPLATE_DB"] = str(template_path.resolve()) except Exception: pass # Fall back to normal Alembic migrations finally: if _lock_fd >= 0: fcntl.flock(_lock_fd, fcntl.LOCK_UN) os.close(_lock_fd) def _install_template_db_patch() -> None: """Monkey-patch MigrationRunner to skip Alembic migrations in tests. For file-based SQLite: copies a pre-migrated template DB (~1 ms). For in-memory SQLite: uses ``Base.metadata.create_all()`` (~5 ms) instead of running 25 sequential Alembic migrations (~0.5-3 s). """ template_path = os.environ.get("CLEVERAGENTS_TEMPLATE_DB") if not template_path or not Path(template_path).is_file(): return try: from cleveragents.infrastructure.database.migration_runner import ( MigrationRunner, ) except ImportError: return _original_init_or_upgrade = MigrationRunner.init_or_upgrade # Prefixes used by before_scenario and step files when creating temp DBs. # "cleveragents_" / "cleveragents_test_" — before_scenario databases # "test_" — databases created inside step files (services_coverage, etc.) _SCENARIO_DB_PREFIXES = ("cleveragents_", "cleveragents_test_", "test_", "db.") def _fast_init_or_upgrade(self: Any, **kwargs: Any) -> None: """Replace Alembic migrations with fast alternatives. - Process-global cache hit -> immediate return (no work at all) - Non-SQLite databases -> fall through to original - In-memory SQLite -> ``Base.metadata.create_all()`` + alembic stamp - File-based SQLite with matching prefix -> copy template - Everything else -> fall through to original """ db_url: str = getattr(self, "database_url", "") # Process-global fast-path: if we have already initialised this # exact URL in the current scenario, skip all work. Eliminates # ~65k redundant function-body executions per full test run. if db_url in _INITIALIZED_DBS: return # Non-SQLite: always fall through if not db_url.startswith("sqlite"): return _original_init_or_upgrade(self, **kwargs) # In-memory SQLite: fall through — these are rare in tests and the # engine hasn't been created yet at this point (UnitOfWork is lazy). if ":memory:" in db_url or db_url == "sqlite://": return _original_init_or_upgrade(self, **kwargs) # Extract the file path from the URL db_file_path = db_url.replace("sqlite:///", "") if not db_file_path.startswith("/"): db_file_path = "/" + db_file_path db_path = Path(db_file_path) # Only apply to scenario-generated temp DBs (avoid hijacking # migration-runner unit tests that use custom URLs). if not any(db_path.name.startswith(p) for p in _SCENARIO_DB_PREFIXES): return _original_init_or_upgrade(self, **kwargs) # Only copy template for databases that don't exist yet or are empty # (SQLite auto-creates a 0-byte file on first engine open). # If the DB already exists and is non-empty, it was either copied from # the template or created by a prior step — either way it is already # fully migrated. Skipping the Alembic check avoids redundant engine # creation, SQLite lock contention, and the cumulative overhead that # causes intermittent hangs in parallel test runs. if db_path.exists() and db_path.stat().st_size > 0: _INITIALIZED_DBS.add(db_url) return # Copy the template — creates a fully-migrated DB in ~1ms db_path.parent.mkdir(parents=True, exist_ok=True) shutil.copy2(template_path, db_file_path) _INITIALIZED_DBS.add(db_url) MigrationRunner.init_or_upgrade = _fast_init_or_upgrade # type: ignore[assignment] def before_scenario(context, scenario): """Set up before each scenario.""" # --- TDD Issue Tag Validation --- # Validate the three-tag system BEFORE any other setup so that # misconfigured TDD tests are caught immediately. # See CONTRIBUTING.md > TDD Issue Test Tags for the full specification. try: validate_tdd_tags(set(scenario.effective_tags)) except ValueError as exc: scenario.hook_failed = True scenario.set_status(Status.failed) _tdd_logger.error("TDD TAG ERROR in %r: %s", scenario.name, exc) return # Clear the process-global set of already-initialised DB paths so # that stale entries from the previous scenario cannot leak into this # one. Each scenario receives fresh temp-DB paths, so old entries are # irrelevant and would only waste memory. _INITIALIZED_DBS.clear() # Store original working directory context.original_cwd = os.getcwd() # Initialize cleanup list context._cleanup_handlers = [] context.stubbed_clients = {} # Reset error attributes to prevent stale state leaking between scenarios. context.acms_error = None context.assemble_error = None # --- Env-var save/restore for provider registry --- # Proactively save env vars so they are restored even if a scenario fails # before reaching the step that normally records them. for attr, env_var in [ ("original_provider_env", "CLEVERAGENTS_DEFAULT_PROVIDER"), ("original_model_env", "CLEVERAGENTS_DEFAULT_MODEL"), ("original_allow_mock_env", "CLEVERAGENTS_ALLOW_MOCK_PROVIDER"), ]: setattr(context, attr, os.environ.get(env_var)) # Ensure mock AI flag is always set so plan service tests can resolve actors os.environ["CLEVERAGENTS_TESTING_USE_MOCK_AI"] = "true" # Re-establish protective env vars every scenario. These are set once # in before_all, but individual scenarios (e.g. migration_runner tests) # temporarily remove them. If any cleanup path fails to restore them # the remaining scenarios in a serial run would be affected. os.environ["BEHAVE_TESTING"] = "true" os.environ["CLEVERAGENTS_AUTO_APPLY_MIGRATIONS"] = "true" # Flush the in-memory engine cache so no stale engines leak between # scenarios. In serial mode (coverage_report) every feature shares # the same process, so a previous scenario's real-or-fake engine can # survive into the next one. try: from cleveragents.infrastructure.database.engine_cache import MEMORY_ENGINES for _url, engine in list(MEMORY_ENGINES.items()): with contextlib.suppress(Exception): engine.dispose() MEMORY_ENGINES.clear() except ImportError: pass # Clean up any lingering test environment variables from previous tests for env_var in [ "CLEVERAGENTS_MOCK_SHOULD_FAIL", "CLEVERAGENTS_MOCK_INVALID_CODE", "CLEVERAGENTS_SERVER_MODE", *LANGSMITH_ENV_VARS, ]: if env_var in os.environ: del os.environ[env_var] # Give each scenario a unique database file so scenarios cannot share # persisted state AND parallel subprocesses never collide on the same # SQLite file. Store the paths for cleanup in after_scenario. # # Features tagged @mock_only use fully mocked services and never touch # the database, so skip the temp-file creation for them (~0.5ms each, # but the real savings come from not triggering MigrationRunner later). context._scenario_db_paths = [] _is_mock_only = "mock_only" in scenario.effective_tags if not _is_mock_only: for env_var, prefix in ( ("CLEVERAGENTS_DATABASE_URL", "cleveragents_"), ("CLEVERAGENTS_TEST_DATABASE_URL", "cleveragents_test_"), ): _fd, db_path = tempfile.mkstemp(suffix=".db", prefix=prefix) os.close(_fd) os.environ[env_var] = f"sqlite:///{db_path}" context._scenario_db_paths.append(db_path) # Clear devcontainer lifecycle registry between scenarios to prevent # test pollution from in-memory lifecycle trackers and health check # threads left by previous scenarios. try: from cleveragents.resource.handlers.devcontainer import ( clear_lifecycle_registry, ) clear_lifecycle_registry() except ImportError: pass # Handler not available in all environments # Re-apply mock AI provider after container reset try: from cleveragents.application.container import override_providers from features.mocks.mock_ai_provider import MockAIProvider # Override the AI provider with mock for all tests mock_provider = MockAIProvider() override_providers(ai_provider=mock_provider) except ImportError: pass # Container not needed for all tests def after_scenario(context, scenario): """Clean up after each scenario.""" # Return to original directory first if hasattr(context, "original_cwd"): os.chdir(context.original_cwd) # Run all cleanup handlers in reverse order if hasattr(context, "_cleanup_handlers"): for handler in reversed(context._cleanup_handlers): with contextlib.suppress(Exception): handler() context._cleanup_handlers = [] # Clean up any test directories if hasattr(context, "test_dir") and context.test_dir: try: if Path(context.test_dir).exists(): shutil.rmtree(context.test_dir) except Exception: pass # Ignore cleanup errors context.test_dir = None # Clean up TemporaryDirectory objects created by ACMS index traversal tests if hasattr(context, "temp_dir") and context.temp_dir is not None: with contextlib.suppress(Exception): context.temp_dir.cleanup() context.temp_dir = None # Clean up environment variables set during tests if hasattr(context, "env_vars_to_clean"): for key in context.env_vars_to_clean: os.environ.pop(key, None) context.env_vars_to_clean = [] for env_var in [ *LANGSMITH_ENV_VARS, "CLEVERAGENTS_DATABASE_URL", "CLEVERAGENTS_TEST_DATABASE_URL", ]: os.environ.pop(env_var, None) # Reset Settings singleton if it was used try: from cleveragents.config.settings import Settings Settings._instance = None except ImportError: pass # Reset global container singleton try: from cleveragents.application.container import reset_container reset_container() except ImportError: pass # Reset A2A facade singleton to prevent state leaking between scenarios. try: from cleveragents.a2a.cli_bootstrap import reset_facade reset_facade() except ImportError: pass # Reset session CLI module-level service singleton so that no stale # service instance leaks into the next scenario. Without this reset, # a scenario that sets _service to a real container-constructed instance # would leave it cached; a subsequent scenario that patches _service # via unittest.mock.patch would restore to the stale real instance on # cleanup, causing the next scenario to hit the real container and fail. try: from cleveragents.cli.commands.session import _reset_session_service _reset_session_service() except ImportError: pass # Clean up service instances for attr in ["context_service", "plan_service", "project_service"]: if hasattr(context, attr): delattr(context, attr) # Clean up any remaining attributes that might hold state for attr in ["plan", "plans", "project", "changes", "added_files", "all_plans"]: if hasattr(context, attr): delattr(context, attr) # Restore provider registry env overrides if used in a scenario for attr, env_var in [ ("original_provider_env", "CLEVERAGENTS_DEFAULT_PROVIDER"), ("original_model_env", "CLEVERAGENTS_DEFAULT_MODEL"), ("original_allow_mock_env", "CLEVERAGENTS_ALLOW_MOCK_PROVIDER"), ]: if hasattr(context, attr): original = getattr(context, attr) if original is None: os.environ.pop(env_var, None) else: os.environ[env_var] = original delattr(context, attr) # Reset provider registry singleton between scenarios try: from cleveragents.providers.registry import ( reset_provider_registry as _reset_registry, ) _reset_registry() except ImportError: pass # Clean up in-memory database engine cache to ensure each scenario # gets a fresh database. This is critical for tests using :memory: databases try: from cleveragents.infrastructure.database.engine_cache import MEMORY_ENGINES # Dispose of all cached engines and clear the cache for _url, engine in list(MEMORY_ENGINES.items()): with contextlib.suppress(Exception): engine.dispose() MEMORY_ENGINES.clear() except ImportError: pass # Remove per-scenario temp database files (and associated journal/WAL # files) now that all engines have been disposed. for db_path in getattr(context, "_scenario_db_paths", []): for suffix in ("", "-journal", "-wal", "-shm"): with contextlib.suppress(OSError): os.unlink(db_path + suffix) # T6: Remove log handlers attached to the async-cleanup logger by # security_async_steps.py so handlers don't accumulate across scenarios. if hasattr(context, "log_handler"): async_logger = logging.getLogger("cleveragents.core.async_cleanup") async_logger.removeHandler(context.log_handler) # T5: Close event loops left open by security_async_steps.py. if hasattr(context, "bridge_loop"): with contextlib.suppress(Exception): context.bridge_loop.close() # NOTE: TDD @tdd_expected_fail result inversion is handled by the # Scenario.run() wrapper installed in _install_tdd_expected_fail_patch(), # NOT in this hook. See before_all() and CONTRIBUTING.md > TDD Issue # Test Tags for the full specification. # Clean up test UnitOfWork fixtures if any were created try: from features.mocks.test_uow_factory import cleanup_test_uow cleanup_test_uow(context) except ImportError: pass