diff --git a/.forgejo/workflows/ci.yml b/.forgejo/workflows/ci.yml index 5b04bd9c6..00b42d1f3 100644 --- a/.forgejo/workflows/ci.yml +++ b/.forgejo/workflows/ci.yml @@ -6,9 +6,6 @@ on: pull_request: branches: [master, develop*] -vars: - docker_prefix: "http://harbor.cleverthis.com/docker/" - env: UV_VERSION: "0.8.0" PYTHON_VERSION: "3.13" @@ -18,7 +15,7 @@ jobs: lint: runs-on: docker container: - image: ${{vars.docker_prefix}}python:3.13-slim + image: python:3.13-slim steps: - name: Install Node.js (required by actions/checkout) run: | @@ -34,35 +31,26 @@ jobs: uses: actions/cache@v3 with: path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} + key: uv-lint-${{ hashFiles('pyproject.toml') }} restore-keys: | - uv- + uv-lint- - name: Run lint via nox run: | - mkdir -p build - nox -s lint 2>&1 | tee build/nox-lint-output.log + nox -s lint env: NOX_DEFAULT_VENV_BACKEND: uv - name: Run format check via nox run: | - nox -s format -- --check 2>&1 | tee -a build/nox-lint-output.log + nox -s format -- --check env: NOX_DEFAULT_VENV_BACKEND: uv - - name: Upload lint log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-lint - path: build/nox-lint-output.log - retention-days: 30 - typecheck: runs-on: docker container: - image: ${{vars.docker_prefix}}python:3.13-slim + image: python:3.13-slim steps: - name: Install Node.js (required by actions/checkout) run: | @@ -78,29 +66,20 @@ jobs: uses: actions/cache@v3 with: path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} + key: uv-typecheck-${{ hashFiles('pyproject.toml') }} restore-keys: | - uv- + uv-typecheck- - name: Run typecheck via nox run: | - mkdir -p build - nox -s typecheck 2>&1 | tee build/nox-typecheck-output.log + nox -s typecheck env: NOX_DEFAULT_VENV_BACKEND: uv - - name: Upload typecheck log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-typecheck - path: build/nox-typecheck-output.log - retention-days: 30 - security: runs-on: docker container: - image: ${{vars.docker_prefix}}python:3.13-slim + image: python:3.13-slim steps: - name: Install Node.js (required by actions/checkout) run: | @@ -116,35 +95,26 @@ jobs: uses: actions/cache@v3 with: path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} + key: uv-security-${{ hashFiles('pyproject.toml') }} restore-keys: | - uv- + uv-security- - name: Run security scan via nox run: | - mkdir -p build - nox -s security_scan 2>&1 | tee build/nox-security-output.log + nox -s security_scan env: NOX_DEFAULT_VENV_BACKEND: uv - name: Run dead code detection via nox run: | - nox -s dead_code 2>&1 | tee -a build/nox-security-output.log + nox -s dead_code env: NOX_DEFAULT_VENV_BACKEND: uv - - name: Upload security log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-security - path: build/nox-security-output.log - retention-days: 30 - quality: runs-on: docker container: - image: ${{vars.docker_prefix}}python:3.13-slim + image: python:3.13-slim steps: - name: Install Node.js (required by actions/checkout) run: | @@ -160,185 +130,20 @@ jobs: uses: actions/cache@v3 with: path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} + key: uv-quality-${{ hashFiles('pyproject.toml') }} restore-keys: | - uv- + uv-quality- - name: Run complexity check via nox run: | - mkdir -p build - nox -s complexity 2>&1 | tee build/nox-quality-output.log + nox -s complexity env: NOX_DEFAULT_VENV_BACKEND: uv - - name: Upload quality log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-quality - path: build/nox-quality-output.log - retention-days: 30 - unit_tests: runs-on: docker container: - image: ${{vars.docker_prefix}}python:3.13-slim - steps: - - name: Install system dependencies (nodejs for checkout, git for merge tests, curl/tar for Helm) - run: | - apt-get update && apt-get install -y -qq nodejs git curl tar && rm -rf /var/lib/apt/lists/* - - - name: Install Helm CLI - run: | - HELM_VERSION="v3.16.4" - ARCH="amd64" - HELM_TARBALL="helm-${HELM_VERSION}-linux-${ARCH}.tar.gz" - curl -fsSL "https://get.helm.sh/${HELM_TARBALL}" -o "/tmp/${HELM_TARBALL}" - curl -fsSL "https://get.helm.sh/${HELM_TARBALL}.sha256sum" -o /tmp/helm.sha256sum - cd /tmp && sha256sum -c helm.sha256sum - tar -xzf "/tmp/${HELM_TARBALL}" -C /tmp - mv /tmp/linux-${ARCH}/helm /usr/local/bin/helm - chmod +x /usr/local/bin/helm - helm version --short - - - uses: actions/checkout@v4 - - - name: Install uv and nox - run: | - pip install -q uv==${{ env.UV_VERSION }} nox - - - name: Cache uv packages - uses: actions/cache@v3 - with: - path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} - restore-keys: | - uv- - - - name: Run unit tests via nox - run: | - mkdir -p build - nox -s unit_tests 2>&1 | tee build/nox-unit-tests-output.log - env: - NOX_DEFAULT_VENV_BACKEND: uv - - - name: Upload unit tests log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-unit-tests - path: build/nox-unit-tests-output.log - retention-days: 30 - - integration_tests: - runs-on: docker - container: - image: ${{vars.docker_prefix}}python:3.13-slim - steps: - - name: Install system dependencies (nodejs for checkout, git for integration tests, curl/tar for Helm) - run: | - apt-get update && apt-get install -y -qq nodejs git curl tar && rm -rf /var/lib/apt/lists/* - - - name: Install Helm CLI - run: | - HELM_VERSION="v3.16.4" - ARCH="amd64" - HELM_TARBALL="helm-${HELM_VERSION}-linux-${ARCH}.tar.gz" - curl -fsSL "https://get.helm.sh/${HELM_TARBALL}" -o "/tmp/${HELM_TARBALL}" - curl -fsSL "https://get.helm.sh/${HELM_TARBALL}.sha256sum" -o /tmp/helm.sha256sum - cd /tmp && sha256sum -c helm.sha256sum - tar -xzf "/tmp/${HELM_TARBALL}" -C /tmp - mv /tmp/linux-${ARCH}/helm /usr/local/bin/helm - chmod +x /usr/local/bin/helm - helm version --short - - - uses: actions/checkout@v4 - - - name: Install uv and nox - run: | - pip install -q uv==${{ env.UV_VERSION }} nox - - - name: Cache uv packages - uses: actions/cache@v3 - with: - path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} - restore-keys: | - uv- - - - name: Run integration tests via nox - run: | - mkdir -p build - nox -s integration_tests 2>&1 | tee build/nox-integration-tests-output.log - env: - NOX_DEFAULT_VENV_BACKEND: uv - CLEVERAGENTS_REQUIRE_HELM_RENDER_ASSERTIONS: "true" - # LLM API keys required for Robot Framework integration tests. - # These secrets must be configured in Forgejo UI: - # Repository Settings > Actions > Secrets - # See docs/development/ci-cd.md for details. - ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} - - - name: Upload integration tests log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-integration-tests - path: build/nox-integration-tests-output.log - retention-days: 30 - - e2e_tests: - runs-on: docker - timeout-minutes: 45 - container: - image: ${{vars.docker_prefix}}python:3.13-slim - steps: - - name: Install system dependencies (nodejs for checkout, git for E2E tests) - run: | - apt-get update && apt-get install -y -qq nodejs git && rm -rf /var/lib/apt/lists/* - - - uses: actions/checkout@v4 - - - name: Install uv and nox - run: | - pip install -q uv==${{ env.UV_VERSION }} nox - - - name: Cache uv packages - uses: actions/cache@v3 - with: - path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} - restore-keys: | - uv- - - - name: Run E2E tests via nox - run: | - mkdir -p build - nox -s e2e_tests 2>&1 | tee build/nox-e2e-tests-output.log - env: - NOX_DEFAULT_VENV_BACKEND: uv - # Run E2E suites in parallel via pabot. 4 workers keeps - # wall-clock time well under the 45-minute timeout while - # staying within the memory budget of the docker runner. - TEST_PROCESSES: "4" - ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} - OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} - GOOGLE_API_KEY: ${{ secrets.GOOGLE_API_KEY }} - - - name: Upload E2E tests log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-e2e-tests - path: build/nox-e2e-tests-output.log - retention-days: 30 - - coverage: - runs-on: docker - container: - image: ${{vars.docker_prefix}}python:3.13-slim - needs: [lint, typecheck, security, quality] + image: python:3.13-slim steps: - name: Install system dependencies (nodejs for checkout, git for merge tests) run: | @@ -354,17 +159,152 @@ jobs: uses: actions/cache@v3 with: path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} + key: uv-tests-${{ hashFiles('pyproject.toml') }} restore-keys: | - uv- + uv-tests- + + - name: Run unit tests via nox + run: | + nox -s unit_tests + env: + NOX_DEFAULT_VENV_BACKEND: uv + + integration_tests: + runs-on: docker + container: + image: python:3.13-slim + steps: + - name: Install system dependencies (nodejs for checkout, git for integration tests) + run: | + apt-get update && apt-get install -y -qq nodejs git && rm -rf /var/lib/apt/lists/* + + - uses: actions/checkout@v4 + + - name: Install uv and nox + run: | + pip install -q uv==${{ env.UV_VERSION }} nox + + - name: Cache uv packages + uses: actions/cache@v3 + with: + path: ~/.cache/uv + key: uv-tests-${{ hashFiles('pyproject.toml') }} + restore-keys: | + uv-tests- + + - name: Run integration tests via nox + run: | + nox -s integration_tests + env: + NOX_DEFAULT_VENV_BACKEND: uv + # LLM API keys required for Robot Framework integration tests. + # These secrets must be configured in Forgejo UI: + # Repository Settings > Actions > Secrets + # See docs/development/ci-cd.md for details. + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + + tdd_quality_gate: + if: forgejo.event_name == 'pull_request' + runs-on: docker + container: + image: python:3.13-slim + steps: + - name: Install system dependencies (nodejs for checkout, git for diff analysis) + run: | + apt-get update && apt-get install -y -qq nodejs git && rm -rf /var/lib/apt/lists/* + + - uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Fetch PR base branch for diff analysis + run: | + git fetch origin "${{ forgejo.base_ref }}" + + - name: Install uv and nox + run: | + pip install -q uv==${{ env.UV_VERSION }} nox + + - name: Cache uv packages + uses: actions/cache@v3 + with: + path: ~/.cache/uv + key: uv-tdd-gate-${{ hashFiles('pyproject.toml') }} + restore-keys: | + uv-tdd-gate- + + - name: Run TDD quality gate via nox + run: | + nox -s tdd_quality_gate + env: + NOX_DEFAULT_VENV_BACKEND: uv + PR_DESCRIPTION: ${{ forgejo.event.pull_request.body }} + PR_BASE_REF: ${{ forgejo.base_ref }} + + e2e_tests: + runs-on: docker + container: + image: python:3.13-slim + steps: + - name: Install system dependencies (nodejs for checkout, git for E2E tests) + run: | + apt-get update && apt-get install -y -qq nodejs git && rm -rf /var/lib/apt/lists/* + + - uses: actions/checkout@v4 + + - name: Install uv and nox + run: | + pip install -q uv==${{ env.UV_VERSION }} nox + + - name: Cache uv packages + uses: actions/cache@v3 + with: + path: ~/.cache/uv + key: uv-tests-${{ hashFiles('pyproject.toml') }} + restore-keys: | + uv-tests- + + - name: Run E2E tests via nox + run: | + nox -s e2e_tests + env: + NOX_DEFAULT_VENV_BACKEND: uv + ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }} + OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }} + GOOGLE_API_KEY: ${{ secrets.GOOGLE_API_KEY }} + + coverage: + runs-on: docker + container: + image: python:3.13-slim + needs: [lint, typecheck] + steps: + - name: Install system dependencies (nodejs for checkout, git for merge tests) + run: | + apt-get update && apt-get install -y -qq nodejs git && rm -rf /var/lib/apt/lists/* + + - uses: actions/checkout@v4 + + - name: Install uv and nox + run: | + pip install -q uv==${{ env.UV_VERSION }} nox + + - name: Cache uv packages + uses: actions/cache@v3 + with: + path: ~/.cache/uv + key: uv-coverage-${{ hashFiles('pyproject.toml') }} + restore-keys: | + uv-coverage- - name: Run coverage report via nox (fail-under 97%) id: coverage run: | mkdir -p build - nox -s coverage_report 2>&1 | tee build/nox-coverage-output.log + nox -s coverage_report 2>&1 | tee build/coverage-output.txt # Extract the single-line CI summary from nox output - grep -E '^(nox > )?COVERAGE (OK|FAILED):' build/nox-coverage-output.log || true + grep -E '^(nox > )?COVERAGE (OK|FAILED):' build/coverage-output.txt || true env: NOX_DEFAULT_VENV_BACKEND: uv @@ -378,7 +318,7 @@ jobs: data = json.load(f) summary = data.get('summary') or data.get('totals') or {} pct = round(summary.get('percent_covered', 0), 1) - threshold = 50 # Temporarily lowered; see issues #4183 and #4184 + threshold = 97 if pct >= threshold: print(f'COVERAGE OK: {pct}% (threshold: {threshold}%)') else: @@ -390,14 +330,6 @@ jobs: exit 1 fi - - name: Upload coverage log artifact - if: always() - uses: actions/upload-artifact@v3 - with: - name: ci-logs-coverage - path: build/nox-coverage-output.log - retention-days: 30 - - name: Upload coverage artifacts if: always() uses: actions/upload-artifact@v3 @@ -409,10 +341,124 @@ jobs: build/htmlcov/ retention-days: 30 + benchmark-regression: + if: forgejo.event_name == 'pull_request' + runs-on: docker-benchmark + container: + image: python:3.13-slim + needs: [lint, typecheck] + steps: + - name: Install system dependencies (nodejs for checkout, git for merge tests) + run: | + apt-get update && apt-get install -y -qq nodejs git && rm -rf /var/lib/apt/lists/* + + - name: Checkout full history + uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Compute base commit + id: hash + run: | + git fetch origin "${{ forgejo.base_ref }}" --depth=200 + BASE_SHA=$(git merge-base HEAD "origin/${{ forgejo.base_ref }}") + echo "ASV_BASE_SHA=${BASE_SHA}" >> $FORGEJO_OUTPUT + + - name: Install dependencies + run: | + python -m pip install -U pip + python -m pip install asv virtualenv uv==${{ env.UV_VERSION }} nox + + - name: Restore prior ASV benchmarks + env: + AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }} + AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }} + AWS_DEFAULT_REGION: ${{ secrets.AWS_DEFAULT_REGION }} + ASV_S3_BUCKET: ${{ secrets.ASV_S3_BUCKET }} + run: | + python -m pip install awscli + mkdir -p build/asv/results + aws s3 sync "s3://${ASV_S3_BUCKET}/asv/results" build/asv/results --delete || true + + - name: Run asv continuous via nox + env: + ASV_BASE_SHA: ${{ steps.hash.outputs.ASV_BASE_SHA }} + run: | + nox -s benchmark_regression + + - name: Archive the results + run: | + tar cf /tmp/asv-results.tar build/asv/results build/asv/html + + - name: Upload benchmark artifacts + if: always() + uses: actions/upload-artifact@v3 + with: + name: asv-results-pr + path: /tmp/asv-results.tar + rentention-days: 30 + + benchmark-publish: + if: forgejo.event_name == 'push' && ( forgejo.ref == 'refs/heads/master' || forgejo.ref == 'refs/heads/develop' ) + runs-on: docker-benchmark + container: python:3.13-slim + steps: + - name: Install system dependencies (nodejs for checkout, git for merge tests) + run: | + apt-get update && apt-get install -y -qq nodejs git && rm -rf /var/lib/apt/lists/* + + - name: Checkout full history + uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Install dependencies + run: | + python -m pip install -U pip + python -m pip install asv virtualenv uv==${{ env.UV_VERSION }} nox + + - name: Restore prior ASV benchmarks + env: + AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }} + AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }} + AWS_DEFAULT_REGION: ${{ secrets.AWS_DEFAULT_REGION }} + ASV_S3_BUCKET: ${{ secrets.ASV_S3_BUCKET }} + run: | + python -m pip install awscli + mkdir -p build/asv/results + aws s3 sync "s3://${ASV_S3_BUCKET}/asv/results" build/asv/results --delete || true + + - name: Run asv via nox + run: | + nox -s benchmark + + - name: Upload updated benchmarks and website to S3 + env: + AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_ID }} + AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_ACCESS_KEY }} + AWS_DEFAULT_REGION: ${{ secrets.AWS_DEFAULT_REGION }} + ASV_S3_BUCKET: ${{ secrets.ASV_S3_BUCKET }} + run: | + python -m pip install awscli + aws s3 sync build/asv/results "s3://${ASV_S3_BUCKET}/asv/results" --delete + aws s3 sync build/asv/html "s3://${ASV_S3_BUCKET}/asv/html" --delete + + - name: Archive the results + run: | + tar cf /tmp/asv-results.tar build/asv/results build/asv/html + + - name: Upload benchmark artifacts + if: always() + uses: actions/upload-artifact@v3 + with: + name: asv-results-pr + path: /tmp/asv-results.tar + rentention-days: 30 + build: runs-on: docker container: - image: ${{vars.docker_prefix}}python:3.13-slim + image: python:3.13-slim steps: - name: Install Node.js (required by actions/checkout) run: | @@ -424,14 +470,6 @@ jobs: run: | pip install -q uv==${{ env.UV_VERSION }} nox - - name: Cache uv packages - uses: actions/cache@v3 - with: - path: ~/.cache/uv - key: uv-${{ hashFiles('pyproject.toml') }} - restore-keys: | - uv- - - name: Build wheel via nox run: | nox -s build @@ -439,10 +477,10 @@ jobs: NOX_DEFAULT_VENV_BACKEND: uv docker: - needs: [lint, typecheck, security, quality, unit_tests] + needs: [lint, typecheck, unit_tests, security] runs-on: docker container: - image: ${{vars.docker_prefix}}docker:dind + image: docker:dind options: --privileged steps: - name: Start Docker daemon and install dependencies @@ -453,163 +491,20 @@ jobs: - uses: actions/checkout@v4 - - name: Build Docker image (CLI) + - name: Build Docker image run: | docker build -t cleverernie:test . - - name: Test Docker image (CLI) + - name: Test Docker image run: | docker run --rm cleverernie:test --version - - name: Build Docker image (Server) - run: | - docker build -f Dockerfile.server -t cleveragents-server:test . - - helm: - runs-on: docker - container: - image: ${{vars.docker_prefix}}python:3.13-slim - steps: - - name: Install system dependencies (nodejs for checkout, curl for Helm and kubeconform) - run: | - apt-get update && apt-get install -y -qq nodejs curl tar && rm -rf /var/lib/apt/lists/* - - - uses: actions/checkout@v4 - - - name: Install Helm CLI - run: | - HELM_VERSION="v3.16.4" - ARCH="amd64" - HELM_TARBALL="helm-${HELM_VERSION}-linux-${ARCH}.tar.gz" - curl -fsSL "https://get.helm.sh/${HELM_TARBALL}" -o "/tmp/${HELM_TARBALL}" - curl -fsSL "https://get.helm.sh/${HELM_TARBALL}.sha256sum" -o /tmp/helm.sha256sum - cd /tmp && sha256sum -c helm.sha256sum - tar -xzf "/tmp/${HELM_TARBALL}" -C /tmp - mv /tmp/linux-${ARCH}/helm /usr/local/bin/helm - chmod +x /usr/local/bin/helm - helm version --short - - - name: Install kubeconform - run: | - KUBECONFORM_VERSION="v0.7.0" - ARCH="amd64" - KUBECONFORM_TARBALL="kubeconform-linux-${ARCH}.tar.gz" - curl -fsSL "https://github.com/yannh/kubeconform/releases/download/${KUBECONFORM_VERSION}/${KUBECONFORM_TARBALL}" \ - -o "/tmp/${KUBECONFORM_TARBALL}" - tar -xzf "/tmp/${KUBECONFORM_TARBALL}" -C /tmp - mv /tmp/kubeconform /usr/local/bin/kubeconform - chmod +x /usr/local/bin/kubeconform - kubeconform -v - - - name: Build Helm chart dependencies - run: | - helm dependency build ./k8s - - - name: Helm lint chart - run: | - helm lint ./k8s \ - --set database.url="postgresql+asyncpg://user:pass@db-host:5432/cleveragents" - - - name: Helm template smoke render - run: | - helm template cleveragents ./k8s \ - --set database.url="postgresql+asyncpg://user:pass@db-host:5432/cleveragents" >/tmp/rendered.yaml - test -s /tmp/rendered.yaml - - - name: Validate rendered manifests with kubeconform - run: | - kubeconform \ - -strict \ - -ignore-missing-schemas \ - -kubernetes-version 1.29.0 \ - -summary \ - /tmp/rendered.yaml - - - push-validation: - # Validates that the CI runner can authenticate and push to the repository. - # Root cause of the push failure: actions/checkout@v4 was not configured with - # explicit push credentials (token + persist-credentials), and no git user - # config (name/email) was set — both are required for any push operation. - runs-on: docker - container: - image: ${{vars.docker_prefix}}python:3.13-slim - steps: - - name: Install system dependencies (nodejs for checkout, git for push validation) - run: | - apt-get update && apt-get install -y -qq nodejs git curl && rm -rf /var/lib/apt/lists/* - - - name: Checkout with explicit write credentials - uses: actions/checkout@v4 - with: - # Pass the Forgejo token explicitly so the credential helper is - # configured for HTTPS push operations. Without this, the default - # checkout may only have read access and push will fail with a - # 403 or authentication error. - token: ${{ secrets.FORGEJO_TOKEN }} - persist-credentials: true - - - name: Configure git user for CI operations - run: | - # Required for any git commit or push operation in CI. - # Uses a bot identity to distinguish CI-generated commits from - # human commits. Without this, git push fails with: - # "Author identity unknown — please tell me who you are." - git config user.name "CleverAgents CI" - git config user.email "ci-bot@cleverthis.com" - - - name: Verify HTTPS credential helper is configured - run: | - # Confirm that the credential helper set up by actions/checkout - # is active. This ensures HTTPS push operations will authenticate - # correctly without prompting for a password. - echo "=== Git credential configuration ===" - git config --list | grep -E "credential|url" || echo "WARNING: No credential helper found" - echo "=== Remote URL ===" - git remote get-url origin - echo "=== Credential helper check ===" - if git config credential.helper > /dev/null 2>&1; then - echo "OK: Credential helper is configured: $(git config credential.helper)" - else - echo "WARNING: No credential helper configured — push may fail" - fi - - - name: Smoke-test push access via API - # Validates write permission using the Forgejo API before attempting - # any real push. This catches credential issues early with a clear - # error message rather than a cryptic git error. - env: - FORGEJO_URL: ${{ secrets.FORGEJO_URL }} - FORGEJO_TOKEN: ${{ secrets.FORGEJO_TOKEN }} - run: | - REPO="${{ forgejo.repository }}" - API_URL="${FORGEJO_URL}/api/v1/repos/${REPO}" - echo "=== Testing repository API access ===" - HTTP_STATUS=$(curl -s -o /dev/null -w "%{http_code}" \ - -H "Authorization: token ${FORGEJO_TOKEN}" \ - "${API_URL}") - if [ "${HTTP_STATUS}" != "200" ]; then - echo "ERROR: FORGEJO_TOKEN cannot access repository API (HTTP ${HTTP_STATUS})." - echo "Ensure FORGEJO_TOKEN is set in Repository Settings > Actions > Secrets" - echo "and that the token has repository (write) scope." - exit 1 - fi - PUSH_ALLOWED=$(curl -s \ - -H "Authorization: token ${FORGEJO_TOKEN}" \ - "${API_URL}" | python3 -c "import sys,json; d=json.load(sys.stdin); print(str(d.get('permissions',{}).get('push',False)).lower())") - if [ "${PUSH_ALLOWED}" != "true" ]; then - echo "ERROR: FORGEJO_TOKEN does not have push (write) permission." - echo "Grant the token Contents: Write permission or use a token with full repository scope." - exit 1 - fi - echo "OK: Push access verified -- FORGEJO_TOKEN has write permission on ${REPO}" - echo "=== Push access smoke-test passed ===" status-check: if: always() - needs: [lint, typecheck, security, quality, unit_tests, integration_tests, e2e_tests, coverage, build, docker, helm, push-validation] + needs: [lint, typecheck, security, quality, unit_tests, coverage, build, docker, tdd_quality_gate] runs-on: docker container: - image: ${{vars.docker_prefix}}python:3.13-slim + image: python:3.13-slim steps: - name: Check required job results run: | @@ -618,27 +513,23 @@ jobs: echo "security: ${{ needs.security.result }}" echo "quality: ${{ needs.quality.result }}" echo "unit_tests: ${{ needs.unit_tests.result }}" - echo "integration_tests: ${{ needs.integration_tests.result }}" - echo "e2e_tests: ${{ needs.e2e_tests.result }}" echo "coverage: ${{ needs.coverage.result }}" echo "build: ${{ needs.build.result }}" echo "docker: ${{ needs.docker.result }}" - echo "helm: ${{ needs.helm.result }}" - echo "push-validation: ${{ needs.push-validation.result }}" + echo "tdd_quality_gate: ${{ needs.tdd_quality_gate.result }}" if [ "${{ needs.lint.result }}" != "success" ] || \ [ "${{ needs.typecheck.result }}" != "success" ] || \ [ "${{ needs.security.result }}" != "success" ] || \ - [ "${{ needs.quality.result }}" != "success" ] || \ [ "${{ needs.unit_tests.result }}" != "success" ] || \ - [ "${{ needs.integration_tests.result }}" != "success" ] || \ - [ "${{ needs.e2e_tests.result }}" != "success" ] || \ - [ "${{ needs.coverage.result }}" != "success" ] || \ - [ "${{ needs.build.result }}" != "success" ] || \ - [ "${{ needs.docker.result }}" != "success" ] || \ - [ "${{ needs.helm.result }}" != "success" ] || \ - [ "${{ needs.push-validation.result }}" != "success" ]; then + [ "${{ needs.coverage.result }}" != "success" ]; then echo "FAILED: One or more required jobs did not succeed" exit 1 fi + + if [ "${{ forgejo.event_name }}" = "pull_request" ] && \ + [ "${{ needs.tdd_quality_gate.result }}" != "success" ]; then + echo "FAILED: tdd_quality_gate did not succeed" + exit 1 + fi echo "All required CI checks passed" diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 700234833..e7f7a802d 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1236,6 +1236,7 @@ session is missing required tooling, add the dependency to the session before re - **All tests (including static checks):** `nox` - **Unit tests only:** `nox -s unit_tests` - **Integration tests:** `nox -s integration_tests` +- **TDD bug-fix quality gate:** `PR_DESCRIPTION="Fixes #123" PR_BASE_REF=master nox -s tdd_quality_gate` - **Benchmarks:** `nox -s benchmark` - **Coverage report:** `nox -s coverage_report` @@ -1274,6 +1275,7 @@ The CI pipeline runs the following jobs. Jobs without explicit dependencies run | `quality` | — | Radon complexity analysis via `nox -s complexity` | | `unit_tests` | — | Behave BDD tests via `nox -s unit_tests` | | `integration_tests` | — | Robot Framework integration tests via `nox -s integration_tests` | +| `tdd_quality_gate` | — (PR-only) | TDD bug-fix workflow enforcement via `nox -s tdd_quality_gate` (checks bug refs, TDD tags, and expected-fail tag removal in PR diff) | | `e2e_tests` | — | End-to-end Robot tests with real LLM keys via `nox -s e2e_tests` | | `coverage` | `lint`, `typecheck` | Slipcover coverage report via `nox -s coverage_report` (fail-under 97%) | | `benchmark-regression` | `lint`, `typecheck` | ASV benchmark regression on PRs via `nox -s benchmark_regression` | @@ -1293,6 +1295,7 @@ mergeable: - **security** — No high-severity Bandit findings, Semgrep rules pass, no dead code - **unit_tests** — All Behave BDD scenarios pass - **coverage** — Test coverage >= 97% +- **tdd_quality_gate** — On PRs only, bug-fix PRs must satisfy the TDD Bug Fix Workflow gate #### How to Read CI Results diff --git a/features/environment.py b/features/environment.py index 4e9d411ae..fdd9385eb 100644 --- a/features/environment.py +++ b/features/environment.py @@ -676,6 +676,22 @@ def after_scenario(context, scenario): pass # Ignore cleanup errors context.test_dir = None + # Clean up temp directories created by tdd_quality_gate and other steps. + # temp_dir may be a Path, str, or tempfile.TemporaryDirectory depending + # on which step file set it; handle all three safely. + # NOTE: do NOT set context.temp_dir = None here — cleanup functions + # registered via context.add_cleanup() run AFTER after_scenario and may + # still reference context.temp_dir (e.g. cli_init_yes_flag_steps.py). + # The scenario context layer is popped after all cleanups finish, which + # removes the attribute automatically. + temp_dir = getattr(context, "temp_dir", None) + if temp_dir is not None: + if isinstance(temp_dir, (str, Path)): + shutil.rmtree(temp_dir, ignore_errors=True) + elif hasattr(temp_dir, "cleanup"): + with contextlib.suppress(Exception): + temp_dir.cleanup() + # Clean up environment variables set during tests if hasattr(context, "env_vars_to_clean"): for key in context.env_vars_to_clean: diff --git a/features/steps/tdd_quality_gate_steps.py b/features/steps/tdd_quality_gate_steps.py new file mode 100644 index 000000000..a71982bd2 --- /dev/null +++ b/features/steps/tdd_quality_gate_steps.py @@ -0,0 +1,541 @@ +"""Step definitions for TDD quality gate feature tests.""" + +from __future__ import annotations + +import os +import shutil +import sys +import tempfile +from pathlib import Path + +from behave import given, then, when + +# Ensure the project root is importable. +_ROOT = str(Path(__file__).resolve().parents[2]) +if _ROOT not in sys.path: + sys.path.insert(0, _ROOT) + +from scripts.tdd_quality_gate import ( # noqa: E402 + check_expected_fail_removed, + find_tdd_tests, + parse_bug_refs, + run_quality_gate, +) +from scripts.tdd_quality_gate import main as quality_gate_main # noqa: E402 + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + + +def _ensure_temp_dir(context: object) -> Path: + """Return (and lazily create) the temporary directory on context.""" + tmp: Path | None = getattr(context, "temp_dir", None) + if tmp is None: + tmp = Path(tempfile.mkdtemp()) + context.temp_dir = tmp # type: ignore[attr-defined] + return tmp + + +def _cleanup_temp_dir(context: object) -> None: + """Remove the temporary directory if it exists.""" + tmp: Path | None = getattr(context, "temp_dir", None) + if tmp is not None and tmp.exists(): + shutil.rmtree(tmp, ignore_errors=True) + + +def _default_pr_diff_for_bug_refs( + bug_refs: list[int], search_root: Path | None = None +) -> str: + """Return a synthetic PR diff that removes expected-fail tags. + + When *search_root* is provided the helper inspects the temp tree to + decide whether each bug's TDD test lives in a ``.feature`` or + ``.robot`` file and emits the diff in the matching format. + """ + chunks: list[str] = [] + for bug_num in bug_refs: + use_robot = False + if search_root is not None: + robot_hits = list(search_root.rglob("*.robot")) + for rp in robot_hits: + try: + if f"tdd_bug_{bug_num}" in rp.read_text(encoding="utf-8"): + use_robot = True + break + except (OSError, UnicodeDecodeError): + continue + + if use_robot: + chunks.append( + "\n".join( + [ + f"diff --git a/robot/bug{bug_num}.robot b/robot/bug{bug_num}.robot", + f"--- a/robot/bug{bug_num}.robot", + f"+++ b/robot/bug{bug_num}.robot", + "@@ -1 +1 @@", + f"-tdd_expected_fail tdd_bug tdd_bug_{bug_num}", + f"+tdd_bug tdd_bug_{bug_num}", + ] + ) + ) + else: + chunks.append( + "\n".join( + [ + f"diff --git a/features/bug{bug_num}.feature b/features/bug{bug_num}.feature", + f"--- a/features/bug{bug_num}.feature", + f"+++ b/features/bug{bug_num}.feature", + "@@ -1 +1 @@", + f"-@tdd_expected_fail @tdd_bug @tdd_bug_{bug_num}", + f"+@tdd_bug @tdd_bug_{bug_num}", + ] + ) + ) + return "\n".join(chunks) + + +# --------------------------------------------------------------------------- +# PR description parsing +# --------------------------------------------------------------------------- + + +@given('a PR description "{description}"') +def step_given_pr_description(context: object, description: str) -> None: + context.pr_description = description # type: ignore[attr-defined] + + +@given("a multiline PR description") +def step_given_multiline_pr_description(context: object) -> None: + context.pr_description = context.text # type: ignore[attr-defined] + + +@when("I parse the bug references") +def step_when_parse_bug_refs(context: object) -> None: + context.bug_refs = parse_bug_refs( # type: ignore[attr-defined] + context.pr_description # type: ignore[attr-defined] + ) + + +@then("the bug references should be [{refs}]") +def step_then_bug_refs_should_be(context: object, refs: str) -> None: + if refs.strip() == "": + expected: list[int] = [] + else: + expected = [int(x.strip()) for x in refs.split(",")] + actual: list[int] = context.bug_refs # type: ignore[attr-defined] + if actual != expected: + raise AssertionError(f"Expected bug refs {expected}, got {actual}") + + +@then("the bug references should be []") +def step_then_bug_refs_empty(context: object) -> None: + actual: list[int] = context.bug_refs # type: ignore[attr-defined] + if actual != []: + raise AssertionError(f"Expected empty bug refs, got {actual}") + + +# --------------------------------------------------------------------------- +# TDD test search +# --------------------------------------------------------------------------- + + +@given('a temporary directory with a file "{filepath}" containing "{content}"') +def step_given_temp_dir_with_file(context: object, filepath: str, content: str) -> None: + _cleanup_temp_dir(context) + tmp = _ensure_temp_dir(context) + full_path = tmp / filepath + full_path.parent.mkdir(parents=True, exist_ok=True) + full_path.write_text(content, encoding="utf-8") + + +@given('a temporary directory also has a file "{filepath}" containing "{content}"') +def step_given_temp_dir_also_has_file( + context: object, filepath: str, content: str +) -> None: + tmp = _ensure_temp_dir(context) + full_path = tmp / filepath + full_path.parent.mkdir(parents=True, exist_ok=True) + full_path.write_text(content, encoding="utf-8") + + +@when("I search for TDD tests for bug {bug_num:d}") +def step_when_search_tdd_tests(context: object, bug_num: int) -> None: + tmp = _ensure_temp_dir(context) + context.found_tests = find_tdd_tests(bug_num, tmp) # type: ignore[attr-defined] + + +@then("the search should find {count:d} test file") +def step_then_search_finds_count(context: object, count: int) -> None: + actual = len(context.found_tests) # type: ignore[attr-defined] + if actual != count: + raise AssertionError(f"Expected {count} test file(s), found {actual}") + + +@then("the search should find {count:d} test files") +def step_then_search_finds_count_plural(context: object, count: int) -> None: + actual = len(context.found_tests) # type: ignore[attr-defined] + if actual != count: + raise AssertionError(f"Expected {count} test file(s), found {actual}") + + +# --------------------------------------------------------------------------- +# Tag removal verification +# --------------------------------------------------------------------------- + + +@when("I check expected fail removal for bug {bug_num:d}") +def step_when_check_removal(context: object, bug_num: int) -> None: + tmp = _ensure_temp_dir(context) + # Filter by bug tag first — matching the production path in run_quality_gate. + test_files = find_tdd_tests(bug_num, tmp) + context.removal_errors = check_expected_fail_removed( # type: ignore[attr-defined] + test_files, bug_num + ) + + +@then("there should be {count:d} removal error") +def step_then_removal_errors_count(context: object, count: int) -> None: + actual = len(context.removal_errors) # type: ignore[attr-defined] + if actual != count: + raise AssertionError( + f"Expected {count} removal error(s), got {actual}: {context.removal_errors}" # type: ignore[attr-defined] + ) + + +@then("there should be {count:d} removal errors") +def step_then_removal_errors_count_plural(context: object, count: int) -> None: + actual = len(context.removal_errors) # type: ignore[attr-defined] + if actual != count: + raise AssertionError( + f"Expected {count} removal error(s), got {actual}: {context.removal_errors}" # type: ignore[attr-defined] + ) + + +@then('the removal error should mention "{text}"') +def step_then_removal_error_mentions(context: object, text: str) -> None: + errors: list[str] = context.removal_errors # type: ignore[attr-defined] + found = any(text in err for err in errors) + if not found: + raise AssertionError( + f"Expected removal error mentioning '{text}', got: {errors}" + ) + + +# --------------------------------------------------------------------------- +# Full quality gate +# --------------------------------------------------------------------------- + + +@given("a temporary search root") +def step_given_temp_search_root(context: object) -> None: + _cleanup_temp_dir(context) + _ensure_temp_dir(context) + context.pr_diff = None # type: ignore[attr-defined] + + +@given('a temporary search root with file "{filepath}" containing "{content}"') +def step_given_temp_search_root_with_file( + context: object, filepath: str, content: str +) -> None: + _cleanup_temp_dir(context) + tmp = _ensure_temp_dir(context) + context.pr_diff = None # type: ignore[attr-defined] + full_path = tmp / filepath + full_path.parent.mkdir(parents=True, exist_ok=True) + full_path.write_text(content, encoding="utf-8") + + +@given('the search root also has file "{filepath}" containing "{content}"') +def step_given_search_root_also_has( + context: object, filepath: str, content: str +) -> None: + tmp = _ensure_temp_dir(context) + full_path = tmp / filepath + full_path.parent.mkdir(parents=True, exist_ok=True) + full_path.write_text(content, encoding="utf-8") + + +@given("the PR diff does not remove expected fail tags") +def step_given_pr_diff_no_expected_fail_removal(context: object) -> None: + context.pr_diff = "" # type: ignore[attr-defined] + + +@when("I run the quality gate") +def step_when_run_quality_gate(context: object) -> None: + tmp = _ensure_temp_dir(context) + pr_desc: str = getattr(context, "pr_description", "") + pr_diff: str | None = getattr(context, "pr_diff", None) + if pr_diff is None: + bug_refs = parse_bug_refs(pr_desc) + pr_diff = _default_pr_diff_for_bug_refs(bug_refs, tmp) + errors, _bug_refs = run_quality_gate(pr_desc, tmp, pr_diff=pr_diff) + context.gate_errors = errors # type: ignore[attr-defined] + + +@then("the quality gate should pass") +def step_then_gate_passes(context: object) -> None: + errors: list[str] = context.gate_errors # type: ignore[attr-defined] + if errors: + raise AssertionError(f"Quality gate should pass but got errors: {errors}") + + +@then("the quality gate should fail") +def step_then_gate_fails(context: object) -> None: + errors: list[str] = context.gate_errors # type: ignore[attr-defined] + if not errors: + raise AssertionError("Quality gate should fail but passed with no errors") + + +@then('the quality gate errors should mention "{text}"') +def step_then_gate_errors_mention(context: object, text: str) -> None: + errors: list[str] = context.gate_errors # type: ignore[attr-defined] + found = any(text in err for err in errors) + if not found: + raise AssertionError( + f"Expected quality gate error mentioning '{text}', got: {errors}" + ) + + +# --------------------------------------------------------------------------- +# Argument validation +# --------------------------------------------------------------------------- + + +@when("I call parse_bug_refs with a non-string argument") +def step_when_parse_bug_refs_non_string(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + parse_bug_refs(123) # type: ignore[arg-type] + except TypeError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@when("I call find_tdd_tests with bug number {num:d}") +def step_when_find_tdd_tests_invalid(context: object, num: int) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + find_tdd_tests(num, Path("/tmp/nonexistent")) + except ValueError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@when("I call find_tdd_tests with a non-Path search root") +def step_when_find_tdd_tests_non_path(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + find_tdd_tests(1, "/tmp/nonexistent") # type: ignore[arg-type] + except TypeError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@then("a TypeError should be raised by the quality gate") +def step_then_type_error_raised(context: object) -> None: + exc = getattr(context, "caught_exception", None) + if exc is None: + raise AssertionError("Expected TypeError but no exception was raised") + if not isinstance(exc, TypeError): + raise AssertionError(f"Expected TypeError, got {type(exc).__name__}: {exc}") + + +@then("a ValueError should be raised by the quality gate") +def step_then_value_error_raised(context: object) -> None: + exc = getattr(context, "caught_exception", None) + if exc is None: + raise AssertionError("Expected ValueError but no exception was raised") + if not isinstance(exc, ValueError): + raise AssertionError(f"Expected ValueError, got {type(exc).__name__}: {exc}") + + +# --------------------------------------------------------------------------- +# Robot-format diff, unreadable files, and extra argument validation +# --------------------------------------------------------------------------- + + +@given("the PR diff removes expected fail for robot bug {bug_num:d}") +def step_given_robot_diff_removes_expected_fail(context: object, bug_num: int) -> None: + context.pr_diff = "\n".join( # type: ignore[attr-defined] + [ + f"diff --git a/robot/bug{bug_num}.robot b/robot/bug{bug_num}.robot", + f"--- a/robot/bug{bug_num}.robot", + f"+++ b/robot/bug{bug_num}.robot", + "@@ -1 +1 @@", + f"-tdd_expected_fail tdd_bug tdd_bug_{bug_num}", + f"+tdd_bug tdd_bug_{bug_num}", + ] + ) + + +@given("a temporary directory with an unreadable feature file for bug {bug_num:d}") +def step_given_unreadable_feature_file(context: object, bug_num: int) -> None: + _cleanup_temp_dir(context) + tmp = _ensure_temp_dir(context) + full_path = tmp / "features" / "bug.feature" + full_path.parent.mkdir(parents=True, exist_ok=True) + # Write invalid UTF-8 bytes so read_text(encoding="utf-8") raises + # UnicodeDecodeError (caught as OSError subclass). This is root-safe + # unlike chmod(0o000) which root bypasses. + full_path.write_bytes( + f"@tdd_expected_fail @tdd_bug @tdd_bug_{bug_num}".encode() + b"\xff\xfe" + ) + + +@when("I call check_expected_fail_removed with a non-list argument") +def step_when_check_expected_fail_removed_non_list(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + check_expected_fail_removed("not-a-list", 1) # type: ignore[arg-type] + except TypeError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@when("I call check_expected_fail_removed with bug number {num:d}") +def step_when_check_expected_fail_removed_bad_bug(context: object, num: int) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + check_expected_fail_removed([], num) + except ValueError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +# --------------------------------------------------------------------------- +# Bool type guard +# --------------------------------------------------------------------------- + + +@when("I call find_tdd_tests with boolean True as bug number") +def step_when_find_tdd_tests_bool(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + find_tdd_tests(True, Path("/tmp/nonexistent")) # type: ignore[arg-type] + except (TypeError, ValueError) as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@when("I call check_expected_fail_removed with boolean True as bug number") +def step_when_check_expected_fail_removed_bool(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + check_expected_fail_removed([], True) # type: ignore[arg-type] + except (TypeError, ValueError) as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +# --------------------------------------------------------------------------- +# Co-located bug false positive guard (M1) +# --------------------------------------------------------------------------- + + +@given("the PR diff only removes expected fail for bug {other:d} not bug {target:d}") +def step_given_pr_diff_removes_wrong_bug( + context: object, other: int, target: int +) -> None: + # Craft a diff that removes expected-fail for bug `other` but NOT for + # bug `target`. The diff must mention bug `other`'s tag on the removed + # line, so the gate should NOT count this as a removal for `target`. + context.pr_diff = "\n".join( # type: ignore[attr-defined] + [ + "diff --git a/features/bugs99.feature b/features/bugs99.feature", + "--- a/features/bugs99.feature", + "+++ b/features/bugs99.feature", + "@@ -1 +1 @@", + f"-@tdd_expected_fail @tdd_bug @tdd_bug_{other}", + f"+@tdd_bug @tdd_bug_{other}", + ] + ) + + +# --------------------------------------------------------------------------- +# run_quality_gate argument validation (L4) +# --------------------------------------------------------------------------- + + +@when("I call run_quality_gate with a non-string PR description") +def step_when_run_quality_gate_non_str_desc(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + run_quality_gate(123, Path("/tmp")) # type: ignore[arg-type] + except TypeError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@when("I call run_quality_gate with a non-Path search root") +def step_when_run_quality_gate_non_path_root(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + run_quality_gate("desc", "/tmp") # type: ignore[arg-type] + except TypeError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@when("I call run_quality_gate with an empty base_ref") +def step_when_run_quality_gate_empty_base_ref(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + run_quality_gate("desc", Path("/tmp"), base_ref=" ") + except ValueError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +@when("I call run_quality_gate with a non-string pr_diff") +def step_when_run_quality_gate_non_str_diff(context: object) -> None: + context.caught_exception = None # type: ignore[attr-defined] + try: + run_quality_gate("desc", Path("/tmp"), pr_diff=123) # type: ignore[arg-type] + except TypeError as exc: + context.caught_exception = exc # type: ignore[attr-defined] + + +# --------------------------------------------------------------------------- +# main() CLI entry point (M2) +# --------------------------------------------------------------------------- + + +@when('I call main with PR_DESCRIPTION "{description}"') +def step_when_call_main_with_desc(context: object, description: str) -> None: + tmp = _ensure_temp_dir(context) + saved_desc = os.environ.get("PR_DESCRIPTION") + saved_cwd = os.getcwd() + try: + os.environ["PR_DESCRIPTION"] = description + os.chdir(tmp) + context.main_exit_code = quality_gate_main() # type: ignore[attr-defined] + finally: + os.chdir(saved_cwd) + if saved_desc is None: + os.environ.pop("PR_DESCRIPTION", None) + else: + os.environ["PR_DESCRIPTION"] = saved_desc + + +@when('I call main with PR_DESCRIPTION "{description}" in an empty search root') +def step_when_call_main_missing_test(context: object, description: str) -> None: + _cleanup_temp_dir(context) + tmp = _ensure_temp_dir(context) + saved_desc = os.environ.get("PR_DESCRIPTION") + saved_base = os.environ.get("PR_BASE_REF") + saved_cwd = os.getcwd() + try: + os.environ["PR_DESCRIPTION"] = description + # Use a base_ref that won't exist in the temp dir (not a git repo), + # so _collect_pr_diff will fail and produce an error about the diff. + os.environ["PR_BASE_REF"] = "master" + os.chdir(tmp) + context.main_exit_code = quality_gate_main() # type: ignore[attr-defined] + finally: + os.chdir(saved_cwd) + if saved_desc is None: + os.environ.pop("PR_DESCRIPTION", None) + else: + os.environ["PR_DESCRIPTION"] = saved_desc + if saved_base is None: + os.environ.pop("PR_BASE_REF", None) + else: + os.environ["PR_BASE_REF"] = saved_base + + +@then("the main exit code should be {code:d}") +def step_then_main_exit_code(context: object, code: int) -> None: + actual = context.main_exit_code # type: ignore[attr-defined] + if actual != code: + raise AssertionError(f"Expected main exit code {code}, got {actual}") diff --git a/features/tdd_quality_gate.feature b/features/tdd_quality_gate.feature new file mode 100644 index 000000000..3b65982d7 --- /dev/null +++ b/features/tdd_quality_gate.feature @@ -0,0 +1,265 @@ +Feature: TDD bug tag quality gate for bug fix PRs + As a CI system + I want to enforce TDD bug fix workflow rules on PRs + So that bug fix PRs follow the required TDD workflow + + # --- PR description parsing --- + + Scenario: Parse single Fixes reference + Given a PR description "Fixes #42" + When I parse the bug references + Then the bug references should be [42] + + Scenario: Parse single Closes reference + Given a PR description "Closes #100" + When I parse the bug references + Then the bug references should be [100] + + Scenario: Parse single Resolves reference + Given a PR description "Resolves #7" + When I parse the bug references + Then the bug references should be [7] + + Scenario: Parse case-insensitive closing keywords + Given a PR description "fixes #10 and CLOSES #20" + When I parse the bug references + Then the bug references should be [10, 20] + + Scenario: Parse ISSUES CLOSED block + Given a PR description "ISSUES CLOSED: #5, #10" + When I parse the bug references + Then the bug references should be [5, 10] + + Scenario: Parse multiple mixed references + Given a PR description "Fixes #42, also closes #99. ISSUES CLOSED: #7" + When I parse the bug references + Then the bug references should be [7, 42, 99] + + Scenario: No bug references returns empty list + Given a PR description "Add new feature for users" + When I parse the bug references + Then the bug references should be [] + + Scenario: Deduplicate repeated bug references + Given a PR description "Fixes #42. Also closes #42" + When I parse the bug references + Then the bug references should be [42] + + Scenario: Parse past tense closing keywords + Given a PR description "Fixed #15, Closed #20, Resolved #25" + When I parse the bug references + Then the bug references should be [15, 20, 25] + + Scenario: Ignore non-closing words that end with keyword substrings + Given a PR description "prefixes #12 and hotfixes #34" + When I parse the bug references + Then the bug references should be [] + + Scenario: Parse bug reference in multi-line PR description + Given a multiline PR description + """ + feat(ci): implement quality gate + + This PR adds the TDD quality gate. + + Fixes #42 + """ + When I parse the bug references + Then the bug references should be [42] + + # --- TDD test search --- + + Scenario: Find TDD test in .feature file + Given a temporary directory with a file "tests/bug.feature" containing "@tdd_bug_42" + When I search for TDD tests for bug 42 + Then the search should find 1 test file + + Scenario: Find TDD test in .robot file + Given a temporary directory with a file "tests/bug.robot" containing "tdd_bug_42" + When I search for TDD tests for bug 42 + Then the search should find 1 test file + + Scenario: Find TDD tests in both .feature and .robot files + Given a temporary directory with a file "features/bug.feature" containing "@tdd_bug_42" + And a temporary directory also has a file "robot/bug.robot" containing "tdd_bug_42" + When I search for TDD tests for bug 42 + Then the search should find 2 test files + + Scenario: No TDD test found for bug number + Given a temporary directory with a file "tests/other.feature" containing "@tdd_bug_99" + When I search for TDD tests for bug 42 + Then the search should find 0 test files + + Scenario: Do not match partial TDD bug tags + Given a temporary directory with a file "tests/partial.feature" containing "@tdd_bug_420" + When I search for TDD tests for bug 42 + Then the search should find 0 test files + + # --- Tag removal verification --- + + Scenario: Expected fail tag still present in .feature file + Given a temporary directory with a file "tests/bug.feature" containing "@tdd_expected_fail @tdd_bug @tdd_bug_42" + When I check expected fail removal for bug 42 + Then there should be 1 removal error + And the removal error should mention "@tdd_expected_fail" + And the removal error should mention "@tdd_bug_42" + + Scenario: Expected fail tag removed from .feature file + Given a temporary directory with a file "tests/bug.feature" containing "@tdd_bug @tdd_bug_42" + When I check expected fail removal for bug 42 + Then there should be 0 removal errors + + Scenario: Expected fail tag still present in .robot file + Given a temporary directory with a file "tests/bug.robot" containing "tdd_expected_fail tdd_bug_42" + When I check expected fail removal for bug 42 + Then there should be 1 removal error + + Scenario: Expected fail tag removed from .robot file + Given a temporary directory with a file "tests/bug.robot" containing "tdd_bug tdd_bug_42" + When I check expected fail removal for bug 42 + Then there should be 0 removal errors + + # --- Full quality gate --- + + Scenario: Quality gate passes when no bug refs in PR + Given a temporary search root + And a PR description "Add new feature" + When I run the quality gate + Then the quality gate should pass + + Scenario: Quality gate fails when no TDD test exists for referenced bug + Given a temporary search root + And a PR description "Fixes #42" + When I run the quality gate + Then the quality gate should fail + And the quality gate errors should mention "No TDD test found for bug #42" + + Scenario: Quality gate fails when expected fail tag is still present + Given a temporary search root with file "features/bug.feature" containing "@tdd_expected_fail @tdd_bug @tdd_bug_42" + And a PR description "Fixes #42" + When I run the quality gate + Then the quality gate should fail + And the quality gate errors should mention "@tdd_expected_fail" + + Scenario: Quality gate passes when expected fail tag has been removed + Given a temporary search root with file "features/bug.feature" containing "@tdd_bug @tdd_bug_42" + And a PR description "Fixes #42" + When I run the quality gate + Then the quality gate should pass + + Scenario: Quality gate handles multiple bug references + Given a temporary search root with file "features/bug10.feature" containing "@tdd_bug @tdd_bug_10" + And the search root also has file "features/bug20.feature" containing "@tdd_expected_fail @tdd_bug @tdd_bug_20" + And a PR description "Fixes #10 and fixes #20" + When I run the quality gate + Then the quality gate should fail + And the quality gate errors should mention "@tdd_bug_20" + + Scenario: Quality gate passes when all bugs have clean TDD tests + Given a temporary search root with file "features/bug10.feature" containing "@tdd_bug @tdd_bug_10" + And the search root also has file "robot/bug20.robot" containing "tdd_bug tdd_bug_20" + And a PR description "Fixes #10 and fixes #20" + When I run the quality gate + Then the quality gate should pass + + Scenario: Quality gate fails when PR diff does not remove expected fail tags + Given a temporary search root with file "features/bug.feature" containing "@tdd_bug @tdd_bug_42" + And a PR description "Fixes #42" + And the PR diff does not remove expected fail tags + When I run the quality gate + Then the quality gate should fail + And the quality gate errors should mention "No removal of @tdd_expected_fail / tdd_expected_fail detected" + + Scenario: Quality gate passes for robot diff with expected fail removed across hunks + Given a temporary search root with file "robot/bug.robot" containing "tdd_bug tdd_bug_42" + And a PR description "Fixes #42" + And the PR diff removes expected fail for robot bug 42 + When I run the quality gate + Then the quality gate should pass + + Scenario: Issue number zero is silently ignored + Given a temporary search root + And a PR description "Fixes #0" + When I run the quality gate + Then the quality gate should pass + + Scenario: find_tdd_tests skips unreadable files + Given a temporary directory with an unreadable feature file for bug 42 + When I search for TDD tests for bug 42 + Then the search should find 0 test files + + Scenario: check_expected_fail_removed skips unreadable files + Given a temporary directory with an unreadable feature file for bug 42 + When I check expected fail removal for bug 42 + Then there should be 0 removal errors + + # --- Argument validation --- + + Scenario: parse_bug_refs rejects non-string input + When I call parse_bug_refs with a non-string argument + Then a TypeError should be raised by the quality gate + + Scenario: find_tdd_tests rejects invalid bug number + When I call find_tdd_tests with bug number 0 + Then a ValueError should be raised by the quality gate + + Scenario: find_tdd_tests rejects non-Path search root + When I call find_tdd_tests with a non-Path search root + Then a TypeError should be raised by the quality gate + + Scenario: check_expected_fail_removed rejects non-list input + When I call check_expected_fail_removed with a non-list argument + Then a TypeError should be raised by the quality gate + + Scenario: check_expected_fail_removed rejects invalid bug number + When I call check_expected_fail_removed with bug number 0 + Then a ValueError should be raised by the quality gate + + # --- Bool type guard --- + + Scenario: find_tdd_tests rejects boolean True as bug number + When I call find_tdd_tests with boolean True as bug number + Then a ValueError should be raised by the quality gate + + Scenario: check_expected_fail_removed rejects boolean True as bug number + When I call check_expected_fail_removed with boolean True as bug number + Then a ValueError should be raised by the quality gate + + # --- Co-located bug false positive guard (M1) --- + + Scenario: Diff detection does not false-positive for co-located bug tests + Given a temporary search root with file "features/bugs.feature" containing "@tdd_bug @tdd_bug_42" + And the search root also has file "features/bugs99.feature" containing "@tdd_bug @tdd_bug_99" + And a PR description "Fixes #42" + And the PR diff only removes expected fail for bug 99 not bug 42 + When I run the quality gate + Then the quality gate should fail + And the quality gate errors should mention "No removal of @tdd_expected_fail / tdd_expected_fail detected" + + # --- run_quality_gate argument validation (L4) --- + + Scenario: run_quality_gate rejects non-string PR description + When I call run_quality_gate with a non-string PR description + Then a TypeError should be raised by the quality gate + + Scenario: run_quality_gate rejects non-Path search root + When I call run_quality_gate with a non-Path search root + Then a TypeError should be raised by the quality gate + + Scenario: run_quality_gate rejects empty base_ref + When I call run_quality_gate with an empty base_ref + Then a ValueError should be raised by the quality gate + + Scenario: run_quality_gate rejects non-string pr_diff + When I call run_quality_gate with a non-string pr_diff + Then a TypeError should be raised by the quality gate + + # --- main() CLI entry point (M2) --- + + Scenario: main returns 0 when no bug refs in PR description + When I call main with PR_DESCRIPTION "Add new feature" + Then the main exit code should be 0 + + Scenario: main returns 1 when TDD test is missing + When I call main with PR_DESCRIPTION "Fixes #99999" in an empty search root + Then the main exit code should be 1 diff --git a/noxfile.py b/noxfile.py index 957467855..39f805683 100644 --- a/noxfile.py +++ b/noxfile.py @@ -873,6 +873,29 @@ def benchmark_regression(session: nox.Session): session.run("asv", "publish", f"--config={config_path}") +@nox.session(python=DEFAULT_PYTHON, reuse_venv=True, venv_backend="uv") +def tdd_quality_gate(session: nox.Session): + """Enforce TDD bug fix workflow rules on PRs. + + Reads the PR description from the ``PR_DESCRIPTION`` environment + variable and verifies that: + + 1. Every bug referenced via closing keywords (``Fixes #N``, + ``Closes #N``, ``Resolves #N``, ``ISSUES CLOSED: #N``) has + a corresponding TDD test tagged ``@tdd_bug_N``. + 2. The ``@tdd_expected_fail`` / ``tdd_expected_fail`` tag has been + removed from each of those tests in the PR diff. + + If no bug references are found the gate passes trivially. + """ + # The quality gate script uses only standard library; no install needed. + pr_description = os.environ.get("PR_DESCRIPTION", "") + pr_base_ref = os.environ.get("PR_BASE_REF", "master") + session.env["PR_DESCRIPTION"] = pr_description + session.env["PR_BASE_REF"] = pr_base_ref + session.run("python", "scripts/tdd_quality_gate.py") + + # Sessions to run by default when running `nox` without arguments nox.options.sessions = [ "lint", # ~5-10 seconds diff --git a/robot/helper_tdd_quality_gate.py b/robot/helper_tdd_quality_gate.py new file mode 100644 index 000000000..790f5fd0f --- /dev/null +++ b/robot/helper_tdd_quality_gate.py @@ -0,0 +1,412 @@ +"""Helper for ``tdd_quality_gate.robot`` — exercises TDD quality gate logic. + +Each sub-command exercises a specific aspect of the quality gate and +prints a sentinel string on success. Exit 0 = check passed, +1 = unexpected outcome. + +See CONTRIBUTING.md > Bug Fix Workflow for the full specification. +""" + +from __future__ import annotations + +__all__: list[str] = [] + +import shutil +import sys +import tempfile +from collections.abc import Callable +from pathlib import Path + +# Ensure the project root is importable. +_ROOT = str(Path(__file__).resolve().parents[1]) +if _ROOT not in sys.path: + sys.path.insert(0, _ROOT) + +from scripts.tdd_quality_gate import ( # noqa: E402 + find_tdd_tests, + parse_bug_refs, + run_quality_gate, +) + + +def _make_temp_tree(files: dict[str, str]) -> Path: + """Create a temporary directory with the given file tree.""" + tmp = Path(tempfile.mkdtemp()) + for filepath, content in files.items(): + full_path = tmp / filepath + full_path.parent.mkdir(parents=True, exist_ok=True) + full_path.write_text(content, encoding="utf-8") + return tmp + + +def _default_pr_diff_for_bug_refs( + bug_refs: list[int], search_root: Path | None = None +) -> str: + """Return a synthetic PR diff that removes expected-fail tags. + + When *search_root* is provided the helper inspects the temp tree to + decide whether each bug's TDD test lives in a ``.feature`` or + ``.robot`` file and emits the diff in the matching format. + """ + chunks: list[str] = [] + for bug_num in bug_refs: + use_robot = False + if search_root is not None: + robot_hits = list(search_root.rglob("*.robot")) + for rp in robot_hits: + try: + if f"tdd_bug_{bug_num}" in rp.read_text(encoding="utf-8"): + use_robot = True + break + except (OSError, UnicodeDecodeError): + continue + + if use_robot: + chunks.append( + "\n".join( + [ + ( + f"diff --git a/robot/bug{bug_num}.robot " + f"b/robot/bug{bug_num}.robot" + ), + f"--- a/robot/bug{bug_num}.robot", + f"+++ b/robot/bug{bug_num}.robot", + "@@ -1 +1 @@", + f"-tdd_expected_fail tdd_bug tdd_bug_{bug_num}", + f"+tdd_bug tdd_bug_{bug_num}", + ] + ) + ) + else: + chunks.append( + "\n".join( + [ + ( + f"diff --git a/features/bug{bug_num}.feature " + f"b/features/bug{bug_num}.feature" + ), + f"--- a/features/bug{bug_num}.feature", + f"+++ b/features/bug{bug_num}.feature", + "@@ -1 +1 @@", + f"-@tdd_expected_fail @tdd_bug @tdd_bug_{bug_num}", + f"+@tdd_bug @tdd_bug_{bug_num}", + ] + ) + ) + return "\n".join(chunks) + + +# --------------------------------------------------------------------------- +# PR description parsing +# --------------------------------------------------------------------------- + + +def parse_fixes_single() -> int: + """Verify parsing a single Fixes #N reference.""" + refs = parse_bug_refs("Fixes #42") + if refs != [42]: + print(f"FAIL: expected [42], got {refs}", file=sys.stderr) + return 1 + print("parse-fixes-single-ok") + return 0 + + +def parse_mixed_refs() -> int: + """Verify parsing multiple mixed closing keywords.""" + refs = parse_bug_refs("Fixes #42, also closes #99. Resolves #7") + if refs != [7, 42, 99]: + print(f"FAIL: expected [7, 42, 99], got {refs}", file=sys.stderr) + return 1 + print("parse-mixed-refs-ok") + return 0 + + +def parse_issues_closed() -> int: + """Verify parsing ISSUES CLOSED: block.""" + refs = parse_bug_refs("ISSUES CLOSED: #5, #10") + if refs != [5, 10]: + print(f"FAIL: expected [5, 10], got {refs}", file=sys.stderr) + return 1 + print("parse-issues-closed-ok") + return 0 + + +def parse_non_closing_words() -> int: + """Verify parser ignores embedded keyword substrings.""" + refs = parse_bug_refs("prefixes #12 and hotfixes #34") + if refs != []: + print(f"FAIL: expected [], got {refs}", file=sys.stderr) + return 1 + print("parse-non-closing-words-ok") + return 0 + + +def no_bug_refs_pass() -> int: + """Verify the quality gate passes when no bug refs in PR.""" + tmp = _make_temp_tree({}) + try: + errors, _refs = run_quality_gate("Add new feature", tmp) + if errors: + print(f"FAIL: expected no errors, got {errors}", file=sys.stderr) + return 1 + print("no-bug-refs-pass-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +# --------------------------------------------------------------------------- +# TDD test search +# --------------------------------------------------------------------------- + + +def find_feature_test() -> int: + """Verify finding @tdd_bug_N in .feature files.""" + tmp = _make_temp_tree( + {"features/bug.feature": "@tdd_bug @tdd_bug_42\nFeature: Test\n"} + ) + try: + tests = find_tdd_tests(42, tmp) + if len(tests) != 1: + print(f"FAIL: expected 1 test, found {len(tests)}", file=sys.stderr) + return 1 + print("find-feature-test-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def find_robot_test() -> int: + """Verify finding tdd_bug_N in .robot files.""" + tmp = _make_temp_tree({"robot/bug.robot": "[Tags] tdd_bug tdd_bug_42\n"}) + try: + tests = find_tdd_tests(42, tmp) + if len(tests) != 1: + print(f"FAIL: expected 1 test, found {len(tests)}", file=sys.stderr) + return 1 + print("find-robot-test-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def find_exact_tag_match() -> int: + """Verify partial tags are not treated as exact bug tag matches.""" + tmp = _make_temp_tree({"features/partial.feature": "@tdd_bug_420\nFeature: Test\n"}) + try: + tests = find_tdd_tests(42, tmp) + if tests: + print(f"FAIL: expected 0 tests, found {len(tests)}", file=sys.stderr) + return 1 + print("find-exact-tag-match-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +# --------------------------------------------------------------------------- +# Quality gate integration +# --------------------------------------------------------------------------- + + +def no_tdd_test_fails() -> int: + """Verify the gate fails when no TDD test exists for a referenced bug.""" + tmp = _make_temp_tree({}) + try: + errors, _refs = run_quality_gate("Fixes #42", tmp, pr_diff="") + if not errors: + print("FAIL: expected errors for missing TDD test", file=sys.stderr) + return 1 + if not any("No TDD test found for bug #42" in e for e in errors): + print( + f"FAIL: expected 'No TDD test found' error, got: {errors}", + file=sys.stderr, + ) + return 1 + print("no-tdd-test-fails-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def expected_fail_present() -> int: + """Verify the gate fails when @tdd_expected_fail is still present.""" + ef_content = "@tdd_expected_fail @tdd_bug @tdd_bug_42\nFeature: Test\n" + tmp = _make_temp_tree({"features/bug.feature": ef_content}) + try: + pr_diff = _default_pr_diff_for_bug_refs([42], tmp) + errors, _refs = run_quality_gate("Fixes #42", tmp, pr_diff=pr_diff) + if not errors: + print( + "FAIL: expected errors for @tdd_expected_fail present", file=sys.stderr + ) + return 1 + if not any("@tdd_expected_fail" in e for e in errors): + print( + f"FAIL: expected '@tdd_expected_fail' error, got: {errors}", + file=sys.stderr, + ) + return 1 + print("expected-fail-present-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def expected_fail_removed() -> int: + """Verify the gate passes when @tdd_expected_fail has been removed.""" + tmp = _make_temp_tree( + {"features/bug.feature": "@tdd_bug @tdd_bug_42\nFeature: Test\n"} + ) + try: + pr_diff = _default_pr_diff_for_bug_refs([42], tmp) + errors, _refs = run_quality_gate("Fixes #42", tmp, pr_diff=pr_diff) + if errors: + print(f"FAIL: expected no errors, got {errors}", file=sys.stderr) + return 1 + print("expected-fail-removed-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def multi_bug_mixed() -> int: + """Verify the gate handles multiple bugs with mixed outcomes.""" + tmp = _make_temp_tree( + { + "features/bug10.feature": "@tdd_bug @tdd_bug_10\nFeature: Bug 10\n", + "features/bug20.feature": ( + "@tdd_expected_fail @tdd_bug @tdd_bug_20\nFeature: Bug 20\n" + ), + } + ) + try: + pr_diff = _default_pr_diff_for_bug_refs([10, 20], tmp) + errors, _refs = run_quality_gate( + "Fixes #10 and fixes #20", tmp, pr_diff=pr_diff + ) + if not errors: + print("FAIL: expected errors for bug #20", file=sys.stderr) + return 1 + if not any("@tdd_bug_20" in e for e in errors): + print(f"FAIL: expected error about bug #20, got: {errors}", file=sys.stderr) + return 1 + # Bug #10 should not have errors + if any("@tdd_bug_10" in e for e in errors): + print(f"FAIL: unexpected error about bug #10: {errors}", file=sys.stderr) + return 1 + print("multi-bug-mixed-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def all_clean_passes() -> int: + """Verify the gate passes when all bugs have clean TDD tests.""" + tmp = _make_temp_tree( + { + "features/bug10.feature": "@tdd_bug @tdd_bug_10\nFeature: Bug 10\n", + "robot/bug20.robot": "[Tags] tdd_bug tdd_bug_20\n", + } + ) + try: + pr_diff = _default_pr_diff_for_bug_refs([10, 20], tmp) + errors, _refs = run_quality_gate( + "Fixes #10 and fixes #20", tmp, pr_diff=pr_diff + ) + if errors: + print(f"FAIL: expected no errors, got {errors}", file=sys.stderr) + return 1 + print("all-clean-passes-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def both_behave_and_robot() -> int: + """Verify the gate checks tests in both .feature and .robot files.""" + tmp = _make_temp_tree( + { + "features/bug.feature": "@tdd_bug @tdd_bug_42\nFeature: Bug\n", + "robot/bug.robot": "[Tags] tdd_bug tdd_bug_42\n", + } + ) + try: + tests = find_tdd_tests(42, tmp) + if len(tests) != 2: + print(f"FAIL: expected 2 tests, found {len(tests)}", file=sys.stderr) + return 1 + pr_diff = _default_pr_diff_for_bug_refs([42], tmp) + errors, _refs = run_quality_gate("Fixes #42", tmp, pr_diff=pr_diff) + if errors: + print(f"FAIL: expected no errors, got {errors}", file=sys.stderr) + return 1 + print("both-behave-and-robot-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +def diff_removal_required() -> int: + """Verify the gate fails when PR diff has no expected-fail removal.""" + tmp = _make_temp_tree( + {"features/bug.feature": "@tdd_bug @tdd_bug_42\nFeature: Test\n"} + ) + try: + errors, _refs = run_quality_gate("Fixes #42", tmp, pr_diff="") + if not errors: + print("FAIL: expected diff-removal error", file=sys.stderr) + return 1 + if not any("No removal of @tdd_expected_fail" in e for e in errors): + print( + f"FAIL: expected diff-removal message, got: {errors}", file=sys.stderr + ) + return 1 + print("diff-removal-required-ok") + return 0 + finally: + shutil.rmtree(tmp, ignore_errors=True) + + +# --------------------------------------------------------------------------- +# Dispatch +# --------------------------------------------------------------------------- + +_COMMANDS: dict[str, Callable[[], int]] = { + "parse_fixes_single": parse_fixes_single, + "parse_mixed_refs": parse_mixed_refs, + "parse_issues_closed": parse_issues_closed, + "parse_non_closing_words": parse_non_closing_words, + "no_bug_refs_pass": no_bug_refs_pass, + "find_feature_test": find_feature_test, + "find_robot_test": find_robot_test, + "find_exact_tag_match": find_exact_tag_match, + "no_tdd_test_fails": no_tdd_test_fails, + "expected_fail_present": expected_fail_present, + "expected_fail_removed": expected_fail_removed, + "multi_bug_mixed": multi_bug_mixed, + "all_clean_passes": all_clean_passes, + "both_behave_and_robot": both_behave_and_robot, + "diff_removal_required": diff_removal_required, +} + + +def main() -> int: + """Dispatch to the sub-command named in sys.argv[1].""" + if len(sys.argv) < 2: + print(f"Usage: {sys.argv[0]} ", file=sys.stderr) + print(f"Commands: {sorted(_COMMANDS)}", file=sys.stderr) + return 2 + + cmd = sys.argv[1] + handler = _COMMANDS.get(cmd) + if handler is None: + print(f"Unknown command: {cmd}", file=sys.stderr) + print(f"Available: {sorted(_COMMANDS)}", file=sys.stderr) + return 2 + + return handler() + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/robot/tdd_quality_gate.robot b/robot/tdd_quality_gate.robot new file mode 100644 index 000000000..dd80d7760 --- /dev/null +++ b/robot/tdd_quality_gate.robot @@ -0,0 +1,149 @@ +*** Settings *** +Documentation Integration tests for the TDD quality gate script. +... +... Exercises ``scripts/tdd_quality_gate.py`` end-to-end by +... invoking it as a subprocess with various PR_DESCRIPTION +... values and temporary file trees, then verifying exit codes +... and output messages. +... +... See CONTRIBUTING.md > Bug Fix Workflow for the full +... specification. +Resource ${CURDIR}/common.resource +Suite Setup Setup Test Environment +Suite Teardown Cleanup Test Environment + +*** Variables *** +${HELPER} ${CURDIR}/helper_tdd_quality_gate.py + +*** Test Cases *** +# =========================================================================== +# PR description parsing +# =========================================================================== + +Parse Single Fixes Reference + [Documentation] Verify the script extracts a single Fixes #N reference + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} parse_fixes_single + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} parse-fixes-single-ok + +Parse Multiple Mixed References + [Documentation] Verify the script extracts multiple closing keywords + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} parse_mixed_refs + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} parse-mixed-refs-ok + +Parse Issues Closed Block + [Documentation] Verify the script extracts ISSUES CLOSED: #N references + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} parse_issues_closed + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} parse-issues-closed-ok + +Parse Non Closing Words + [Documentation] Verify parser ignores embedded keyword substrings + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} parse_non_closing_words + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} parse-non-closing-words-ok + +No Bug References Passes Trivially + [Documentation] Verify the quality gate passes when no bug refs in PR + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} no_bug_refs_pass + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} no-bug-refs-pass-ok + +# =========================================================================== +# TDD test search and tag verification +# =========================================================================== + +Find TDD Test In Feature File + [Documentation] Verify the script finds @tdd_bug_N in .feature files + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} find_feature_test + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} find-feature-test-ok + +Find TDD Test In Robot File + [Documentation] Verify the script finds tdd_bug_N in .robot files + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} find_robot_test + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} find-robot-test-ok + +Find TDD Test Uses Exact Tag Match + [Documentation] Verify partial tags are not matched as exact bug tags + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} find_exact_tag_match + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} find-exact-tag-match-ok + +No TDD Test Found Fails Gate + [Documentation] Verify the gate fails when no TDD test exists + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} no_tdd_test_fails + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} no-tdd-test-fails-ok + +Expected Fail Tag Still Present Fails Gate + [Documentation] Verify the gate fails when @tdd_expected_fail is still present + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} expected_fail_present + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} expected-fail-present-ok + +Expected Fail Tag Removed Passes Gate + [Documentation] Verify the gate passes when @tdd_expected_fail is removed + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} expected_fail_removed + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} expected-fail-removed-ok + +# =========================================================================== +# Full quality gate integration +# =========================================================================== + +Full Gate Multiple Bugs Mixed Results + [Documentation] Verify the gate handles multiple bugs with mixed outcomes + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} multi_bug_mixed + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} multi-bug-mixed-ok + +Full Gate All Clean Passes + [Documentation] Verify the gate passes when all referenced bugs have clean tests + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} all_clean_passes + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} all-clean-passes-ok + +TDD Tests In Both Behave And Robot + [Documentation] Verify the gate checks tests in both .feature and .robot files + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} both_behave_and_robot + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} both-behave-and-robot-ok + +Diff Removal Is Required + [Documentation] Verify the gate fails when PR diff has no expected-fail removal + [Tags] ci quality tdd + ${result}= Run Process ${PYTHON} ${HELPER} diff_removal_required + ... cwd=${WORKSPACE} + Should Be Equal As Integers ${result.rc} 0 msg=${result.stderr} + Should Contain ${result.stdout} diff-removal-required-ok diff --git a/scripts/tdd_quality_gate.py b/scripts/tdd_quality_gate.py new file mode 100644 index 000000000..43fb0a172 --- /dev/null +++ b/scripts/tdd_quality_gate.py @@ -0,0 +1,402 @@ +#!/usr/bin/env python3 +"""TDD bug tag quality gate for bug fix PRs. + +Enforces the TDD bug fix workflow rules described in CONTRIBUTING.md: + +1. Parses the PR description for closing keywords that reference bug issues + (``Closes #N``, ``Fixes #N``, ``Resolves #N``, ``ISSUES CLOSED: #N``). +2. Searches the codebase for tests tagged ``@tdd_bug_N`` (Behave ``.feature`` + files) or ``tdd_bug_N`` (Robot ``.robot`` files). +3. Verifies that every such test has had its ``@tdd_expected_fail`` / + ``tdd_expected_fail`` tag removed — the fix PR must remove the + expected-fail marker as proof the bug is now fixed. + +Exit codes: + 0 — All checks passed (or PR references no bugs). + 1 — One or more violations detected. + +Usage: + PR_DESCRIPTION="Fixes #42" python scripts/tdd_quality_gate.py + +Or via nox:: + + PR_DESCRIPTION="Fixes #42" nox -s tdd_quality_gate +""" + +from __future__ import annotations + +import functools +import os +import re +import subprocess +import sys +from pathlib import Path + +# --------------------------------------------------------------------------- +# PR description parsing +# --------------------------------------------------------------------------- + +# Matches: Closes #N, Fixes #N, Resolves #N (case-insensitive) +_CLOSING_KEYWORD_RE = re.compile( + r"\b(?:close[sd]?|fix(?:e[sd])?|resolve[sd]?)\b\s+#(\d+)", + re.IGNORECASE, +) + +# Matches: ISSUES CLOSED: #N, #M, ... +_ISSUES_CLOSED_RE = re.compile( + r"ISSUES\s+CLOSED\s*:\s*((?:#\d+[\s,]*)+)", + re.IGNORECASE, +) + +# Extracts individual issue numbers from the ISSUES CLOSED value +_ISSUE_NUMBER_RE = re.compile(r"#(\d+)") + + +@functools.lru_cache(maxsize=64) +def _tag_token_pattern(tag: str) -> re.Pattern[str]: + """Return a compiled regex that matches ``tag`` as a full token.""" + escaped = re.escape(tag) + return re.compile(rf"(? bool: + """Return True when ``tag`` appears as a full token in ``content``.""" + return _tag_token_pattern(tag).search(content) is not None + + +def _collect_pr_diff(search_root: Path, base_ref: str) -> str: + """Collect unified diff between the PR branch and the base branch.""" + if not isinstance(search_root, Path): + raise TypeError(f"search_root must be a Path, got {type(search_root).__name__}") + if not isinstance(base_ref, str): + raise TypeError(f"base_ref must be a str, got {type(base_ref).__name__}") + if base_ref.strip() == "": + raise ValueError("base_ref must not be empty") + + ranges = (f"origin/{base_ref}...HEAD", f"{base_ref}...HEAD") + for ref_range in ranges: + try: + proc = subprocess.run( + [ + "git", + "diff", + "--no-color", + ref_range, + "--", + "*.feature", + "*.robot", + ], + cwd=search_root, + check=True, + capture_output=True, + text=True, + ) + return proc.stdout + except (OSError, subprocess.CalledProcessError): + continue + + raise RuntimeError( + "Unable to compute PR diff against base branch. " + "Ensure git history for the base branch is available and retry." + ) + + +def _diff_has_expected_fail_removal_for_bug(pr_diff: str, bug_number: int) -> bool: + """Return True when PR diff removes expected-fail for ``bug_number``.""" + if not isinstance(pr_diff, str): + raise TypeError(f"pr_diff must be a str, got {type(pr_diff).__name__}") + if ( + isinstance(bug_number, bool) + or not isinstance(bug_number, int) + or bug_number < 1 + ): + raise ValueError(f"bug_number must be a positive integer, got {bug_number!r}") + + current_suffix = "" + in_hunk = False + file_has_bug_tag = False + file_removed_expected_fail = False + + for line in pr_diff.splitlines(): + if line.startswith("+++ "): + if file_has_bug_tag and file_removed_expected_fail: + return True + raw_path = line[4:] + if raw_path.startswith("b/"): + raw_path = raw_path[2:] + current_suffix = Path(raw_path).suffix.lower() + in_hunk = False + file_has_bug_tag = False + file_removed_expected_fail = False + continue + + if current_suffix not in {".feature", ".robot"}: + continue + + if line.startswith("@@"): + in_hunk = True + continue + + if not in_hunk: + continue + + if not line or line[0] not in {" ", "+", "-"}: + continue + + content = line[1:] + if current_suffix == ".feature": + bug_tag = f"@tdd_bug_{bug_number}" + expected_fail_tag = "@tdd_expected_fail" + else: + bug_tag = f"tdd_bug_{bug_number}" + expected_fail_tag = "tdd_expected_fail" + + if _contains_tag_token(content, bug_tag): + file_has_bug_tag = True + if ( + line[0] == "-" + and _contains_tag_token(content, expected_fail_tag) + and _contains_tag_token(content, bug_tag) + ): + file_removed_expected_fail = True + + return file_has_bug_tag and file_removed_expected_fail + + +def parse_bug_refs(pr_description: str) -> list[int]: + """Extract bug issue numbers from PR closing keywords. + + Recognises ``Closes #N``, ``Fixes #N``, ``Resolves #N`` + (case-insensitive) and ``ISSUES CLOSED: #N, #M``. + + Returns a deduplicated, sorted list of issue numbers. + """ + if not isinstance(pr_description, str): + raise TypeError( + f"pr_description must be a str, got {type(pr_description).__name__}" + ) + + refs: set[int] = set() + + # Standard closing keywords + for match in _CLOSING_KEYWORD_RE.finditer(pr_description): + num = int(match.group(1)) + if num > 0: + refs.add(num) + + # ISSUES CLOSED: #N, #M block + for block_match in _ISSUES_CLOSED_RE.finditer(pr_description): + block = block_match.group(1) + for num_match in _ISSUE_NUMBER_RE.finditer(block): + num = int(num_match.group(1)) + if num > 0: + refs.add(num) + + return sorted(refs) + + +# --------------------------------------------------------------------------- +# TDD test search +# --------------------------------------------------------------------------- + + +def find_tdd_tests( + bug_number: int, + search_root: Path, +) -> list[Path]: + """Find test files tagged with ``@tdd_bug_``. + + Searches ``.feature`` files for ``@tdd_bug_`` and ``.robot`` + files for ``tdd_bug_``. + + Returns a list of paths that contain the tag. + """ + if ( + isinstance(bug_number, bool) + or not isinstance(bug_number, int) + or bug_number < 1 + ): + raise ValueError(f"bug_number must be a positive integer, got {bug_number!r}") + if not isinstance(search_root, Path): + raise TypeError(f"search_root must be a Path, got {type(search_root).__name__}") + + tag_behave = f"@tdd_bug_{bug_number}" + tag_robot = f"tdd_bug_{bug_number}" + matches: list[Path] = [] + + # Search .feature files + for feature_file in sorted(search_root.rglob("*.feature")): + try: + content = feature_file.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError): + continue + if _contains_tag_token(content, tag_behave): + matches.append(feature_file) + + # Search .robot files + for robot_file in sorted(search_root.rglob("*.robot")): + try: + content = robot_file.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError): + continue + if _contains_tag_token(content, tag_robot): + matches.append(robot_file) + + return matches + + +# --------------------------------------------------------------------------- +# Tag removal verification +# --------------------------------------------------------------------------- + + +def check_expected_fail_removed( + test_files: list[Path], + bug_number: int, +) -> list[str]: + """Verify ``@tdd_expected_fail`` has been removed from test files. + + Returns a list of error messages for files that still contain the + expected-fail tag. + """ + if not isinstance(test_files, list): + raise TypeError(f"test_files must be a list, got {type(test_files).__name__}") + if ( + isinstance(bug_number, bool) + or not isinstance(bug_number, int) + or bug_number < 1 + ): + raise ValueError(f"bug_number must be a positive integer, got {bug_number!r}") + + errors: list[str] = [] + + for path in test_files: + try: + content = path.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError): + continue + + suffix = path.suffix.lower() + if suffix == ".feature": + tag = "@tdd_expected_fail" + elif suffix == ".robot": + tag = "tdd_expected_fail" + else: + continue + + if _contains_tag_token(content, tag): + bug_tag_display = ( + f"@tdd_bug_{bug_number}" + if suffix == ".feature" + else f"tdd_bug_{bug_number}" + ) + errors.append( + f"Bug fix PR must remove the {tag} tag from tests tagged " + f"{bug_tag_display}. " + f"See CONTRIBUTING.md > Bug Fix Workflow." + ) + + return errors + + +# --------------------------------------------------------------------------- +# Main gate logic +# --------------------------------------------------------------------------- + + +def run_quality_gate( + pr_description: str, + search_root: Path, + *, + pr_diff: str | None = None, + base_ref: str = "master", +) -> tuple[list[str], list[int]]: + """Run the full TDD quality gate. + + Returns a ``(errors, bug_refs)`` tuple. An empty error list means all + checks passed. ``bug_refs`` contains the parsed bug issue numbers so + callers can avoid re-parsing the PR description. + """ + if not isinstance(pr_description, str): + raise TypeError( + f"pr_description must be a str, got {type(pr_description).__name__}" + ) + if not isinstance(search_root, Path): + raise TypeError(f"search_root must be a Path, got {type(search_root).__name__}") + if pr_diff is not None and not isinstance(pr_diff, str): + raise TypeError(f"pr_diff must be a str or None, got {type(pr_diff).__name__}") + if not isinstance(base_ref, str): + raise TypeError(f"base_ref must be a str, got {type(base_ref).__name__}") + if base_ref.strip() == "": + raise ValueError("base_ref must not be empty") + + bug_refs = parse_bug_refs(pr_description) + if not bug_refs: + return [], bug_refs + + if pr_diff is None: + try: + pr_diff = _collect_pr_diff(search_root, base_ref) + except RuntimeError as exc: + return [str(exc)], bug_refs + + all_errors: list[str] = [] + + for bug_num in bug_refs: + test_files = find_tdd_tests(bug_num, search_root) + + if not test_files: + all_errors.append( + f"No TDD test found for bug #{bug_num}. " + f"The TDD workflow requires a test tagged @tdd_bug_{bug_num} " + f"to exist before the bug can be fixed. " + f"See CONTRIBUTING.md > Bug Fix Workflow." + ) + continue + + removal_errors = check_expected_fail_removed(test_files, bug_num) + all_errors.extend(removal_errors) + + # Only check the diff when the file-level check found no tag issues; + # otherwise the diff error would be redundant. + if not removal_errors and not _diff_has_expected_fail_removal_for_bug( + pr_diff, bug_num + ): + all_errors.append( + "No removal of @tdd_expected_fail / tdd_expected_fail detected " + f"in PR diff for bug #{bug_num}. " + "The bug fix PR must remove the expected-fail tag in this branch. " + "See CONTRIBUTING.md > Bug Fix Workflow." + ) + + return all_errors, bug_refs + + +# --------------------------------------------------------------------------- +# CLI entry point +# --------------------------------------------------------------------------- + + +def main() -> int: + """CLI entry point. Reads PR_DESCRIPTION from the environment.""" + pr_description = os.environ.get("PR_DESCRIPTION", "") + base_ref = os.environ.get("PR_BASE_REF", "master") + search_root = Path.cwd() + + errors, bug_refs = run_quality_gate(pr_description, search_root, base_ref=base_ref) + + if errors: + for error in errors: + print(f"ERROR: {error}", file=sys.stderr) + return 1 + + if bug_refs: + print(f"TDD quality gate passed for bug(s): {bug_refs}") + else: + print("TDD quality gate: no bug references found in PR description (pass).") + + return 0 + + +if __name__ == "__main__": + sys.exit(main())