From f10fcd7d19bad75bc671ff357c36985502e2d0f9 Mon Sep 17 00:00:00 2001 From: Danny Avila Date: Mon, 24 Aug 2026 09:04:24 -0400 Subject: [PATCH] =?UTF-8?q?=F0=9F=8E=B0=20ci:=20Vote=20on=20the=20Full=20M?= =?UTF-8?q?ock=20Suite=20to=20End=20Phantom=20Spec=20Trials=20(#15162)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * ci: votes run the full mock suite; covered list from Playwright's own discovery The vote workflow passed the merged PR's skippable tier as CLI path filters, but playwright.config.mock.ts scopes discovery to testDir specs/mock/ — tier entries outside that directory matched nothing, and the covered-list log line still claimed them. Run 32701691037 proves it: a11y/keys/messages in the covered list, zero of their tests executed, '120 passed' all from specs/mock/. The graduation ledger was minting clean trials for specs that never ran. Now every dev push runs the full mock suite (no path filters to mismatch), the covered list is derived from playwright --list --reporter=json (git-enumeration fallback over the same testDir), and each merge is one trial for every pool spec — ~4x faster accrual toward the pre-registered graduation bars, plus the post-merge Playwright safety net the jest workflows already have. Timeout 30->45 for the wider run; newest merge still cancels older votes; observe-only, continue-on-error, kill switch CODEGRAPH_E2E_VOTES unchanged. * ci: covered list from executed results, not discovery (Codex P1) Env-gated suites (mcp-tool-list-changed needs E2E_MCP_LIST_CHANGED, enforced- model-specs needs E2E_MODEL_SPECS_ENFORCE) are discovered by --list yet skip every test under the vote job's default env — counting them as covered would mint phantom trials, the exact class this PR exists to kill. The run now emits line+json reporters and the ledger step derives covered from specs with at least one non-skipped test outcome; no results json means no trials logged. Verified against a synthetic suite: gated spec excluded, nested dirs handled, crash branch logs nothing. --- .github/workflows/codegraph-e2e-votes.yml | 125 +++++++++------------- 1 file changed, 51 insertions(+), 74 deletions(-) diff --git a/.github/workflows/codegraph-e2e-votes.yml b/.github/workflows/codegraph-e2e-votes.yml index fdcbe79626..35637a3041 100644 --- a/.github/workflows/codegraph-e2e-votes.yml +++ b/.github/workflows/codegraph-e2e-votes.yml @@ -1,15 +1,22 @@ # Codegraph e2e VOTES — observe-only, post-merge, time-boxed. # # Playwright never runs on pushes to dev, so evidence for the e2e skip election would -# otherwise wait on rare organic PR spec failures. This workflow runs EXACTLY the skippable -# tier the merged PR's selection computed — every merge becomes a direct trial of "would -# skipping these specs have missed a failure". A green run is a confirmation vote; a failing -# spec here is a tier-miss vote counted AGAINST enabling skipping. The shadow evaluator on -# the codegraph droplet harvests these runs and attributes them back to the merged PR. +# otherwise wait on rare organic PR spec failures. This workflow runs the FULL mock suite on +# every merge: each run is one graduation trial for every spec it executes, and doubles as +# the post-merge safety net the jest workflows already have via their dev-push triggers. # -# It cannot fail the branch: the tier lookup exits 0 on every path and the test step is -# continue-on-error. The newest merge cancels older vote runs. The whole campaign switches -# off by setting repo variable CODEGRAPH_E2E_VOTES=off once the election passes. +# It previously ran only the merged PR's skippable tier, passing the tier as CLI path +# filters. playwright.config.mock.ts scopes discovery to testDir specs/mock/, so tier +# entries outside that directory matched nothing — and the covered-list log line still +# claimed them, minting graduation trials for specs that never executed (run 32701691037: +# a11y/keys/messages in the covered list, zero of their tests run). The covered list below +# is therefore derived from the run's EXECUTED results — discovery is not enough either, +# since env-gated suites self-skip under this job's default env — and the run takes no +# path filters at all. +# +# It cannot fail the branch: the test step is continue-on-error. The newest merge cancels +# older vote runs. The whole campaign switches off by setting repo variable +# CODEGRAPH_E2E_VOTES=off once the election passes. name: Codegraph E2E Votes on: @@ -20,6 +27,7 @@ on: - '**' - '!**.md' - '!.github/workflows/**' + - '.github/workflows/codegraph-e2e-votes.yml' permissions: contents: read @@ -34,10 +42,10 @@ env: jobs: vote: - name: vote (skippable tier) + name: vote (full suite) if: vars.CODEGRAPH_E2E_VOTES != 'off' runs-on: ubuntu-latest - timeout-minutes: 30 + timeout-minutes: 45 env: CI: 'true' E2E_CHROMIUM_CHANNEL: chrome @@ -45,56 +53,12 @@ jobs: steps: - uses: actions/checkout@v5 - - name: Ask codegraph for this merge's skippable tier - id: tiers - env: - URL: ${{ secrets.CODEGRAPH_URL }} - TOKEN: ${{ secrets.CODEGRAPH_TOKEN }} - GH_TOKEN: ${{ github.token }} - run: | - set +e - N=0 - if [ -n "$URL" ] && [ -n "$TOKEN" ]; then - gh api "repos/${{ github.repository }}/commits/${{ github.sha }}" \ - --jq '[.files[] | {path: .filename, status: .status}]' > files.json 2>/dev/null - if [ -s files.json ]; then - jq -c '{files: .}' files.json > body.json - RESP=$(curl -sS -m 45 -H "Authorization: Bearer $TOKEN" \ - -H 'content-type: application/json' --data-binary @body.json "$URL/v1/select") - # fail_open reflects the JEST floors (root config, lockfile, stale graph); the - # e2e tiers come from the testid bridge and are valid whenever they computed at - # all. The old fail-open skip silently excused exactly the big backend merges - # whose trials matter most (LibreChat#14957's merge produced no vote because - # api/package.json tripped the jest floor). Tiers present => vote. - if ! echo "$RESP" | jq -e '.e2e.skippable' >/dev/null 2>&1; then - echo "codegraph unavailable or no tiers; skipping" - else - echo "$RESP" | jq -r '.e2e.skippable[]' | sed 's|^e2e/||' > skippable.txt - N=$(wc -l < skippable.txt | tr -d ' ') - fi - else - echo "could not read merge commit files; skipping" - fi - else - echo "no codegraph config; skipping" - fi - echo "codegraph-votes: running $N skippable specs" - # The exact list, one log line: the shadow's per-spec graduation ledger counts a clean - # trial for every spec a green vote run covered, and until this line existed it had to - # approximate coverage from the decision event's tier (drift: the tier is recomputed - # here at the merge commit against a possibly newer graph head). - if [ "$N" != "0" ]; then echo "codegraph-votes-specs: $(tr '\n' ' ' < skippable.txt)"; fi - echo "count=$N" >> "$GITHUB_OUTPUT" - exit 0 - - name: Use Node.js 24.16.0 - if: steps.tiers.outputs.count != '0' uses: actions/setup-node@v5 with: node-version: '24.16.0' - name: Restore node_modules cache - if: steps.tiers.outputs.count != '0' id: cache-node-modules uses: actions/cache@v5 with: @@ -109,11 +73,10 @@ jobs: key: node-modules-e2e-${{ runner.os }}-24.16.0-${{ hashFiles('package-lock.json') }} - name: Install dependencies - if: steps.tiers.outputs.count != '0' && steps.cache-node-modules.outputs.cache-hit != 'true' + if: steps.cache-node-modules.outputs.cache-hit != 'true' run: npm ci - name: Restore data-provider build cache - if: steps.tiers.outputs.count != '0' id: cache-data-provider uses: actions/cache@v5 with: @@ -121,11 +84,10 @@ jobs: key: build-data-provider-${{ runner.os }}-${{ hashFiles('package.json', 'package-lock.json', 'packages/data-provider/src/**', 'packages/data-provider/tsconfig*.json', 'packages/data-provider/tsdown.config.mjs', 'packages/data-provider/package.json') }} - name: Build data-provider - if: steps.tiers.outputs.count != '0' && steps.cache-data-provider.outputs.cache-hit != 'true' + if: steps.cache-data-provider.outputs.cache-hit != 'true' run: npm run build:data-provider - name: Restore data-schemas build cache - if: steps.tiers.outputs.count != '0' id: cache-data-schemas uses: actions/cache@v5 with: @@ -133,11 +95,10 @@ jobs: key: build-data-schemas-${{ runner.os }}-${{ hashFiles('package.json', 'package-lock.json', 'packages/data-schemas/src/**', 'packages/data-schemas/tsconfig*.json', 'packages/data-schemas/tsdown.config.mjs', 'packages/data-schemas/package.json', 'packages/data-provider/src/**', 'packages/data-provider/tsconfig*.json', 'packages/data-provider/tsdown.config.mjs', 'packages/data-provider/package.json') }} - name: Build data-schemas - if: steps.tiers.outputs.count != '0' && steps.cache-data-schemas.outputs.cache-hit != 'true' + if: steps.cache-data-schemas.outputs.cache-hit != 'true' run: npm run build:data-schemas - name: Restore api build cache - if: steps.tiers.outputs.count != '0' id: cache-api uses: actions/cache@v5 with: @@ -145,11 +106,10 @@ jobs: key: build-api-${{ runner.os }}-${{ hashFiles('package.json', 'package-lock.json', 'packages/api/src/**', 'packages/api/tsconfig*.json', 'packages/api/tsdown.config.mjs', 'packages/api/package.json', 'packages/data-provider/src/**', 'packages/data-provider/tsconfig*.json', 'packages/data-provider/tsdown.config.mjs', 'packages/data-provider/package.json', 'packages/data-schemas/src/**', 'packages/data-schemas/tsconfig*.json', 'packages/data-schemas/tsdown.config.mjs', 'packages/data-schemas/package.json') }} - name: Build api - if: steps.tiers.outputs.count != '0' && steps.cache-api.outputs.cache-hit != 'true' + if: steps.cache-api.outputs.cache-hit != 'true' run: npm run build:api - name: Restore client-package build cache - if: steps.tiers.outputs.count != '0' id: cache-client-package uses: actions/cache@v5 with: @@ -157,11 +117,10 @@ jobs: key: build-client-package-${{ runner.os }}-${{ hashFiles('package.json', 'package-lock.json', 'packages/client/src/**', 'packages/client/tsconfig*.json', 'packages/client/tsdown.config.mjs', 'packages/client/package.json', 'packages/data-provider/src/**', 'packages/data-provider/tsconfig*.json', 'packages/data-provider/tsdown.config.mjs', 'packages/data-provider/package.json') }} - name: Build client-package - if: steps.tiers.outputs.count != '0' && steps.cache-client-package.outputs.cache-hit != 'true' + if: steps.cache-client-package.outputs.cache-hit != 'true' run: npm run build:client-package - name: Restore client app build cache - if: steps.tiers.outputs.count != '0' id: cache-client-app uses: actions/cache@v5 with: @@ -169,24 +128,21 @@ jobs: key: build-client-app-e2e-${{ runner.os }}-${{ hashFiles('package.json', 'package-lock.json', 'client/src/**', 'client/public/**', 'client/scripts/post-build.cjs', 'client/index.html', 'client/package.json', 'client/vite.config.*', 'client/tsconfig*.json', 'client/tailwind.config.*', 'client/postcss.config.*', 'packages/client/src/**', 'packages/client/tailwind.preset.cjs', 'packages/client/tsconfig*.json', 'packages/client/tsdown.config.mjs', 'packages/client/package.json', 'packages/data-provider/src/**', 'packages/data-provider/tsconfig*.json', 'packages/data-provider/tsdown.config.mjs', 'packages/data-provider/package.json') }} - name: Build client app - if: steps.tiers.outputs.count != '0' && steps.cache-client-app.outputs.cache-hit != 'true' + if: steps.cache-client-app.outputs.cache-hit != 'true' run: npm run build:client - name: Verify Chrome is present - if: steps.tiers.outputs.count != '0' run: google-chrome --version # ffmpeg for retry video — see the note in playwright-mock.yml. - name: Resolve Playwright version id: playwright-version - if: steps.tiers.outputs.count != '0' run: | version=$(node -p "require('./package-lock.json').packages['node_modules/playwright-core'].version") echo "version=${version}" >> "$GITHUB_OUTPUT" - name: Restore Playwright ffmpeg cache id: cache-ffmpeg - if: steps.tiers.outputs.count != '0' uses: actions/cache/restore@v5 with: path: ~/.cache/ms-playwright @@ -194,7 +150,7 @@ jobs: - name: Install Playwright ffmpeg (best effort) id: install-ffmpeg - if: steps.tiers.outputs.count != '0' && steps.cache-ffmpeg.outputs.cache-hit != 'true' + if: steps.cache-ffmpeg.outputs.cache-hit != 'true' timeout-minutes: 3 continue-on-error: true run: | @@ -202,7 +158,7 @@ jobs: .github/scripts/verify-playwright-ffmpeg.sh - name: Save Playwright ffmpeg cache - if: steps.tiers.outputs.count != '0' && steps.install-ffmpeg.outcome == 'success' + if: steps.install-ffmpeg.outcome == 'success' continue-on-error: true uses: actions/cache/save@v5 with: @@ -211,15 +167,36 @@ jobs: # Optional fonts only — see the note in playwright-mock.yml. - name: Install optional Playwright font dependencies (best effort) - if: steps.tiers.outputs.count != '0' timeout-minutes: 4 continue-on-error: true run: .github/scripts/install-playwright-fonts.sh - - name: Vote — run the skippable tier (cannot fail the branch) - if: steps.tiers.outputs.count != '0' + - name: Vote — run the full mock suite (cannot fail the branch) continue-on-error: true - run: npx playwright test --config=e2e/playwright.config.mock.ts $(tr '\n' ' ' < skippable.txt) + env: + PLAYWRIGHT_JSON_OUTPUT_NAME: pw-results.json + run: npx playwright test --config=e2e/playwright.config.mock.ts --reporter=line,json + + - name: Ledger — log the specs that actually executed + run: | + set +e + # The shadow's per-spec graduation ledger counts a clean trial for every spec a green + # run covered, so the covered list must come from EXECUTED tests, not from discovery: + # env-gated suites (mcp-tool-list-changed needs E2E_MCP_LIST_CHANGED, enforced-model- + # specs needs E2E_MODEL_SPECS_ENFORCE) are discovered by --list yet skip every test + # under this job's default env — counting them as covered would mint phantom trials, + # the exact bug this workflow revision exists to kill (Codex P1 on #15162). A spec is + # covered iff at least one of its tests reached a non-skipped outcome. + if jq -e '.suites' pw-results.json >/dev/null 2>&1; then + jq -r '[.suites[] | recurse(.suites[]?) | .specs[]? | select([.tests[]?.status] | any(. != "skipped")) | .file] | unique | .[]' pw-results.json \ + | sed 's|^|specs/mock/|' > covered.txt + N=$(wc -l < covered.txt | tr -d ' ') + echo "codegraph-votes: running $N specs (executed, full suite)" + if [ "$N" != "0" ]; then echo "codegraph-votes-specs: $(tr '\n' ' ' < covered.txt)"; fi + else + echo "codegraph-votes: no results json — run crashed before reporting; no trials logged" + fi + exit 0 - name: Done if: always()