From 37895febc1648078ceec4dd93b674cba416ac9be Mon Sep 17 00:00:00 2001 From: KT <677465+kevintseng@users.noreply.github.com> Date: Thu, 1 Oct 2026 03:28:42 +0800 Subject: [PATCH 1/2] ci: shorten pull request feedback - The test step puts TEMP/TMP on the runner's temp disk. On Windows, TEMP pointed at the OS disk (C:) while the workspace and runner.temp are on D:, and the suite creates a temp folder and a SQLite database per test; the Windows legs spent 1556 s of test time against 373 s on ubuntu. Node reads TMPDIR on ubuntu and macOS, so nothing changes there. - Matrix legs no longer lint twice: verify:release already runs `npm run lint` first, in the same leg. - The ubuntu/Node 24 leg runs the suite with the coverage floor instead of a separate job running the same suite on the same OS and Node again. - The packaged dashboard e2e runs inside SDLC verify, which already ran the same command on the same OS and Node; the separate job is gone. - Tests pin that the coverage leg exists, that the coverage thresholds are not lowered, and that the verify:release, coverage, SDLC verify and temp-folder settings cannot be skipped or removed quietly. --- .github/workflows/ci.yml | 131 +++++++++------------------ scripts/audit/baseline.json | 8 +- tests/release-scripts-safety.test.ts | 71 +++++++++++++++ 3 files changed, 117 insertions(+), 93 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0cbd81ab7..ebce8267c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -43,7 +43,7 @@ on: # precisely while the branch was hardest to reason about. # # No inputs: the branch is chosen by the ref the run is dispatched on, and - # every job below is unconditional — there is no `if:` in this file — so a + # every job below is unconditional — no job has an `if:` — so a # dispatched run does exactly what a pull request run does, rather than a # quieter subset that would make a green tick mean something different. # @@ -71,55 +71,18 @@ concurrency: cancel-in-progress: true jobs: - # One leg, ubuntu/24, alongside the matrix rather than inside it: coverage - # instruments every worker and roughly doubles a run, and the numbers do not - # differ by OS enough to buy three more instrumented legs. Before this job, - # `npm run test:coverage` had ZERO automated callers — installed provider, - # measured include globs, and nothing anywhere that would ever go red. - coverage: - name: Coverage floor - runs-on: ubuntu-latest - timeout-minutes: 20 - steps: - - name: Checkout - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 - with: - persist-credentials: false - fetch-tags: true - fetch-depth: 0 - - - name: Setup Node.js - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 - with: - node-version: '24' - cache: 'npm' - - - name: Install dependencies - run: npm ci - - # The suite spawns the built CLI, hooks and packaged binaries; without - # dist/ those tests fail for a reason that has nothing to do with - # coverage. - - name: Build - run: npm run build - - - name: Run tests with the coverage floor - run: npm run test:coverage - build-and-test: name: Build & Test (${{ matrix.os }}, Node ${{ matrix.node }}) runs-on: ${{ matrix.os }} - # 40, not 15 — measured, twice (2026-08-13, runs 31697660004 and - # 31711750310, attempt 1 of each). A degraded windows-latest runner ran - # the suite at ~5× its usual pace: steady per-file progress right up to - # the kill, no hang — 51/127 and 48/127 test files done when the - # 15-minute cap cancelled the job — which projects to ~34 minutes - # end-to-end. A healthy Windows leg takes 6–10 minutes and ubuntu ~2, so - # this cap costs nothing when runners are healthy; it exists so - # runner-lottery slowness self-recovers as a slow green instead of a - # cancelled leg a human has to notice and re-run. The suite is fully - # serial (shared SQLite HOME) and spawn-heavy, which is why Windows - # magnifies runner variance this much. + # 40 is a bound, not a target. Measured over 19 runs (2026-09-28..30): + # green ubuntu legs take 5-15 minutes, macOS 7-13, Windows 22-40. Four of + # 38 Windows legs hit this cap with no failing test (one had finished + # 223 of 286 files) — a slower runner paying the same per-test cost, not + # a hang. That cost is Windows-only: tests that create a fresh temp dir + # and SQLite database per test pay ~1 s each there and ~30 ms on ubuntu. + # Raising the cap would only move the cancellations; removing that + # per-test cost is the fix. The suite is fully serial (shared SQLite + # HOME), so nothing inside one leg can run in parallel to absorb it. timeout-minutes: 40 strategy: fail-fast: false @@ -148,10 +111,21 @@ jobs: # macOS/arm64 and needs a full toolchain — MSVC on Windows. One # Linux leg gives early warning that Node 26 works at all # without paying for a source build on three runners. + # + # ubuntu/24 also enforces the coverage floor. Coverage runs the same + # suite with v8 instrumentation and thresholds, and the numbers do not + # differ by OS enough to instrument more than one leg. It used to be a + # separate job on this same OS and Node, running the whole suite a + # second time; the first `include` entry adds `coverage` to the existing + # ubuntu/24 leg (it creates no new leg), so the suite runs once there. + # An include entry matches only the original os/node values, so it cannot attach to the Node 26 leg. matrix: os: [ubuntu-latest, macos-latest, windows-latest] node: ['22', '24'] include: + - os: ubuntu-latest + node: '24' + coverage: true - os: ubuntu-latest node: '26' @@ -182,13 +156,6 @@ jobs: - name: Install dependencies run: npm ci - - name: Lint (hard gate — max-warnings 0) - # Run BEFORE typecheck because lint is significantly faster and - # most regressions land here first. Fail-fast cuts CI minutes - # on bad PRs. Hard-gated at zero warnings since v4.2.0; new - # `eslint-disable` comments require explicit justification. - run: npm run lint - - name: Build source and generated artifacts run: npm run build @@ -208,12 +175,22 @@ jobs: # publish path again. Must come after Build — the mirror gate diffs what # the build regenerates. # - # Lint stays as its own step above for fail-fast; verify:release runs it - # again for ~3s, which is the cost of having one authoritative list. + # Its first command is `npm run lint` (zero warnings), so lint is not a + # separate step: that ran the same script twice in every leg. run: npm run verify:release - - name: Run tests - run: npm test -- --run + - name: Run tests${{ matrix.coverage && ' with the coverage floor' || '' }} + # Temp folders on the runner's own temp disk. On Windows, TEMP points + # at C:\Users\runneradmin\AppData\Local\Temp, on the OS disk, while + # the workspace and runner.temp are on D:. The suite creates a temp + # folder and a SQLite database per test, and there the Windows legs + # spent 1556 s of test time against 373 s on ubuntu (run 36757693602). + # Node's os.tmpdir() reads TEMP/TMP on Windows and TMPDIR elsewhere, + # so this changes nothing on ubuntu or macOS. + env: + TEMP: ${{ runner.temp }} + TMP: ${{ runner.temp }} + run: ${{ matrix.coverage && 'npm run test:coverage' || 'npm test -- --run' }} - name: Doctor (manifest + hooks integrity gate) # Must run AFTER the explicit build so @@ -253,7 +230,7 @@ jobs: exit 1 fi - # The three jobs below used to carry `needs: build-and-test`, which made each + # The jobs below used to carry `needs: build-and-test`, which made each # of them wait for ALL eleven matrix legs before starting. Measured on run # 30735734852: the matrix finished at 06:28:38, these started at 06:28:40, and # the run ended at 06:30:46 — a 2m06s serial tail on a 10m15s run, 20% of every @@ -265,7 +242,7 @@ jobs: # and on a public repo that trade is backwards: GitHub reports # `billable.duration_ms = 0` for every leg of this workflow, so the currency is # wall-clock feedback time, and this was paying 2 minutes of it on every - # PASSING run to save three ubuntu jobs on the rarer failing one. + # PASSING run to save the ubuntu jobs on the rarer failing one. package-smoke: name: Packaged Artifact Smoke Test runs-on: ubuntu-latest @@ -299,35 +276,6 @@ jobs: - name: Verify every derived packaged upgrade path run: npm run test:packaged:upgrade - dashboard-e2e-smoke: - name: Packaged Dashboard E2E Smoke - runs-on: ubuntu-latest - timeout-minutes: 15 - - steps: - - name: Checkout - uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4 - with: - persist-credentials: false - - - name: Setup Node.js - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 - with: - node-version: '22' - cache: 'npm' - - - name: Install dependencies - run: npm ci - - - name: Install Playwright Chromium - run: npx playwright install --with-deps chromium - - - name: Build - run: npm run build - - - name: Run packaged dashboard e2e smoke - run: npm run test:e2e-dashboard - # The "verify before publish" gate (per user directive 2026-05-08). # Catches drift that unit tests miss — e.g., installation.test.ts asserting # 5 hook types when hooks.json shipped 6 went unnoticed for a full release @@ -367,6 +315,11 @@ jobs: # hash) can therefore be compared with what CI saw for the same tree; a # receipt for a different tree is a finding. The test step before it keeps # scripts/verify.mjs and scripts/verify-receipt.mjs honest about themselves. + # + # This is the only job that runs the packaged dashboard e2e on a pull + # request. A separate job used to run the same command on the same OS and + # Node; tests/release-scripts-safety.test.ts pins that this job runs the + # full verify and that the e2e step is still in scripts/verify.config.json. sdlc-verify: name: SDLC verify runs-on: ubuntu-latest diff --git a/scripts/audit/baseline.json b/scripts/audit/baseline.json index b456e0c79..6d5f1353b 100644 --- a/scripts/audit/baseline.json +++ b/scripts/audit/baseline.json @@ -6,15 +6,15 @@ "reason": "Source-only typography and palette guard, explicitly not rendered usability evidence. Controlled old palette and inline font failures establish sensitivity; independent browser review separately exercises computed contrast, clipping and responsive layout.", "triaged": "2026-09-08" }, - "C4 .github/workflows/ci.yml:245": { + "C4 .github/workflows/ci.yml:222": { "class": "SAFE-CAPTURED-EXIT", "reason": "The doctor exit code is captured and rejected before this grep; the grep is an additional content assertion, not the verdict owner.", - "triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement)" + "triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement); re-keyed 2026-10-01 (the coverage job and the lint step above the unchanged statement were folded away); re-keyed 2026-10-01 245→222 (the matrix comments and steps above reworked to shorten pull request feedback; statement byte-identical)" }, - "C4 .github/workflows/ci.yml:251": { + "C4 .github/workflows/ci.yml:228": { "class": "SAFE-CAPTURED-EXIT", "reason": "The doctor exit code is captured and rejected first; this grep independently checks the skills and hooks content.", - "triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement)" + "triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement); re-keyed 2026-10-01 (the coverage job and the lint step above the unchanged statement were folded away); re-keyed 2026-10-01 251→228 (the matrix comments and steps above reworked to shorten pull request feedback; statement byte-identical)" }, "C4 scripts/release-verify.sh:106": { "class": "SAFE-DIAGNOSTIC", diff --git a/tests/release-scripts-safety.test.ts b/tests/release-scripts-safety.test.ts index 52a59b69e..30eb395a2 100644 --- a/tests/release-scripts-safety.test.ts +++ b/tests/release-scripts-safety.test.ts @@ -362,6 +362,77 @@ describe('Feature: release scripts never edit the real ~/.memesh', () => { expect(releaseJob).not.toContain('--quick'); }); + // The coverage floor and the packaged dashboard e2e each used to be their + // own job, and their required check names were what guaranteed they ran on + // every pull request. Both now live inside other jobs, so these two tests + // are what notices if either one quietly stops running. + function ciJob(id: string): string { + const ci = read('.github/workflows/ci.yml'); + return ci.match(new RegExp(`\\n {2}${id}:\\n[\\s\\S]*?(?=\\n {2}[A-Za-z0-9_-]+:\\n|$)`))?.[0] ?? ''; + } + // The one step of a job whose `run:` line is exactly `run`. A step-level + // `if:` or `continue-on-error` turns it into a green step that proved nothing. + function unconditionalStep(job: string, run: string): string { + const step = job.split(/\n {6}- /).find((s) => s.split('\n').some((line) => line.trim() === `run: ${run}`)) ?? ''; + expect(step, `no step runs exactly: ${run}`).not.toBe(''); + expect(step).not.toMatch(/\n\s+if:/); + expect(step).not.toContain('continue-on-error'); + return step; + } + + it('enforces the coverage floor in the ubuntu/Node 24 matrix leg, unconditionally', () => { + const buildJob = ciJob('build-and-test'); + expect(buildJob).not.toBe(''); + // A job-level `if:` can skip the whole leg, and GitHub counts a skipped + // job as a passing check. + expect(buildJob).not.toMatch(/\n {4}if:/); + expect(buildJob).not.toContain('continue-on-error'); + // Adds `coverage` to the existing ubuntu/24 combination; it is not a new leg. + expect(buildJob).toMatch(/- os: ubuntu-latest\n\s+node: '24'\n\s+coverage: true\n/); + unconditionalStep(buildJob, "${{ matrix.coverage && 'npm run test:coverage' || 'npm test -- --run' }}"); + const pkg = JSON.parse(read('package.json')) as { scripts: Record }; + expect(pkg.scripts['test:coverage']).toBe('node scripts/run-tests-isolated.mjs --coverage'); + // The leg itself must exist: an `exclude:` would drop it without an `if:`. + expect(buildJob).not.toMatch(/\n\s+exclude:/); + // The floor must not be lowered in passing (raising it is fine). + const thresholds = read('vitest.config.ts').match(/thresholds:\s*\{([^}]*)\}/)?.[1] ?? ''; + for (const [metric, floor] of [['statements', 52], ['branches', 48], ['functions', 55], ['lines', 54]] as const) { + expect(Number(thresholds.match(new RegExp(`${metric}:\\s*(\\d+)`))?.[1] ?? 0), metric).toBeGreaterThanOrEqual(floor); + } + // Lint runs inside verify:release; this step is its only home in the leg. + unconditionalStep(buildJob, 'npm run verify:release'); + }); + + it('puts the test temp folders on the runner temp disk, on every leg', () => { + const step = ciJob('build-and-test').split(/\n {6}- /).find((s) => s.includes('npm test -- --run')) ?? ''; + expect(step, 'no test step').not.toBe(''); + expect(step).toMatch(/\n\s+TEMP: \$\{\{ runner\.temp \}\}\n/); + expect(step).toMatch(/\n\s+TMP: \$\{\{ runner\.temp \}\}\n/); + }); + + it('runs the packaged dashboard e2e on every pull request, inside SDLC verify', () => { + const sdlcJob = ciJob('sdlc-verify'); + expect(sdlcJob).not.toBe(''); + expect(sdlcJob).not.toMatch(/\n {4}if:/); + expect(sdlcJob).not.toMatch(/\n {4}continue-on-error/); + unconditionalStep(sdlcJob, 'npx playwright install --with-deps chromium'); + // The full run, not `--journeys`: the full run is the one with the suite. + unconditionalStep(sdlcJob, 'node scripts/verify.mjs'); + const config = JSON.parse(read('scripts/verify.config.json')) as { + verify: { steps: Array<{ command: string; args?: string[] }> }; + }; + const commands = config.verify.steps.map((step) => [step.command, ...(step.args ?? [])].join(' ')); + expect(commands).toEqual( + expect.arrayContaining([ + 'npm run build', + 'npm run verify:release', + 'node scripts/run-tests-isolated.mjs', + 'npm run test:packaged', + 'npm run test:e2e-dashboard', + ]), + ); + }); + it('fails the sync gate when an installed adapter artifact is missing', () => { const root = fs.mkdtempSync(path.join(os.tmpdir(), 'message-sync-fixture-')); const write = (relative: string, content = '') => { From 08af0f353dcf933ab7297395c2eea476bdf44b2a Mon Sep 17 00:00:00 2001 From: KT <677465+kevintseng@users.noreply.github.com> Date: Thu, 1 Oct 2026 03:59:22 +0800 Subject: [PATCH 2/2] ci: say what the temp setting does off Windows, and pin lint and Windows - The comment on the test step's TEMP/TMP now says that ubuntu and macOS read TMPDIR first, and that where it is unset temp moves to runner.temp. - Tests pin that verify:release still starts with `npm run lint`, and that the matrix still includes windows-latest. --- .github/workflows/ci.yml | 4 ++-- tests/release-scripts-safety.test.ts | 3 +++ 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ebce8267c..36d955357 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -185,8 +185,8 @@ jobs: # the workspace and runner.temp are on D:. The suite creates a temp # folder and a SQLite database per test, and there the Windows legs # spent 1556 s of test time against 373 s on ubuntu (run 36757693602). - # Node's os.tmpdir() reads TEMP/TMP on Windows and TMPDIR elsewhere, - # so this changes nothing on ubuntu or macOS. + # On ubuntu and macOS Node reads TMPDIR first; where that is unset, + # temp moves to runner.temp too, on the workspace's disk (harmless). env: TEMP: ${{ runner.temp }} TMP: ${{ runner.temp }} diff --git a/tests/release-scripts-safety.test.ts b/tests/release-scripts-safety.test.ts index 30eb395a2..cfff1f4b5 100644 --- a/tests/release-scripts-safety.test.ts +++ b/tests/release-scripts-safety.test.ts @@ -401,6 +401,9 @@ describe('Feature: release scripts never edit the real ~/.memesh', () => { } // Lint runs inside verify:release; this step is its only home in the leg. unconditionalStep(buildJob, 'npm run verify:release'); + expect(pkg.scripts['verify:release']).toMatch(/^npm run lint && /); + // Windows stays in the matrix: the legs the temp-disk step is for. + expect(buildJob).toMatch(/\n\s+os: \[[^\]]*\bwindows-latest\b[^\]]*\]/); }); it('puts the test temp folders on the runner temp disk, on every leg', () => {