Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
131 changes: 42 additions & 89 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@ on:
# precisely while the branch was hardest to reason about.
#
# No inputs: the branch is chosen by the ref the run is dispatched on, and
# every job below is unconditional — there is no `if:` in this file — so a
# every job below is unconditional — no job has an `if:` — so a
# dispatched run does exactly what a pull request run does, rather than a
# quieter subset that would make a green tick mean something different.
#
Expand Down Expand Up @@ -71,55 +71,18 @@ concurrency:
cancel-in-progress: true

jobs:
# One leg, ubuntu/24, alongside the matrix rather than inside it: coverage
# instruments every worker and roughly doubles a run, and the numbers do not
# differ by OS enough to buy three more instrumented legs. Before this job,
# `npm run test:coverage` had ZERO automated callers — installed provider,
# measured include globs, and nothing anywhere that would ever go red.
coverage:
name: Coverage floor
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- name: Checkout
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
persist-credentials: false
fetch-tags: true
fetch-depth: 0

- name: Setup Node.js
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: '24'
cache: 'npm'

- name: Install dependencies
run: npm ci

# The suite spawns the built CLI, hooks and packaged binaries; without
# dist/ those tests fail for a reason that has nothing to do with
# coverage.
- name: Build
run: npm run build

- name: Run tests with the coverage floor
run: npm run test:coverage

build-and-test:
name: Build & Test (${{ matrix.os }}, Node ${{ matrix.node }})
runs-on: ${{ matrix.os }}
# 40, not 15 — measured, twice (2026-08-13, runs 31697660004 and
# 31711750310, attempt 1 of each). A degraded windows-latest runner ran
# the suite at ~5× its usual pace: steady per-file progress right up to
# the kill, no hang — 51/127 and 48/127 test files done when the
# 15-minute cap cancelled the job — which projects to ~34 minutes
# end-to-end. A healthy Windows leg takes 6–10 minutes and ubuntu ~2, so
# this cap costs nothing when runners are healthy; it exists so
# runner-lottery slowness self-recovers as a slow green instead of a
# cancelled leg a human has to notice and re-run. The suite is fully
# serial (shared SQLite HOME) and spawn-heavy, which is why Windows
# magnifies runner variance this much.
# 40 is a bound, not a target. Measured over 19 runs (2026-09-28..30):
# green ubuntu legs take 5-15 minutes, macOS 7-13, Windows 22-40. Four of
# 38 Windows legs hit this cap with no failing test (one had finished
# 223 of 286 files) — a slower runner paying the same per-test cost, not
# a hang. That cost is Windows-only: tests that create a fresh temp dir
# and SQLite database per test pay ~1 s each there and ~30 ms on ubuntu.
# Raising the cap would only move the cancellations; removing that
# per-test cost is the fix. The suite is fully serial (shared SQLite
# HOME), so nothing inside one leg can run in parallel to absorb it.
timeout-minutes: 40
strategy:
fail-fast: false
Expand Down Expand Up @@ -148,10 +111,21 @@ jobs:
# macOS/arm64 and needs a full toolchain — MSVC on Windows. One
# Linux leg gives early warning that Node 26 works at all
# without paying for a source build on three runners.
#
# ubuntu/24 also enforces the coverage floor. Coverage runs the same
# suite with v8 instrumentation and thresholds, and the numbers do not
# differ by OS enough to instrument more than one leg. It used to be a
# separate job on this same OS and Node, running the whole suite a
# second time; the first `include` entry adds `coverage` to the existing
# ubuntu/24 leg (it creates no new leg), so the suite runs once there.
# An include entry matches only the original os/node values, so it cannot attach to the Node 26 leg.
matrix:
os: [ubuntu-latest, macos-latest, windows-latest]
node: ['22', '24']
include:
- os: ubuntu-latest
node: '24'
coverage: true
- os: ubuntu-latest
node: '26'

Expand Down Expand Up @@ -182,13 +156,6 @@ jobs:
- name: Install dependencies
run: npm ci

- name: Lint (hard gate — max-warnings 0)
# Run BEFORE typecheck because lint is significantly faster and
# most regressions land here first. Fail-fast cuts CI minutes
# on bad PRs. Hard-gated at zero warnings since v4.2.0; new
# `eslint-disable` comments require explicit justification.
run: npm run lint

- name: Build source and generated artifacts
run: npm run build

Expand All @@ -208,12 +175,22 @@ jobs:
# publish path again. Must come after Build — the mirror gate diffs what
# the build regenerates.
#
# Lint stays as its own step above for fail-fast; verify:release runs it
# again for ~3s, which is the cost of having one authoritative list.
# Its first command is `npm run lint` (zero warnings), so lint is not a
# separate step: that ran the same script twice in every leg.
run: npm run verify:release

- name: Run tests
run: npm test -- --run
- name: Run tests${{ matrix.coverage && ' with the coverage floor' || '' }}
# Temp folders on the runner's own temp disk. On Windows, TEMP points
# at C:\Users\runneradmin\AppData\Local\Temp, on the OS disk, while
# the workspace and runner.temp are on D:. The suite creates a temp
# folder and a SQLite database per test, and there the Windows legs
# spent 1556 s of test time against 373 s on ubuntu (run 36757693602).
# On ubuntu and macOS Node reads TMPDIR first; where that is unset,
# temp moves to runner.temp too, on the workspace's disk (harmless).
env:
TEMP: ${{ runner.temp }}
TMP: ${{ runner.temp }}
run: ${{ matrix.coverage && 'npm run test:coverage' || 'npm test -- --run' }}

- name: Doctor (manifest + hooks integrity gate)
# Must run AFTER the explicit build so
Expand Down Expand Up @@ -253,7 +230,7 @@ jobs:
exit 1
fi

# The three jobs below used to carry `needs: build-and-test`, which made each
# The jobs below used to carry `needs: build-and-test`, which made each
# of them wait for ALL eleven matrix legs before starting. Measured on run
# 30735734852: the matrix finished at 06:28:38, these started at 06:28:40, and
# the run ended at 06:30:46 — a 2m06s serial tail on a 10m15s run, 20% of every
Expand All @@ -265,7 +242,7 @@ jobs:
# and on a public repo that trade is backwards: GitHub reports
# `billable.duration_ms = 0` for every leg of this workflow, so the currency is
# wall-clock feedback time, and this was paying 2 minutes of it on every
# PASSING run to save three ubuntu jobs on the rarer failing one.
# PASSING run to save the ubuntu jobs on the rarer failing one.
package-smoke:
name: Packaged Artifact Smoke Test
runs-on: ubuntu-latest
Expand Down Expand Up @@ -299,35 +276,6 @@ jobs:
- name: Verify every derived packaged upgrade path
run: npm run test:packaged:upgrade

dashboard-e2e-smoke:
name: Packaged Dashboard E2E Smoke
runs-on: ubuntu-latest
timeout-minutes: 15

steps:
- name: Checkout
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
persist-credentials: false

- name: Setup Node.js
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: '22'
cache: 'npm'

- name: Install dependencies
run: npm ci

- name: Install Playwright Chromium
run: npx playwright install --with-deps chromium

- name: Build
run: npm run build

- name: Run packaged dashboard e2e smoke
run: npm run test:e2e-dashboard

# The "verify before publish" gate (per user directive 2026-05-08).
# Catches drift that unit tests miss — e.g., installation.test.ts asserting
# 5 hook types when hooks.json shipped 6 went unnoticed for a full release
Expand Down Expand Up @@ -367,6 +315,11 @@ jobs:
# hash) can therefore be compared with what CI saw for the same tree; a
# receipt for a different tree is a finding. The test step before it keeps
# scripts/verify.mjs and scripts/verify-receipt.mjs honest about themselves.
#
# This is the only job that runs the packaged dashboard e2e on a pull
# request. A separate job used to run the same command on the same OS and
# Node; tests/release-scripts-safety.test.ts pins that this job runs the
# full verify and that the e2e step is still in scripts/verify.config.json.
sdlc-verify:
name: SDLC verify
runs-on: ubuntu-latest
Expand Down
8 changes: 4 additions & 4 deletions scripts/audit/baseline.json
Original file line number Diff line number Diff line change
Expand Up @@ -6,15 +6,15 @@
"reason": "Source-only typography and palette guard, explicitly not rendered usability evidence. Controlled old palette and inline font failures establish sensitivity; independent browser review separately exercises computed contrast, clipping and responsive layout.",
"triaged": "2026-09-08"
},
"C4 .github/workflows/ci.yml:245": {
"C4 .github/workflows/ci.yml:222": {
"class": "SAFE-CAPTURED-EXIT",
"reason": "The doctor exit code is captured and rejected before this grep; the grep is an additional content assertion, not the verdict owner.",
"triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement)"
"triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement); re-keyed 2026-10-01 (the coverage job and the lint step above the unchanged statement were folded away); re-keyed 2026-10-01 245→222 (the matrix comments and steps above reworked to shorten pull request feedback; statement byte-identical)"
},
"C4 .github/workflows/ci.yml:251": {
"C4 .github/workflows/ci.yml:228": {
"class": "SAFE-CAPTURED-EXIT",
"reason": "The doctor exit code is captured and rejected first; this grep independently checks the skills and hooks content.",
"triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement)"
"triaged": "2026-09-06 provider-vector retirement audit; re-keyed 2026-09-12 (the workflow_dispatch trigger added 23 lines above the unchanged statement); re-keyed 2026-10-01 (the coverage job and the lint step above the unchanged statement were folded away); re-keyed 2026-10-01 251→228 (the matrix comments and steps above reworked to shorten pull request feedback; statement byte-identical)"
},
"C4 scripts/release-verify.sh:106": {
"class": "SAFE-DIAGNOSTIC",
Expand Down
74 changes: 74 additions & 0 deletions tests/release-scripts-safety.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -362,6 +362,80 @@ describe('Feature: release scripts never edit the real ~/.memesh', () => {
expect(releaseJob).not.toContain('--quick');
});

// The coverage floor and the packaged dashboard e2e each used to be their
// own job, and their required check names were what guaranteed they ran on
// every pull request. Both now live inside other jobs, so these two tests
// are what notices if either one quietly stops running.
function ciJob(id: string): string {
const ci = read('.github/workflows/ci.yml');
return ci.match(new RegExp(`\\n {2}${id}:\\n[\\s\\S]*?(?=\\n {2}[A-Za-z0-9_-]+:\\n|$)`))?.[0] ?? '';
}
// The one step of a job whose `run:` line is exactly `run`. A step-level
// `if:` or `continue-on-error` turns it into a green step that proved nothing.
function unconditionalStep(job: string, run: string): string {
const step = job.split(/\n {6}- /).find((s) => s.split('\n').some((line) => line.trim() === `run: ${run}`)) ?? '';
expect(step, `no step runs exactly: ${run}`).not.toBe('');
expect(step).not.toMatch(/\n\s+if:/);
expect(step).not.toContain('continue-on-error');
return step;
}

it('enforces the coverage floor in the ubuntu/Node 24 matrix leg, unconditionally', () => {
const buildJob = ciJob('build-and-test');
expect(buildJob).not.toBe('');
// A job-level `if:` can skip the whole leg, and GitHub counts a skipped
// job as a passing check.
expect(buildJob).not.toMatch(/\n {4}if:/);
expect(buildJob).not.toContain('continue-on-error');
// Adds `coverage` to the existing ubuntu/24 combination; it is not a new leg.
expect(buildJob).toMatch(/- os: ubuntu-latest\n\s+node: '24'\n\s+coverage: true\n/);
unconditionalStep(buildJob, "${{ matrix.coverage && 'npm run test:coverage' || 'npm test -- --run' }}");
const pkg = JSON.parse(read('package.json')) as { scripts: Record<string, string> };
expect(pkg.scripts['test:coverage']).toBe('node scripts/run-tests-isolated.mjs --coverage');
// The leg itself must exist: an `exclude:` would drop it without an `if:`.
expect(buildJob).not.toMatch(/\n\s+exclude:/);
// The floor must not be lowered in passing (raising it is fine).
const thresholds = read('vitest.config.ts').match(/thresholds:\s*\{([^}]*)\}/)?.[1] ?? '';
for (const [metric, floor] of [['statements', 52], ['branches', 48], ['functions', 55], ['lines', 54]] as const) {
expect(Number(thresholds.match(new RegExp(`${metric}:\\s*(\\d+)`))?.[1] ?? 0), metric).toBeGreaterThanOrEqual(floor);
}
// Lint runs inside verify:release; this step is its only home in the leg.
unconditionalStep(buildJob, 'npm run verify:release');
expect(pkg.scripts['verify:release']).toMatch(/^npm run lint && /);
// Windows stays in the matrix: the legs the temp-disk step is for.
expect(buildJob).toMatch(/\n\s+os: \[[^\]]*\bwindows-latest\b[^\]]*\]/);
});

it('puts the test temp folders on the runner temp disk, on every leg', () => {
const step = ciJob('build-and-test').split(/\n {6}- /).find((s) => s.includes('npm test -- --run')) ?? '';
expect(step, 'no test step').not.toBe('');
expect(step).toMatch(/\n\s+TEMP: \$\{\{ runner\.temp \}\}\n/);
expect(step).toMatch(/\n\s+TMP: \$\{\{ runner\.temp \}\}\n/);
});

it('runs the packaged dashboard e2e on every pull request, inside SDLC verify', () => {
const sdlcJob = ciJob('sdlc-verify');
expect(sdlcJob).not.toBe('');
expect(sdlcJob).not.toMatch(/\n {4}if:/);
expect(sdlcJob).not.toMatch(/\n {4}continue-on-error/);
unconditionalStep(sdlcJob, 'npx playwright install --with-deps chromium');
// The full run, not `--journeys`: the full run is the one with the suite.
unconditionalStep(sdlcJob, 'node scripts/verify.mjs');
const config = JSON.parse(read('scripts/verify.config.json')) as {
verify: { steps: Array<{ command: string; args?: string[] }> };
};
const commands = config.verify.steps.map((step) => [step.command, ...(step.args ?? [])].join(' '));
expect(commands).toEqual(
expect.arrayContaining([
'npm run build',
'npm run verify:release',
'node scripts/run-tests-isolated.mjs',
'npm run test:packaged',
'npm run test:e2e-dashboard',
]),
);
});

it('fails the sync gate when an installed adapter artifact is missing', () => {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'message-sync-fixture-'));
const write = (relative: string, content = '') => {
Expand Down
Loading