diff --git a/contracts/model-capability-ladder-v1.yaml b/contracts/model-capability-ladder-v1.yaml index c141463952..a006a7526e 100644 --- a/contracts/model-capability-ladder-v1.yaml +++ b/contracts/model-capability-ladder-v1.yaml @@ -44,10 +44,24 @@ entity: # for that arch; a claimed backend that falls back is RED. # sha256 pins the exact file: a rung measured on a different file is a different # measurement (Q5_K taught us three readers can each invent a layout). +# +# #3712 (operator 2026-09-21: "you must ensure all models Q4_K CUDA work; the end"; +# publishing with "most working" is a "p0 tire fire"): +# * NO Q4_K rung is optional. `required: false` on a Q4_K rung is refused by the +# gate, and so is a Q4_K rung whose `backends` does not claim cuda. +# * The universe is what each host HOLDS, not this list. `inventory` names where +# and what to look for; scripts/model_ladder.sh measures every match on the host +# (on `inventory.backends`) and writes the list into the receipt (schema v2). +# An inventory model missing from the run, or not green on CUDA, is a FAIL +# naming it. A rung is still how a model gets a pinned sha256 and a host list. ladder: hosts: - { id: lambda, isa: x86_64, gpu: "RTX 4090", cc: sm_89, required: true } - { id: gx10, isa: aarch64, gpu: "GB10", cc: sm_121, required: true } + inventory: + dirs: ["~/models", "~/.apr/models", "~/.cache/apr/models"] # depth 1, the fleet's model dirs + patterns: ["*q4_k*.gguf", "*q4k*.gguf", "*q4_k*.apr", "*q4k*.apr"] # case-insensitive + backends: [cuda] rungs: - id: qwen2-1.5b-q4km family: qwen2.5 @@ -71,8 +85,8 @@ ladder: gguf: Qwen3-8B-Q4_K_M.gguf sha256: d98cdcbd03e17ce47681435b5150e34c1417f50b5c0019dd560e4882c5745785 backends: [cpu, cuda] - required: false - note: "second dense-Qwen3 size; required once #3413 lands so the fix is proved at two widths" + required: false # REFUSED by check_model_ladder.sh (#3712): T-2 is RED while this key stands + note: "second dense-Qwen3 size. No Q4_K model is optional (#3712), so the release gate refuses this key and the release cannot ship on it. It is not flipped here because the ARMED ladder-green shape reads the committed 0.68.2 lambda receipt (golden_output: Empty output — the 0.69.0 hold) and would turn pv lint red on every PR; the flip to true lands with the Qwen3-8B CUDA fix and a committed green lambda receipt" - id: qwen35-0.8b-q4km family: qwen3.5 arch: qwen35 @@ -124,21 +138,24 @@ equations: - "a rung the host does not hold is absent, and absent is FAIL for a required rung: an unmeasured claim is not a passed claim" preconditions: - "binary pinned via scripts/apr_bin.sh (built from HEAD) or DOGFOOD_ALLOW_UNPINNED=1 for the published-crate mode" + - "every apr call runs through apr_locked: the fleet GPU lock (flock, bounded wait; a held lock is an ENV decline naming its holder) and choom -n 1000, so a measurement and never a CI job is the OOM victim (#3712; gx10 global OOM 2026-09-21 15:56:08Z killed CI containers)" host_receipt: - formula: "receipt(host) = {host, version, sha, gpu, cc, rungs: [{id, present, capability_match, golden_output, backends: {b: {ran, fallback}}}]}" + formula: "receipt(host) = {schema: apr-model-ladder-receipt/v2, host, version, sha, apr_sha, gpu, cc, inventory: [{file, sha256, bytes}], rungs: [{id, file, present, capability_match, golden_output, backends: {b: {ran, fallback, rc}}}]}" domain: "evidence/dogfood/models//.json, written only by scripts/model_ladder.sh" invariants: - "receipt.version equals the cargo root version of the tree it was measured on" - "receipt.sha equals the git HEAD the binary was built from" - "receipt.executed >= 1 — a receipt that measured nothing is a decline (exit 2), never a pass" + - "receipt.inventory is MEASURED on the host (every file under inventory.dirs matching inventory.patterns), never copied from the ladder; an empty inventory is FAIL" preconditions: - "python3 with json on the host (yaml is read by python3 too)" gate_verdict: - formula: "T2_models = ∀ host required: receipt(host) fresh(version) ∧ ∀ rung required: green(rung, host)" + formula: "T2_models = ∀ host required: receipt(host) fresh(version) ∧ ∀ rung required: green(rung, host) ∧ ∀ f ∈ receipt(host).inventory: green(f, host, inventory.backends)" domain: "scripts/check_model_ladder.sh, invoked by scripts/dogfood.sh through [package.metadata.dogfood]" invariants: - "the rung list at origin/main is the floor: a PR may add rungs, hosts or backends and may not remove any (same construction as check_multiplatform_dogfood.sh layer 2)" - "a missing receipt is FAIL, not DEFER: unlike the multiplatform gate this needs no published crate, a dev build measures it" + - "a Q4_K rung (file or id matches q4_?k) is required and claims cuda — the key `required: false` on one is refused, not tolerated (#3712)" preconditions: - "git can read origin/main:contracts/model-capability-ladder-v1.yaml, or the run is the bootstrap" @@ -168,6 +185,16 @@ proof_obligations: property: "The ladder can grow and cannot shrink in the same PR that reads it" formal: "rungs(origin/main) ⊆ rungs(HEAD) ∧ hosts(origin/main) ⊆ hosts(HEAD) ∧ ∀ rung: backends_main(rung) ⊆ backends_head(rung)" applies_to: gate_verdict + - id: MCL-INV-006 + type: invariant + property: "No Q4_K rung is optional, and every one claims CUDA" + formal: "∀ rung ∈ ladder.rungs: q4k(rung) ⟹ rung.required = true ∧ cuda ∈ rung.backends" + applies_to: gate_verdict + - id: MCL-INV-007 + type: invariant + property: "The universe is the host's measured inventory — every Q4_K model it holds is in the run and green on CUDA" + formal: "∀ host required, ∀ f ∈ receipt(host).inventory: ∃ row ∈ receipt(host).rungs: row.file = f ∧ row.present ∧ (f ∈ files(ladder) ∨ green(row, inventory.backends)) ∧ receipt(host).inventory ≠ ∅" + applies_to: host_receipt falsification_tests: - id: FALSIFY-MCL-001 @@ -220,6 +247,31 @@ falsification_tests: prediction: "narrowing a rung from every host at origin/main to hosts: [gx10] makes the gate exit 1 with 'hosts DROPPED'" test: "bash scripts/check_model_ladder.sh --self-test --case red-dropped-host-on-rung" if_fails: "a required (rung, host) pair could be dropped by adding a key instead of removing one" + - id: FALSIFY-MCL-011 + rule: "A Q4_K rung cannot be made optional or CPU-only" + prediction: "`required: false` on a Q4_K rung exits 1 naming the rung (case red-q4k-required-false); a Q4_K rung with backends [cpu] exits 1 (case red-q4k-rung-cpu-only)" + test: "bash scripts/check_model_ladder.sh --self-test --case red-q4k-required-false && bash scripts/check_model_ladder.sh --self-test --case red-q4k-rung-cpu-only" + if_fails: "the 0.69.0 hold: qwen3-8b-q4km was required: false, so an empty-output CUDA model was a note, not a failure" + - id: FALSIFY-MCL-012 + rule: "An inventory model missing from the run is named" + prediction: "a receipt whose inventory lists a file with no present row exits 1 with 'is MISSING from the run' naming the file (case red-inventory-model-missing)" + test: "bash scripts/check_model_ladder.sh --self-test --case red-inventory-model-missing" + if_fails: "the universe is the list again: a model the host holds is never measured" + - id: FALSIFY-MCL-013 + rule: "An inventory model beyond the ladder is judged on CUDA like a rung" + prediction: "an inventory-only row whose golden_output is skipped, or whose cuda fell back, exits 1 naming inv: (cases red-inventory-model-golden-skipped, red-inventory-model-fell-back); green on cuda exits 0 (case green-inventory-beyond-ladder)" + test: "bash scripts/check_model_ladder.sh --self-test --case red-inventory-model-golden-skipped && bash scripts/check_model_ladder.sh --self-test --case red-inventory-model-fell-back && bash scripts/check_model_ladder.sh --self-test --case green-inventory-beyond-ladder" + if_fails: "an unlisted model is recorded but never judged — present is not green" + - id: FALSIFY-MCL-014 + rule: "A receipt without a measured inventory is not evidence" + prediction: "a v1 receipt (no inventory) exits 1; an empty inventory exits 1; a ladder with no inventory block exits 1 (cases red-receipt-v1-no-inventory, red-inventory-empty, red-ladder-without-inventory)" + test: "bash scripts/check_model_ladder.sh --self-test --case red-receipt-v1-no-inventory && bash scripts/check_model_ladder.sh --self-test --case red-inventory-empty && bash scripts/check_model_ladder.sh --self-test --case red-ladder-without-inventory" + if_fails: "absence scored as conformance: a host that listed nothing proved nothing and passed" + - id: FALSIFY-MCL-015 + rule: "Every GPU apr call runs under the fleet lock, choom'd to 1000, with a bounded wait" + prediction: "check_model_ladder.sh --self-test: model_ladder.sh makes no \"$APR\" call outside apr_locked; a fake apr called through --lock-probe sees the lock held and oom_score_adj 1000; a lock held elsewhere declines with exit 2 naming the holder's pid; four producer mutants (raw call, no flock, no choom, no -w) are each killed. The real run is RED when the audit finds a raw call" + test: "bash scripts/check_model_ladder.sh --self-test" + if_fails: "a ladder run on gx10's unified memory OOM-kills the CI pool (15:56:08Z, 18 kills in 10 s) or hangs a release on a stuck lock" qa_gate: id: F-MCL-001 @@ -231,5 +283,7 @@ qa_gate: - required_rungs_present_and_green - claimed_backends_ran_without_fallback - ladder_not_shrunk_vs_origin_main - pass_criteria: "check_model_ladder.sh exits 0 with executed >= 1 on every required host" + - q4k_rungs_required_and_claim_cuda + - every_inventory_model_measured_and_green_on_cuda + pass_criteria: "check_model_ladder.sh exits 0 with executed >= 1 and a non-empty measured inventory on every required host" falsification: "plant a receipt with cuda.fallback=true → exit 1" diff --git a/docs/roadmaps/entries/PMAT-3712.yaml b/docs/roadmaps/entries/PMAT-3712.yaml new file mode 100644 index 0000000000..69f9f9c31b --- /dev/null +++ b/docs/roadmaps/entries/PMAT-3712.yaml @@ -0,0 +1,17 @@ +- id: PMAT-3712 + github_issue: 3712 + item_type: task + title: Model gate universe = each host's measured Q4_K inventory; no Q4_K rung optional; skip/fallback/rc!=0 RED + status: in_progress + priority: critical + assigned_to: null + created: '2026-09-21T00:00:00Z' + updated: '2026-09-21T00:00:00Z' + spec: null + acceptance_criteria: [] + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'ACCEPTANCE (hand-entered from gh#3712; this row = done_when 1 + 2, the GATE side; cop assignment 2026-09-21: worker B). Issue done_when 1, verbatim: "Universe = measured inventory, not a list. The release gate derives the model set from what''s ON each required host (every `*Q4_K*` GGUF/APR under the declared models dir on lambda AND gx10), unioned with the ladder. A model present on a host but missing from the gate''s run is a FAIL naming it. The ladder has no `required: false` for any Q4_K rung; a guard refuses the key on a Q4_K rung (case row + mutant)." Issue done_when 2, verbatim: "Every (model, host) cell is GREEN on CUDA: capability_match passed (not skipped), golden_output passed (not skipped; #3711), `apr run --gpu` rc 0 with no fallback line. One red cell = NO-GO. There is no known, optional or pre-existing exemption, and no threshold like >= N%." WHAT THIS ROW DOES: (a) contracts/model-capability-ladder-v1.yaml gains `ladder.inventory` {dirs, patterns (case-insensitive *q4_k*/*q4k* .gguf/.apr), backends [cuda]}; (b) scripts/model_ladder.sh measures ladder UNION the host''s inventory (every match, depth 1), judges each inventory model on cuda with the SAME measure() as a rung, and writes receipt schema apr-model-ladder-receipt/v2 carrying `inventory` [{file, sha256, bytes}], `file` per row, and `apr_sha` (full 40-hex, for #3715); (c) scripts/check_model_ladder.sh FAILs: `required` not true on a Q4_K rung; a Q4_K rung that does not claim cuda; a ladder with no inventory; a receipt that is not v2 or has no inventory; an EMPTY inventory; an inventory file with no present row (MISSING from the run, named); an inventory-only model not green on cuda (skip, fallback, rc != 0 are RED); (d) case table grows 18 -> 26 (every case red for exactly its own reason, one FAIL line each except the two-host cases), plus 3 self-mutants each killed by its case (q4k-required-false, q4k-without-cuda, inventory-missing); --case with no such case is now RED, and the root is derived from the script path, not git rev-parse (#3581). WHAT IT DOES NOT DO (cop ruling 2026-09-21): qwen3-8b-q4km stays `required: false` in the contract, and the gate refuses that key at T-2, so the release cannot ship on it. Flipping it here would turn pv lint red on EVERY PR: the armed ladder-green shape reads the committed 0.68.2 lambda receipt (golden_output: Empty output), measured rc 0 -> 1 with only that flip. The flip folds into the 0.69.1 batch with the qwen3-8b fix and a committed green lambda receipt. So done_when 1''s `no required: false` clause is NOT met by this row: Refs, not Closes. Done_when 2''s cells turn green only through the model fixes under EPIC #3710; done_when 3 and 4 are aprender-f0''s (#3708). cells[] per verb x context (the widened bar, #3715 proposal) ships with the cop''s widened-bar delta, not here. WHERE THE GATE RUNS (so its rc=1 on the real contract is the release NO-GO, not a PR red): check_model_ladder.sh is declared ONLY in Cargo.toml [package.metadata.dogfood] gates (line 612, the T-2 pre-publish dogfood) and is listed in scripts/unwired_guards_baseline.txt (line 12): no PR workflow runs it, and guard_tree skips it (unwired-baseline). pv lint is a different tool that never runs this script; it reads the contract and PASSES with qwen3-8b at required: false. Measured on this head: guard_tree --no-cargo 0 failed, pv lint contracts/ PASS, cargo test -p aprender-contracts --lib 0 failed. THE GPU LOCK (cop ruling 2026-09-21, after a gx10 global OOM at 15:56:08Z killed CI containers): every apr call in model_ladder.sh goes through apr_locked = flock -E 75 -w ${MODEL_LADDER_LOCK_WAIT:-1800} /tmp/apr-gpu.lock choom -n 1000 --; a lock still held after the wait is an ENV decline (exit 2) naming the holder pid from /proc/locks, never a hang and never a model verdict. T-1 wraps the script in choom only (a second flock outside would deadlock). check_model_ladder.sh audits it statically in the REAL run (a raw "$APR" call is RED) and behaviourally in --self-test (fake apr via --lock-probe: lock held + oom 1000; a held lock declines naming the pid), with four producer mutants each killed: raw-apr-call, no-lock, no-choom, unbounded. FALSIFY-MCL-015.' diff --git a/docs/roadmaps/roadmap.yaml b/docs/roadmaps/roadmap.yaml index 38f3913d9c..c044f2da80 100644 --- a/docs/roadmaps/roadmap.yaml +++ b/docs/roadmaps/roadmap.yaml @@ -19946,3 +19946,20 @@ roadmap: labels: - kind:code notes: 'ANDON 2, found by aprender-62 as first responder on #3689: workspace-test-shard 1/3 (job 106309701273, runner yoga-build3) failed ont4c3_parity_receipts::the_committed_tree_agrees_with_its_own_denominator with "scripts/parity_receipt_denominator.sh: line 61: python3: command not found" for every receipt. classify() then printed nothing, the loop read that as "other", and the script said "the tree holds 0", exit 1. That is a missing interpreter scored as a count, the class of the git refusal in #3682. The cop ruled (~11:40Z) to fold it into #3689''s push under four conditions, which are done_when 1-5 above. SCOPE: scripts/parity_receipt_denominator.sh, crates/aprender-contracts-cli/tests/ont4c3_parity_receipts.rs, and this fragment. HOW IT IS MET. PY_BIN (seam PARITY_PYTHON, default python3). count_and_check checks `command -v "$PY_BIN"` after listing the tree and before any file: absent -> UNMEASURED line to stderr, return 3, which verify and --print propagate. Each file''s class must now be exactly record|legacy|other; anything else, including nothing, is "ENV the classifier () gave no answer for ", return 2 (the old `*) :` fall-through was the defect). --self-test on a runner with no interpreter prints an UNMEASURED line and exits 0: the classification rows cannot run there. The Rust test accepts exit 3 only when stderr carries both "UNMEASURED runner=" and "reason=no-interpreter interpreter="; any other non-success still fails. MEASURED at this commit: `--self-test` rc 0 with 7 rows, the 5 existing plus "no interpreter is UNMEASURED exit 3 naming it, never a count of 0" and "an interpreter that answers nothing is ENV exit 2, never a count of 0"; a bare run with python3 gives rc 0, "PASS 7 receipt(s) under evidence/parity/**, and evidence/parity/EXPECTED_RECEIPTS says 7." (the T-2 requirement, measured); PARITY_PYTHON=/nonexistent gives rc 3 "UNMEASURED runner=nopy-probe reason=no-interpreter interpreter=/nonexistent/python3 …"; PARITY_PYTHON=/bin/true gives rc 2 "ENV the classifier (/bin/true) gave no answer for evidence/parity/derived_expiries.json". THE YOGA SHAPE, a PATH made of every /usr/bin and /bin executable except python*: the PRE-FIX script (25d2ee264) prints "python3: command not found" 118 times, then "FAIL evidence/parity/EXPECTED_RECEIPTS says 7; the tree holds 0.", rc 1, which is #3689''s failure reproduced; the fix prints "UNMEASURED runner=yoga-shape reason=no-interpreter interpreter=python3 …", rc 3. MUTANT, the silent-classifier arm restored to the old `*) :` fall-through: self-test rc 1, "FAIL a silent interpreter gave exit 1: FAIL evidence/parity/EXPECTED_RECEIPTS says 2; the tree holds 0." THE RUST TEST: `cargo test -p aprender-contracts-cli --test ont4c3_parity_receipts` gives 9 passed with python3; PARITY_PYTHON=/nonexistent/python3 gives 1 passed, printing the UNMEASURED line; PARITY_PYTHON=/bin/true gives FAILED, "exit Some(2)" plus the ENV line, so a crash is still RED and only the named UNMEASURED line buys exit 3. `cargo fmt --all -- --check` rc 0.' +- id: PMAT-3712 + github_issue: 3712 + item_type: task + title: Model gate universe = each host's measured Q4_K inventory; no Q4_K rung optional; skip/fallback/rc!=0 RED + status: in_progress + priority: critical + assigned_to: null + created: '2026-09-21T00:00:00Z' + updated: '2026-09-21T00:00:00Z' + spec: null + acceptance_criteria: [] + phases: [] + subtasks: [] + estimated_effort: null + labels: + - kind:code + notes: 'ACCEPTANCE (hand-entered from gh#3712; this row = done_when 1 + 2, the GATE side; cop assignment 2026-09-21: worker B). Issue done_when 1, verbatim: "Universe = measured inventory, not a list. The release gate derives the model set from what''s ON each required host (every `*Q4_K*` GGUF/APR under the declared models dir on lambda AND gx10), unioned with the ladder. A model present on a host but missing from the gate''s run is a FAIL naming it. The ladder has no `required: false` for any Q4_K rung; a guard refuses the key on a Q4_K rung (case row + mutant)." Issue done_when 2, verbatim: "Every (model, host) cell is GREEN on CUDA: capability_match passed (not skipped), golden_output passed (not skipped; #3711), `apr run --gpu` rc 0 with no fallback line. One red cell = NO-GO. There is no known, optional or pre-existing exemption, and no threshold like >= N%." WHAT THIS ROW DOES: (a) contracts/model-capability-ladder-v1.yaml gains `ladder.inventory` {dirs, patterns (case-insensitive *q4_k*/*q4k* .gguf/.apr), backends [cuda]}; (b) scripts/model_ladder.sh measures ladder UNION the host''s inventory (every match, depth 1), judges each inventory model on cuda with the SAME measure() as a rung, and writes receipt schema apr-model-ladder-receipt/v2 carrying `inventory` [{file, sha256, bytes}], `file` per row, and `apr_sha` (full 40-hex, for #3715); (c) scripts/check_model_ladder.sh FAILs: `required` not true on a Q4_K rung; a Q4_K rung that does not claim cuda; a ladder with no inventory; a receipt that is not v2 or has no inventory; an EMPTY inventory; an inventory file with no present row (MISSING from the run, named); an inventory-only model not green on cuda (skip, fallback, rc != 0 are RED); (d) case table grows 18 -> 26 (every case red for exactly its own reason, one FAIL line each except the two-host cases), plus 3 self-mutants each killed by its case (q4k-required-false, q4k-without-cuda, inventory-missing); --case with no such case is now RED, and the root is derived from the script path, not git rev-parse (#3581). WHAT IT DOES NOT DO (cop ruling 2026-09-21): qwen3-8b-q4km stays `required: false` in the contract, and the gate refuses that key at T-2, so the release cannot ship on it. Flipping it here would turn pv lint red on EVERY PR: the armed ladder-green shape reads the committed 0.68.2 lambda receipt (golden_output: Empty output), measured rc 0 -> 1 with only that flip. The flip folds into the 0.69.1 batch with the qwen3-8b fix and a committed green lambda receipt. So done_when 1''s `no required: false` clause is NOT met by this row: Refs, not Closes. Done_when 2''s cells turn green only through the model fixes under EPIC #3710; done_when 3 and 4 are aprender-f0''s (#3708). cells[] per verb x context (the widened bar, #3715 proposal) ships with the cop''s widened-bar delta, not here. WHERE THE GATE RUNS (so its rc=1 on the real contract is the release NO-GO, not a PR red): check_model_ladder.sh is declared ONLY in Cargo.toml [package.metadata.dogfood] gates (line 612, the T-2 pre-publish dogfood) and is listed in scripts/unwired_guards_baseline.txt (line 12): no PR workflow runs it, and guard_tree skips it (unwired-baseline). pv lint is a different tool that never runs this script; it reads the contract and PASSES with qwen3-8b at required: false. Measured on this head: guard_tree --no-cargo 0 failed, pv lint contracts/ PASS, cargo test -p aprender-contracts --lib 0 failed. THE GPU LOCK (cop ruling 2026-09-21, after a gx10 global OOM at 15:56:08Z killed CI containers): every apr call in model_ladder.sh goes through apr_locked = flock -E 75 -w ${MODEL_LADDER_LOCK_WAIT:-1800} /tmp/apr-gpu.lock choom -n 1000 --; a lock still held after the wait is an ENV decline (exit 2) naming the holder pid from /proc/locks, never a hang and never a model verdict. T-1 wraps the script in choom only (a second flock outside would deadlock). check_model_ladder.sh audits it statically in the REAL run (a raw "$APR" call is RED) and behaviourally in --self-test (fake apr via --lock-probe: lock held + oom 1000; a held lock declines naming the pid), with four producer mutants each killed: raw-apr-call, no-lock, no-choom, unbounded. FALSIFY-MCL-015.' diff --git a/scripts/check_model_ladder.sh b/scripts/check_model_ladder.sh index 40be43269b..a190fe683d 100755 --- a/scripts/check_model_ladder.sh +++ b/scripts/check_model_ladder.sh @@ -3,7 +3,17 @@ # # Reads one receipt per REQUIRED host (written by scripts/model_ladder.sh) for # the version being cut and is green only when every required rung is present -# and green on every one of them. It is declared in Cargo.toml +# and green on every one of them, AND every model the host HOLDS is too. +# +# #3712 (operator 2026-09-21: "you must ensure all models Q4_K CUDA work; the end", +# and publishing with "most working" is a "p0 tire fire"). Three rules, no exemptions +# and no thresholds: +# * No Q4_K rung is optional. `required: false` on a Q4_K rung is REFUSED, and so is +# a Q4_K rung that does not claim cuda. +# * The universe is the host's measured inventory. A receipt must be schema v2 and +# carry a non-empty `inventory`. Every inventory model must appear in the run and be +# green on CUDA, or it is a FAIL naming it. +# * A skipped capability_match or golden_output, a fallback line, or a run rc != 0 is RED. It is declared in Cargo.toml # [package.metadata.dogfood] so scripts/dogfood.sh runs it in every phase; a # missing receipt is FAIL, not DEFER — a dev build can measure this, no # published crate is needed. @@ -31,11 +41,15 @@ while [ $# -gt 0 ]; do --ladder) [ $# -ge 2 ] || { echo "--ladder needs a value" >&2; exit 2; }; LADDER="$2"; shift 2 ;; --ladder-main) [ $# -ge 2 ] || { echo "--ladder-main needs a value" >&2; exit 2; }; LADDER_MAIN_OVERRIDE="$2"; shift 2 ;; --version) [ $# -ge 2 ] || { echo "--version needs a value" >&2; exit 2; }; VERSION_OVERRIDE="$2"; shift 2 ;; - -h|--help) sed -n '2,22p' "$0"; exit 0 ;; + -h|--help) awk 'NR == 1 { next } !/^#/ { exit } { sub(/^# ?/, ""); print }' "$0"; exit 0 ;; *) echo "check_model_ladder: unknown argument '$1'" >&2; exit 2 ;; esac done -cd "$(git rev-parse --show-toplevel 2>/dev/null || pwd)" || exit 2 +SELF="$(cd "$(dirname "$0")" && pwd)/$(basename "$0")" # before the cd: the mutants copy this file +# The root is derived from this file's path, never from `git rev-parse` (it dies in the CI container, +# aprender#3581) or from the caller's cwd. MODEL_LADDER_ROOT is how a mutant copy, which lives in a +# temp dir, is told the tree it judges. +cd "${MODEL_LADDER_ROOT:-$(dirname "$SELF")/..}" || exit 2 # ---------------------------------------------------------------- the judge # judge → exit 0/1/2 @@ -52,6 +66,32 @@ rungs = L.get("rungs", []) if not hosts: print("decline: ladder names no required host"); sys.exit(2) if not rungs: print("decline: ladder has no rungs"); sys.exit(2) rc = 0 +import re +def is_q4k(r): # a Q4_K model, by its file or its id (#3712) + return bool(re.search(r"q4_?k", f"{r.get('gguf', '')} {r.get('id', '')}", re.I)) +def why_of(x, backends): # every reason a measured row is not green on the claimed backends + why = [] + cm, go = x.get("capability_match") or {}, x.get("golden_output") or {} + claims_gpu = bool({"cuda", "gpu"} & set(backends)) + cap_ok = (cm.get("passed") and not cm.get("skipped")) or (cm.get("skipped") and not claims_gpu) + if not cap_ok: why.append("capability_match " + ("SKIPPED" if cm.get("skipped") else "FAIL") + ": " + str(cm.get("message", ""))[:60]) + if not (go.get("passed") and not go.get("skipped")): why.append("golden_output " + ("SKIPPED" if go.get("skipped") else "FAIL") + ": " + str(go.get("message", ""))[:60]) + be = x.get("backends") or {} + for b in backends: + v = be.get(b) + if v is None: why.append(f"{b}: not measured") + elif v.get("fallback"): why.append(f"{b}: FELL BACK — the claimed backend did not run") + elif not v.get("ran"): why.append(f"{b}: did not run (rc={v.get('rc')})") + return why +# #3712: no Q4_K rung is optional, and every one claims cuda. The key is refused, not tolerated. +for r in rungs: + if is_q4k(r) and r.get("required") is not True: + print(f"FAIL rung {r['id']} is a Q4_K rung with required: {r.get('required')!r} — no Q4_K model is optional; every one must be green on CUDA (#3712)"); rc = 1 + if is_q4k(r) and "cuda" not in (r.get("backends") or []): + print(f"FAIL rung {r['id']} is a Q4_K rung that does not claim cuda — every Q4_K model must be green on CUDA (#3712)"); rc = 1 +inv_backends = list((L.get("inventory") or {}).get("backends") or []) +if not (L.get("inventory") or {}).get("patterns") or "cuda" not in inv_backends: + print("FAIL the ladder declares no inventory (patterns + backends incl. cuda) — the universe cannot be the host's measured Q4_K models (#3712)"); rc = 1 # anti-shrink vs origin/main if main_p and os.path.exists(main_p): try: @@ -89,9 +129,28 @@ for h in hosts: print(f"FAIL {h['id']:7} receipt is for {R.get('version')!r}, this cut is {version!r} — STALE"); rc = 1; continue if int(R.get("executed", 0)) < 1: print(f"FAIL {h['id']:7} receipt executed=0 — a receipt that measured nothing is not evidence"); rc = 1; continue + inv = R.get("inventory") + if R.get("schema") != "apr-model-ladder-receipt/v2" or not isinstance(inv, list): + print(f"FAIL {h['id']:7} receipt carries no measured inventory (schema {R.get('schema')!r}) — the universe is what the host HOLDS, not a list (#3712)"); rc = 1; continue + if not inv: + print(f"FAIL {h['id']:7} measured inventory is EMPTY — a host holding no Q4_K model proved nothing (#3712)"); rc = 1; continue by = {r.get("id"): r for r in R.get("rungs", [])} + by_file = {x.get("file"): x for x in R.get("rungs", []) if x.get("file")} + ladder_files = {r.get("gguf") for r in rungs} + inv_green = 0 + for item in inv: + f = item.get("file") + x = by_file.get(f) + if x is None or not x.get("present"): # held by the host, absent from the run + print(f"FAIL {h['id']:7} inventory model {f} is MISSING from the run — the host holds it, so the release must prove it (#3712)"); rc = 1; continue + if f in ladder_files: + continue # a ladder rung: judged, required, in the rung loop below + why = why_of(x, inv_backends) + if why: print(f"FAIL {h['id']:7} inv:{f:22} " + "; ".join(why)); rc = 1 + else: inv_green += 1; print(f"ok {h['id']:7} inv:{f:22} green on {','.join(inv_backends)}") + print(f"ok {h['id']:7} inventory: {len(inv)} Q4_K model(s) held, every one in the run") for r in rungs: - rid = r["id"]; req = bool(r.get("required")) + rid = r["id"]; req = bool(r.get("required")) or is_q4k(r) x = by.get(rid) tag = "required" if req else "optional" listed = r.get("hosts") @@ -103,18 +162,7 @@ for h in hosts: continue if x.get("sha_ok") is False: print(f"FAIL {h['id']:7} {rid:22} sha256 mismatch — a different file is a different measurement"); rc = 1; continue - why = [] - cm, go = x.get("capability_match") or {}, x.get("golden_output") or {} - claims_gpu = bool({"cuda", "gpu"} & set(r.get("backends", []))) - cap_ok = (cm.get("passed") and not cm.get("skipped")) or (cm.get("skipped") and not claims_gpu) - if not cap_ok: why.append("capability_match " + ("SKIPPED" if cm.get("skipped") else "FAIL") + ": " + str(cm.get("message",""))[:60]) - if not (go.get("passed") and not go.get("skipped")): why.append("golden_output " + ("SKIPPED" if go.get("skipped") else "FAIL") + ": " + str(go.get("message",""))[:60]) - be = x.get("backends") or {} - for b in r.get("backends", []): - v = be.get(b) - if v is None: why.append(f"{b}: not measured") - elif v.get("fallback"): why.append(f"{b}: FELL BACK — the claimed backend did not run") - elif not v.get("ran"): why.append(f"{b}: did not run (rc={v.get('rc')})") + why = why_of(x, r.get("backends", [])) if why: if req: print(f"FAIL {h['id']:7} {rid:22} " + "; ".join(why)); rc = 1 else: print(f"warn {h['id']:7} {rid:22} ({tag}) " + "; ".join(why)) @@ -125,6 +173,45 @@ PY } # ---------------------------------------------------------------- self-test +# ---------------------------------------------------------------- the GPU lock +# Cop ruling 2026-09-21 (#3712): every apr call scripts/model_ladder.sh makes runs under the fleet GPU +# lock (flock, bounded wait) and choom -n 1000, through ONE function, apr_locked. Two halves: +# lock_audit (static, also in the REAL run): no apr subcommand is invoked on "$APR" directly. +# lock_probe (behavioural, self-test): a fake apr, called through --lock-probe, must see the lock +# held and its own oom_score_adj at 1000; with the lock held elsewhere the call must +# decline (exit 2) within the bounded wait, naming the holder's pid. +# lock_audit -> prints FAIL lines, exit 1 on any raw call +lock_audit() { + python3 - "$1" <<'LOCKPY' +import re, sys +bad = 0 +for n, line in enumerate(open(sys.argv[1]), 1): + code = re.sub(r"(^|\s)#.*$", "", line) # a call in a comment is not a call + for m in re.finditer(r'(? -> prints ok/FAIL lines, exit 1 on any failure +lock_probe() { + local prod=$1 w=$2 out rc hp bad=0 + mkdir -p "$w"; : > "$w/lock" + printf '#!/usr/bin/env bash\nif flock -n "$FAKE_LOCK" true; then l=UNLOCKED; else l=LOCKED; fi\necho "fake-apr $1 lock=$l oom=$(cat /proc/self/oom_score_adj)"\n' > "$w/apr" + chmod +x "$w/apr" + out=$(FAKE_LOCK="$w/lock" MODEL_LADDER_ROOT="$PWD" MODEL_LADDER_GPU_LOCK="$w/lock" DOGFOOD_ALLOW_UNPINNED=1 APR="$w/apr" timeout 60 bash "$prod" --lock-probe qa probe 2>&1); rc=$? + if [ "$rc" = 0 ] && grep -q 'lock=LOCKED oom=1000' <<< "$out"; then echo "ok lock: an apr call runs holding the lock, at oom_score_adj 1000" + else echo "FAIL lock: the probe call did not run holding the lock at oom 1000 (rc=$rc): $out"; bad=1; fi + python3 -c 'import fcntl, sys, time; f = open(sys.argv[1], "a"); fcntl.flock(f, fcntl.LOCK_EX); time.sleep(60)' "$w/lock" & + hp=$! + sleep 0.5 + out=$(FAKE_LOCK="$w/lock" MODEL_LADDER_ROOT="$PWD" MODEL_LADDER_GPU_LOCK="$w/lock" MODEL_LADDER_LOCK_WAIT=1 DOGFOOD_ALLOW_UNPINNED=1 APR="$w/apr" timeout 8 bash "$prod" --lock-probe run probe 2>&1); rc=$? + kill "$hp" 2> /dev/null; wait "$hp" 2> /dev/null + if [ "$rc" = 2 ] && grep -q 'was not free after 1s' <<< "$out" && grep -q "holder: pid $hp" <<< "$out"; then echo "ok lock: a held lock declines (exit 2) in the bounded wait, naming the holder's pid" + else echo "FAIL lock: a held lock did not decline in the bounded wait naming its holder (rc=$rc): $out"; bad=1; fi + return "$bad" +} + if [ "$SELF_TEST" = 1 ]; then n=0; bad=0 for c in "$CASES_DIR"/*/; do @@ -144,6 +231,39 @@ if [ "$SELF_TEST" = 1 ]; then fi done if [ "$n" -lt 6 ] && [ -z "$ONLY_CASE" ]; then echo "FAIL only $n case(s) ran; the table needs >= 6 to discriminate"; bad=$((bad+1)); fi + if [ -n "$ONLY_CASE" ] && [ "$n" -eq 0 ]; then echo "FAIL no case named $ONLY_CASE under $CASES_DIR -- a case that did not run is not a pass"; bad=$((bad+1)); fi + # Mutants (#3712): each refusal is deleted in a copy of this script, and the case that + # names it must go RED under the copy. A rule no case can tell from its absence is theater. + if [ -z "$ONLY_CASE" ]; then + mdir=$(mktemp -d "${TMPDIR:-/tmp}/ladder-mut.XXXXXX") || exit 2 + mutant() { # mutant