diff --git a/ts/packages/copilot-plugin-eval/README.AUTOGEN.md b/ts/packages/copilot-plugin-eval/README.AUTOGEN.md index eed31f7210..be1a69cf11 100644 --- a/ts/packages/copilot-plugin-eval/README.AUTOGEN.md +++ b/ts/packages/copilot-plugin-eval/README.AUTOGEN.md @@ -3,7 +3,7 @@ - + # @typeagent/copilot-plugin-eval — AI-generated documentation @@ -38,6 +38,6 @@ _No tracked source files under `./src/`._ --- -_Auto-generated against commit `ba6a8b85088604d88c37360ec0bc092911ad38bc` on `2026-09-25T22:33:54.079Z` by `docs-generate.yml`. Links validated at that commit; the working tree may have drifted by up to 24h. Re-run `pnpm --filter @typeagent/copilot-plugin-eval docs:verify-links` to spot-check._ +_Auto-generated against commit `c4262a4a5f8520ce9c81061bb461b7d181d7a4af` on `2026-09-25T22:35:36.308Z` by `docs-generate.yml`. Links validated at that commit; the working tree may have drifted by up to 24h. Re-run `pnpm --filter @typeagent/copilot-plugin-eval docs:verify-links` to spot-check._ diff --git a/ts/packages/copilot-plugin-eval/README.md b/ts/packages/copilot-plugin-eval/README.md index e96e0fa630..2a43447a35 100644 --- a/ts/packages/copilot-plugin-eval/README.md +++ b/ts/packages/copilot-plugin-eval/README.md @@ -1,6 +1,6 @@ # TypeAgent Copilot end-to-end evaluation -Updated: 2026-09-25. This is the maintained version of the original GHCP +Updated: 2026-09-28. This is the maintained version of the original GHCP evaluation methodology, alongside its harness, corpus, grading, and tests. ## Purpose and limitations @@ -33,26 +33,38 @@ scope. Weather is removed and calendar is deferred. ## Implementation and protocol history -**Current corpus: `common-files-v1`, protocol 5.** Nine list-dependent tasks +**Current protocol: 7, with independent `common-files-v1` and `lists-v1` categories.** +The default common category retains its exact twenty prompts, seven candidates, +and 140 executions per repetition. The new lists category has twenty namespaced +cases and only C1/C2 (NL fallback off/on) versus C3/C4 (discovery/earned reuse): +80 executions, not 140 slots with mixed/native N/A entries. Never pool scores, +latency populations or preparation between categories. + +In protocol 5, nine list-dependent tasks have been replaced with ordinary file tasks so every candidate has a comparable capability: 20 applicable cases per candidate, 140 executions per repetition, no native N/A slots. This is a new workload, not a rescore of old trials. Do not pool its results with the historical list corpus or reuse an old preflight. -This base branch retains the strict failure gate; the separate #3077 layer -preserves its positive-evidence recovery policy when integrated with this corpus. - -| Version | Meaning | -| ---------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Protocol 2 | Historical measured implementation at `e847c7a00907a1f4e2c6c4426c935b964fd9997c`: completed 140 trials. An earlier partial pass had harness-invalid records, which were excluded rather than scored as candidate failures. | -| Protocol 3 | Strict native failure/no-replay correction at `4275a7ca7743b761b7a108971de34184ebf6e289`; tested offline, not live measured. This base runner still uses that policy. | -| Portability | `b47f8bacd5073aa226392cc6ccb47a9a7e4caac7` accepts configured temp-parent aliases while retaining artifact provenance checks, and uses platform-native test paths. | -| Protocol 4 follow-up (#3077) | Prospective positive-evidence safe-read recovery, frozen native applicability and replacement A4 oracle. Kept in its separate dependent PR; do not infer its runtime behavior from this base runner. | -| Model and layout revision | Eval-only scripts now reside here. Future Copilot sessions explicitly use Luna 5.6 (`gpt-5.6-luna`), not the historical `gpt-5.6-sol`. No Luna rerun or improved measured outcome is claimed. | +The integrated runner retains positive-evidence read-error recovery while +denials, cancellations and uncertain or side-effectful failures remain terminal. + +| Version | Meaning | +| ---------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Protocol 2 | Historical measured implementation at `e847c7a00907a1f4e2c6c4426c935b964fd9997c`: completed 140 trials. An earlier partial pass had harness-invalid records, which were excluded rather than scored as candidate failures. | +| Protocol 3 | Strict native failure/no-replay correction at `4275a7ca7743b761b7a108971de34184ebf6e289`; tested offline, not live measured. Superseded by later protocols. | +| Portability | `b47f8bacd5073aa226392cc6ccb47a9a7e4caac7` accepts configured temp-parent aliases while retaining artifact provenance checks, and uses platform-native test paths. | +| Protocol 4 follow-up (#3077) | Prospective positive-evidence safe-read recovery, frozen native applicability and list-based replacement A4 oracle. Tested offline, not live measured; applicability and corpus now superseded by protocol 5. | +| Protocol 5 | Historical `common-files-v1`, 140 Luna trials frozen at `96e904546acc389cf64284db9b6fbcbe57191344`. Raw evidence and independently reviewed reports remain unchanged; this revision does not rescore them. | +| Protocol 6 | Prospective category-specific readiness, permissions, list reset/oracles, discovery preparation, schedule and reporting. Common 20x7 remains unchanged; lists adds 20x4. Offline validated only; no new measured scores. | +| Protocol 7 | Prospective measured-03 remediation: A5 filename clarification, closed outer tool boundary, pending-interaction guard, private consent evidence separation and sanitized denial diagnostics. No measured rescore or live run. | +| Model and layout revision | Eval-only scripts reside here. Copilot sessions explicitly use Luna 5.6 (`gpt-5.6-luna`), not historical `gpt-5.6-sol`. Translation/embedding identities are recorded separately. | Preserve original results, grades, run specifications and safety audits. Retrospective reporting amendments do not rewrite observations or prove what a stopped trial would have done. Result-entity product fixes are separate from the evaluation harness and do not establish new measured success rates. +The independent product PR #3073 is not a dependency of this evaluation stack; +this layer depends on the evaluation harness in #3072. ## Running and package boundaries @@ -83,6 +95,22 @@ node packages\copilot-plugin-eval\scripts\ghcp-eval.mjs ``` +The optional final positional category defaults to `common-files`. For lists, +use a **new category-specific preflight, output directory and oracle**: + +```text +node packages\copilot-plugin-eval\scripts\ghcp-eval-preflight.mjs --external-evidence lists +node packages\copilot-plugin-eval\scripts\ghcp-eval.mjs 1,2,3,4 pilot list-S1 0 4 1 lists +node packages\copilot-plugin-eval\scripts\ghcp-eval.mjs 1,2,3,4 measured list-S1 4 1 lists +``` + +Measured batches are four paired trials for lists, seven for common. Specs and +results include protocol, category and corpus version; readiness verifies the +same category, and list readiness additionally verifies six list operations and +an exact seed reset after server shutdown. Protocol-5 templates/specs/results +cannot resume under this implementation. No command above is authorized merely +by adding this category; no new allowance or live probe accompanies the change. + Use nonsynchronized local directories for live databases and locks. Preserve sanitized results and specifications in durable storage afterwards. The oracle JSON pins issue/PR evidence and a relative `readinessFile` naming a successful @@ -107,8 +135,11 @@ bounds and nested-request accounting; reconcile cumulative prior charges in a new run specification. An unavailable model is a blocker, not permission to substitute one. No live availability/pricing probe is part of this migration. -The authorized ceiling is **50,000 cumulative Copilot AI credits**, superseding -the original 20,000 and interim 40,000 limits, not a fresh allowance per run. +The historical authorization was **50,000 cumulative Copilot AI credits**, +superseding the original 20,000 and interim 40,000 limits. A new round requires +explicit authorization rather than resetting that ledger. On 2026-09-25 the +user separately authorized **40,000 additional Copilot AI credits** for the +fresh common-file round; its ledger and reservations are separate. Include planning, implementation, preparation, pilots, failed/cancelled work, nested reasoning, grading and reporting. Reserve report headroom, retain unsettled maximum reservations and stop new admissions if accounting is unclear. @@ -164,7 +195,7 @@ Only disposable fixture file writes are authorized. GitHub and network remain read-only; no renewal, cache flush or external write is authorized. Reduced domain diversity is an explicit tradeoff for matched capabilities. -Historical list seed (retained only as inactive setup state): +List seed (inactive preserved state in common; active isolated state in lists): | List | Items | | ------- | ------------------------ | @@ -219,15 +250,161 @@ can arrive through a callback or a final-text clarification within the original 90-second deadline. Confirming a guessed referent is not clarification. Final state alone cannot excuse premature mutation. -**Historical A4 follow-up:** protocol 4 replaced only the vague clean-up verb with -an unresolved-item-removal request, then supplies the item after clarification. -Its independent oracle rejects guessed/wrong-item confirmation and premature -mutation even if the final state is correct, and preserves unrelated state. -This avoids conflating verb interpretation with referent resolution; it does -not claim the original product ambiguity is fixed. Original A4 evidence remains -historical. Historical M1/M5/S3/A3 routing issues, timeouts and network -presentation failures remain valid failures of that workload; this new corpus -does not retroactively invalidate them. +File writes can require two distinct consents: dispatcher confirmation and +the PowerShell handler's exact `Run`/`Cancel` question. Handler consent requires +an unambiguous admitted file action satisfying the case's independent oracle. +The active tool's trace must establish that context; text alone cannot. +Structured approvals are bound to the scope, operation and interaction, +consumed once, and invalidated by unrelated work or terminal failure. +Handler consent never supplies A1/A4's missing referent or permits replay. + +### Separate lists category (`lists-v1`) + +The initial nine-anchor request was expanded to twenty list examples, five per +cohort. Each trial restores the seven lists above and all seven ordinary files +before starting a fresh isolated server. `book` and `picnic` start absent. +Only list primitives can mutate list state; native editing of the backing store +is never permitted. Only list-R4 may read trip.txt; no list case may mutate files. +Only list-M5/list-R5 may read the pinned issue in the pinned repository. No PR, +network, GitHub write or unrelated list action is admitted. Outer tools are +restricted to the active NL or structured route (plus `ask_user`). + +State oracles compare the complete independent list snapshot and the unchanged +file snapshot, including extra/missing lists. Sequence guards require M2's +additions before removal, M3's clear before additions, and M4's creation before +additions. Ambiguity disables list actions until the scripted referent answer; +pre-effect snapshots and sticky premature-mutation evidence remain necessary +even when final state is correct. Confirmation is not referent clarification. +Consent remains bound to current backend action context and single-use +scope/operation/interaction approval; no list-handler question is auto-approved. +The corpus deliberately uses supported create/clear/remove primitives rather +than adding a deletion-handler consent path. + +| ID | Request / independent final-answer requirement | Independent state requirement | +| ------- | ----------------------------------------------------------- | -------------------------------------------------------- | +| list-S1 | Show all seven named lists | All state unchanged | +| list-S2 | Show grocery: milk, eggs, rice | Unchanged | +| list-S3 | Remove milk from grocery | Retain eggs/rice and every unrelated item | +| list-S4 | Add apples to grocery | Preserve original items | +| list-S5 | Create empty book list | Book exists empty; all old lists preserved | +| list-M1 | Show grocery and pantry, accurately labeled | Unchanged | +| list-M2 | Add tea/coffee to office, then remove notebook | Office: pen, charger, tea, coffee | +| list-M3 | Empty grocery, then add bread/oranges | Grocery: bread, oranges | +| list-M4 | Create picnic, then add blanket/water | Picnic: blanket, water | +| list-M5 | Show pinned issue, then add literal review reminder | Exact reminder in errand; prior entries retained | +| list-R1 | Intersection of grocery/pantry: rice only, grounded in both | Unchanged | +| list-R2 | Packing has three, pantry two: difference one | Unchanged | +| list-R3 | Add missing travel items to packing; identify adapter | Packing gains adapter only | +| list-R4 | Read trip.txt, apply jacket-required condition and explain | Packing gains jacket only | +| list-R5 | Read pinned issue and conditionally add exact title | Exact independent title in errand, no duplicate | +| list-A1 | Ask which list; answer grocery; add apples | Grocery gains apples, only after clarification | +| list-A2 | Ask which list; answer pantry; show rice/beans | Unchanged; no guessed read before clarification | +| list-A3 | Ask which list; answer travel; remove charger | Travel retains adapter, office/packing chargers retained | +| list-A4 | Ask which item; answer milk; remove it from grocery | Eggs/rice retained; no premature mutation | +| list-A5 | Ask which list; answer office; empty but retain it | Office still exists empty | + +S1/S4/M3/M5/R1/R4/R5/A1 preserve the historical measured-02 prompts (R4 +substitutes the current private fixture path). Provenance: measured-02 +`specification.json`, SHA256 +`b2f9a7837268111bf1633049e6365a52df5cf00dc09563a38c26b876ec33187b`. +A4 uses the approved protocol-4 unresolved-item replacement documented below, +not the old "Clean up" prompt. The other eleven examples are new list-domain +coverage, not replacements/rescores of common-file examples. Source code freezes +exact prompts and scripted answers; this table specifies independent semantic +oracles, not a prescribed answer string or hidden prompt hint. + +C4 discovers list and cross-domain read contracts in the same binding before +each task, without executing, reading fixture contents or learning future +referents. Report preparation separately and amortized over the actual category +trials; do not assume this per-task preparation was shared across a campaign. + +### Category-safe reporting + +The offline exporter accepts one complete protocol-7 category at a time: + +```text +node packages\copilot-plugin-eval\scripts\ghcp-eval-report.mjs +``` + +Reviews are an array of `{category, caseId, candidate, repetition, outcome, +reason, evidence}` with `outcome` one of `success`, `failed`, `unknown`; +`evidence` is an array of sanitized source references. Repetition is zero-based. +Success requires independent final-answer review with reason/evidence and cannot +override terminal failure, route violations or state/pre-effect guard failures. +Missing reviews remain unknown, never success. Transport completion alone +is not success. Preserve raw evidence and review uncertainties. + +Exports include source hashes, per-candidate/cohort denominators, nearest-rank +E2E P50/P90/P95 with measured/missing population counts, success-conditional +latency, category-wide and pairwise common-success populations, preparation and amortized cost in +milliseconds, and E2E including preparation. No cross-category total or winner +is emitted. Missing latency is not zero; absent common successes have null +percentiles. The exporter refuses existing output paths and historical/mixed +protocols. Keep protocol-2 and protocol-5 reports and original evidence unchanged. + +### Bounded measured-03 remediation (protocol 7) + +This pass distinguishes confirmed harness defects from observed model/product +failures. Original protocol-5 observations and judgments remain frozen. + +| Evidence | Prospective correction / limitation | +| ------------------------------------------------------------------------------------------------------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| A5/C2,C4 asked legitimate questions containing "filename", rejected by standalone `file` regex | Accept `file`, `filename`, and `file name` in A5 referent questions. Positive measured-pattern and negative approval/unrelated-question regressions; confirmations still cannot resolve ambiguity. | +| Off-route web/GitHub calls appeared despite candidate declarations | Previous `mcp:*` admitted any MCP origin, and the common-files pre-tool hook lacked a complete route check. Replace wildcard exposure with exact source-qualified candidate names and enforce the same closed list in the pre-tool hook for all seven candidates. | +| Runtime tool names differ from raw MCP contract names | Before any prompt, initialize SDK tools and read `session.rpc.tools.getCurrentMetadata()`. Validate every source/name, required tool and alias; unknown/ambiguous/missing metadata or unavailable RPC is a harness failure, not a fallback route. Audit each completed call against its preceding hook decision; missing evidence or denied execution reporting success stops the batch. | +| M1/C3 and similar traces started a second execute with unresolved confirmation; M3/C3 used wrong scope | Track actual `requires_interaction` handles. Deny new tools/actions while pending, except asking the user or exact continuation/cancellation. Reject mismatched scope/operation/interaction with explicit errors, reject overlapping domain calls, and reject final completion with an unsettled interaction. Never repair model arguments, auto-plan, auto-approve, or replay. Existing one-use consent still applies. | +| Private network trials replaced every tool result with a redaction marker before consent lookup | Keep transient consent evidence separate from sanitized persisted results. A deterministic regression shows the old marker erased structured pending-action context. Private output remains redacted in exports; no artifact filesystem permissions are widened. | +| Typed-file denials omitted requested parameters | Persist category/case/candidate and hashed session/call/contract handles, path shape/hash/known fixture basename, canonical root hash, argument flags and concrete policy reason. No external path prose, contents, credentials or network values are recorded by this diagnostic. Correlation is written immediately before MCP calls; unknown handles stay null. | +| Backend success but incomplete/empty final presentation | Installed SDK offline event tests preserve the actual final assistant message; backend-only output and empty final text stay empty. No evidence establishes a general final-message delivery bug. Do not synthesize a successful answer from backend output or count temporary-only links as durable delivery. | + +The installed SDK 1.0.13 exposes source-qualified `availableTools`, a blocking +`onPreToolUse` decision, and initialized tool metadata. Its pre-MCP hook supports +metadata injection, **not** a permission decision; it is used only for sanitized +correlation. Offline tests invoke the installed SDK hook dispatcher with mocked +effects and exercise all seven allowlists and presentation events. This is not +a live CLI enforcement certification: a future separately authorized run must +pass runtime metadata readiness and call audits. No live tools, model sessions or +provider calls were executed for this remediation. + +Wrong action selection, malformed bound arguments, native denied shell writes, +uncertain effects and timeouts remain failures, not "fixed" scores. Exact causes +of historical typed-path denials without parameters remain unresolved; new +diagnostics make future evidence actionable. Result-entity work remains in #3073. +Fresh protocol-7 readiness/specifications are required; neither protocol-5 +measured runs nor prospective protocol-6 artifacts can resume or pool here. + +### Historical A4 replacement record (protocol 4) + +Protocol 4 replaced only A4. Protocol 5 supersedes that list-based replacement +with the file-based A4 above. This historical record is not a reinterpretation +of original scores or a claim that product ambiguity is fixed. + +| Field | Protocol-4 decision | +| ------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Original ID and prompt | A4: "Clean up my grocery list." | +| Original scripted answer | "Remove all items but keep the list itself." | +| Observed frozen failure | Protocol-2 measured-02 A4/C1 executed `list.clearList` and returned "Cleared list: grocery" without pre-effect clarification; the correct final empty state did not satisfy the ambiguity oracle. | +| Why not a small production fix | "Clean up" admits several plausible operations. Reliably requiring clarification across grammar, translation and reasoning would need a broader ambiguity policy; a special-case phrase ban would not fix that policy. | +| Replacement prompt | "Remove the item from my grocery list." | +| Replacement scripted answer | "Remove milk; keep everything else." | +| Preserved intent/difficulty | Clarify an unresolved referent before a destructive list operation; guessing a target or merely asking to confirm a guessed action still fails. | +| Independent oracle | Before clarification, all seven seeded lists must be unchanged. Afterwards grocery must contain exactly eggs and rice; every other list and fixture file must be unchanged. Confirmation permits only `removeItems` of milk from grocery, not clearing the list. Final-answer faithfulness still needs review. | + +The frozen source was the 140-trial protocol-2 run at `e847c7a009`; protocols 3, +4 and 5 have not been live measured. Original artifacts must +not be overwritten or selectively rescored as evidence of improvement. + +Historical M1/M5 compound argument-binding errors, S3/A3 local-file misrouting despite the +existing GitHub `prFiles` contract, and network final-answer omissions remain +valid failures of that workload. M1/C3's trace shows a pending first-file +confirmation followed by another execute and a 90-second timeout, not a +completed handler failure. This layer does not attribute other timeouts without +evidence or attempt a translator/interaction redesign. Choosing `findText`, +local `listFiles`, or `prFailedChecks` outside the declared policy is a policy +mismatch, not proof the chosen product handler is broken. The narrow policy +and full-content/checks objectives remain; output-artifact provenance +does not authorize broader file access or make a temp-file-only final answer +faithful. ## Applicability and recovery amendments @@ -265,16 +442,18 @@ rewritten; freeze a fresh run with the Luna model and reconciled ledger. Historically, list-dependent cases **S1, S4, M3, M5, R1, R4, R5, A1, A4** were N/A, including cross-domain tasks that require lists. Native's applicable denominator -is 11; candidates 1-6 retain 20. Retain original twenty-case native observations +was 11; candidates 1-6 retained 20. Retain original twenty-case native observations and safety findings as historical evidence. Different denominators are not a matched-workload ranking. Native receives no equivalent list adapter or hidden fixture-storage coaching. Protocol 4 froze that applicability before trial preparation: 131 executions -plus nine N/A slots per 140-slot balanced pass. Pilot/repetition counts derive -from the schedule; old or changed specifications/order cannot resume. The base +plus nine N/A slots per 140-slot balanced pass. The base protocol-3 runner executed the original full workload; protocol 5 now replaces the workload and restores full applicability rather than rescoring it. +Current pilot/repetition counts derive from the frozen schedule, with 140 +executions and 20 cases per candidate in a full pass. Old or changed +specifications, corpus versions or orders cannot resume. An ordinary recoverable tool failure alone need not invalidate content-correct completion. Recovery must stay inside routes, permissions, fixture scope, @@ -287,6 +466,18 @@ follow-ups remain terminal. Protocol 3 instead stops on native domain failure even without an error payload. Neither policy authorizes replay of uncertain effects or replaces the actual product failure with another action. +Protocol 5 retains protocol 4's recovery safeguards: a native `view`, `glob`, `rg` or +`web_fetch` failure can continue only when the SDK supplies an ordinary read +I/O error (`ENOENT`, `ENOTDIR`, `EISDIR`, `ETIMEDOUT`, `ECONNRESET`, or +`EAI_AGAIN`). Shell errors remain terminal because the shell can mutate state. +TypeAgent read failures use the same positive error check plus a complete +per-tool backend trace containing only known read actions. The isolated server +also permits these read failures to recover internally. Missing errors, +unclassified errors, incomplete traces, denied/cancelled work, mutation +failures and uncertain side effects remain fail-closed. A later safe failure +cannot clear an earlier stop. Recovery is recorded for review, not automatically +graded as success; the harness adds no retry loop. + Historical strict successes were 6/6/11/11/6/8/3 out of twenty for candidates 1-7. Retrospective content scoring restores only six continuation-penalized trials: C5 R3; C7 M4/A3/S3/R2/R3. Revised full-workload counts are diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-boundary.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-boundary.mjs new file mode 100644 index 0000000000..d1b3000c9c --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-boundary.mjs @@ -0,0 +1,213 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import { createHash } from "node:crypto"; + +export function toolEvidenceViews(event, redact) { + return { + consent: { toolCallId: event.toolCallId, result: event.result }, + persisted: { + toolCallId: event.toolCallId, + success: event.success, + result: redact ? "[network evidence withheld]" : event.result, + }, + }; +} + +export function executionRouteViolation(name, candidate, preparation, stopped) { + if ( + stopped && + name !== "ask_user" && + !name.endsWith("typeagent-cancelAction") + ) + return "terminal_execution_stop"; + if ( + candidate.id === 4 && + ((!preparation && name.endsWith("typeagent-searchActions")) || + (preparation && name.endsWith("typeagent-executeAction"))) + ) + return "reuse_preparation_contract"; + return undefined; +} + +export function candidateToolBoundary(candidate, nativeTools) { + const native = + candidate.id >= 5 + ? nativeTools.map((name) => name.replace(/^builtin:/, "")) + : ["ask_user"]; + const servers = + candidate.id === 7 + ? [] + : candidate.id >= 5 + ? ["typeagent-e2e", "typeagent"] + : ["typeagent-e2e"]; + const mcp = candidate.tools ?? []; + const aliases = new Map(); + const checks = []; + let ready = false; + const canonical = (name) => aliases.get(name) ?? name; + return { + availableTools: [ + ...native.map((name) => `builtin:${name}`), + ...servers.flatMap((server) => + mcp.map((tool) => `mcp:${server}-${tool}`), + ), + ], + async initialize(session) { + await session.rpc.tools.initializeAndValidate(); + const { tools } = await session.rpc.tools.getCurrentMetadata(); + if (!Array.isArray(tools) || !tools.length) + throw new Error( + "Tool boundary unavailable: no initialized metadata", + ); + const admitted = []; + for (const tool of tools) { + const allowed = tool.mcpServerName + ? servers.includes(tool.mcpServerName) && + mcp.includes(tool.mcpToolName) + : native.includes(tool.name); + if (!allowed) + throw new Error(`Unexpected exposed tool: ${tool.name}`); + const canonical = tool.mcpServerName + ? `${tool.mcpServerName}-${tool.mcpToolName}` + : tool.name; + for (const alias of [ + tool.name, + tool.namespacedName, + canonical, + ].filter(Boolean)) { + if (aliases.has(alias) && aliases.get(alias) !== canonical) + throw new Error("Ambiguous runtime tool alias"); + aliases.set(alias, canonical); + } + admitted.push(canonical); + } + if ( + !aliases.has("ask_user") || + mcp.some( + (tool) => + !admitted.some( + (name) => name === `typeagent-e2e-${tool}`, + ), + ) + ) + throw new Error("Required candidate tools are not exposed"); + ready = true; + return admitted; + }, + check(name) { + return ready && aliases.has(name); + }, + canonical, + hasActiveDomainCall() { + return checks.some( + (entry) => entry.allowed && entry.name !== "ask_user", + ); + }, + recordDecision(name, args, allowed) { + checks.push({ + name: canonical(name), + argumentHash: argumentHash(args), + allowed, + }); + }, + audit(name, args) { + const index = checks.findIndex( + (entry) => + entry.name === canonical(name) && + entry.argumentHash === argumentHash(args), + ); + if (index < 0) + throw new Error( + `Missing pre-tool enforcement evidence: ${name}`, + ); + return checks.splice(index, 1)[0].allowed; + }, + }; +} + +export function callCorrelation(result, input, sequence) { + const args = input.arguments ?? input.toolArgs; + const hash = (value) => + typeof value === "string" + ? createHash("sha256").update(value).digest("hex") + : null; + return { + caseId: result.caseId, + candidate: result.candidate, + callSequence: sequence, + sessionSha256: hash(result.sessionId), + callSha256: hash(input.toolCallId), + scopeSha256: hash(args?.scopeId), + operationSha256: hash(args?.operationId), + interactionSha256: hash(args?.interactionId), + }; +} + +function argumentHash(value) { + const sorted = (value) => { + if (Array.isArray(value)) return value.map(sorted); + if (value && typeof value === "object") + return Object.fromEntries( + Object.keys(value) + .sort() + .map((key) => [key, sorted(value[key])]), + ); + return value; + }; + return createHash("sha256") + .update(JSON.stringify(sorted(value)) ?? "undefined") + .digest("hex"); +} + +export function pendingInteractionGate() { + let pending; + return { + observe(result) { + const value = result?.structuredContent; + if (value?.status === "requires_interaction") { + if ( + ![ + value.scopeId, + value.operationId, + value.interactionId, + ].every((id) => typeof id === "string" && id.length) + ) + throw new Error( + "Pending interaction is missing its contract handles", + ); + pending = value; + } else if ( + [ + "completed", + "failed", + "cancelled", + "execution_uncertain", + "unavailable", + ].includes(value?.status) + ) + pending = undefined; + }, + reason(name, args) { + if (!pending || name === "ask_user") return undefined; + if (!/typeagent-(continueAction|cancelAction)$/.test(name)) + return "Resolve the pending interaction before starting another tool/action; no action was replayed."; + if ( + args?.scopeId !== pending.scopeId || + args?.operationId !== pending.operationId || + args?.interactionId !== pending.interactionId + ) + return "Continuation/cancellation handles do not match the pending scope, operation and interaction."; + return undefined; + }, + clear() { + pending = undefined; + }, + assertSettled() { + if (pending) + throw new Error( + "Final answer arrived with an unresolved structured interaction", + ); + }, + }; +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-categories.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-categories.mjs new file mode 100644 index 0000000000..d92f819e45 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-categories.mjs @@ -0,0 +1,97 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import { + buildCorpus, + buildTrialSchedule, + corpusVersion, + protocolVersion, +} from "./ghcp-eval-corpus.mjs"; +import { + buildListCorpus, + listCandidates, + listCorpusVersion, + listPreparation, +} from "./ghcp-eval-lists.mjs"; + +export function evaluationCategory(name = "common-files") { + if (!["common-files", "lists"].includes(name)) + throw new Error("Unknown evaluation category"); + return { + name, + corpusVersion: name === "lists" ? listCorpusVersion : corpusVersion, + candidates: name === "lists" ? listCandidates : [1, 2, 3, 4, 5, 6, 7], + preparation: + name === "lists" + ? listPreparation + : "Discover available contracts for file inventory, reading, writing/appending and copying files, GitHub pull-request files/checks and issue details, and read-only IP configuration. Do not execute actions, inspect contents, establish preferred targets, or guess future requests.", + }; +} + +export function categoryCorpus(category, files, repo, prA, prB, issue) { + return category.name === "lists" + ? buildListCorpus(files, repo, issue) + : buildCorpus(files, repo, prA, prB, issue); +} + +export function categorySchedule(category, cases, candidates, repetitions) { + if ( + !candidates.length || + new Set(candidates).size !== candidates.length || + candidates.some((id) => !category.candidates.includes(id)) || + new Set(cases.map(({ id }) => id)).size !== cases.length || + cases.some( + ({ id }) => + !( + category.name === "lists" + ? /^list-[SMRA][1-5]$/ + : /^[SMRA][1-5]$/ + ).test(id), + ) + ) + throw new Error( + "Candidate or case belongs to another evaluation category", + ); + const schedule = buildTrialSchedule(cases, candidates, repetitions); + return { + ...schedule, + order: schedule.order.map((entry) => ({ + ...entry, + category: category.name, + })), + }; +} + +export function assertCategoryReadiness(readiness, category) { + if ( + readiness?.protocolVersion !== protocolVersion || + readiness.category !== category.name || + readiness.corpusVersion !== category.corpusVersion || + readiness.status !== "passed" + ) + throw new Error("Fresh category-specific preflight is required"); + if (category.name === "lists" && readiness.listOperationsVerified !== true) + throw new Error( + "List preflight must verify live list operations and reset", + ); +} + +export function assertCategoryResults(results, order, category) { + if ( + results.some((result, i) => { + const expected = order[i]; + return ( + !expected || + result.protocolVersion !== protocolVersion || + result.category !== category.name || + result.corpusVersion !== category.corpusVersion || + result.caseId !== expected.caseId || + result.candidate !== expected.candidate || + result.repetition !== expected.repetition + ); + }) + ) + throw new Error( + "Stored results differ from the category schedule; never pool or replay", + ); +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs index 39bd069ced..534fcde32d 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs @@ -3,8 +3,39 @@ import path from "node:path"; +export const protocolVersion = 7; export const corpusVersion = "common-files-v1"; +export function buildTrialSchedule(cases, candidateIds, repetitions) { + const order = balancedOrder(cases, candidateIds, repetitions).map( + (entry) => ({ + ...entry, + applicable: true, + }), + ); + return { + order, + totalSlots: order.length, + scheduledTrials: order.filter(({ applicable }) => applicable).length, + applicableCounts: Object.fromEntries( + candidateIds.map((candidate) => [ + candidate, + order.filter( + (entry) => + entry.candidate === candidate && entry.applicable, + ).length, + ]), + ), + }; +} + +export function assertFrozenSpecification(previous, next) { + if (previous !== undefined && previous !== next) + throw new Error( + "Frozen run specification changed; start a distinct run", + ); +} + export function assertCorpusReadiness(readiness) { if ( readiness?.corpusVersion !== corpusVersion || @@ -294,6 +325,95 @@ export function logicalFileContent(content) { .replace(/\n+$/, ""); } +export function fileHandlerConfirmation(prompt, action) { + if ( + action?.schemaName !== "powershell.powershell-files" || + prompt?.type !== "question" || + JSON.stringify(prompt.choices) !== JSON.stringify(["Run", "Cancel"]) || + (prompt.defaultId !== undefined && prompt.defaultId !== 1) + ) + return undefined; + const message = { + copyFile: "Copy the requested file or directory?", + writeFile: "Write content to the requested file?", + }[action.actionName]; + return message && prompt.message === message + ? { type: "question", selected: 0 } + : undefined; +} + +export function pendingFileAction(events) { + const pending = []; + for (const { event, detail } of events) { + if (event === "action.admitted") pending.push(detail); + else if (event === "action.completed") { + if ( + pending.length !== 1 || + pending[0].schemaName !== detail?.schemaName || + pending[0].actionName !== detail?.actionName + ) + return undefined; + pending.pop(); + } else if (event === "action.denied" || event === "action.failed") + return undefined; + } + return pending.length === 1 && + pending[0]?.schemaName === "powershell.powershell-files" + ? pending[0] + : undefined; +} + +export function consumeFixtureContinuation(approvals, args, stopped) { + const approval = approvals.get(args?.interactionId); + approvals.delete(args?.interactionId); + return ( + !stopped && + approval !== undefined && + approval.operationId === args.operationId && + approval.scopeId === args.scopeId && + args.response !== null && + typeof args.response === "object" && + Object.keys(approval.response).length === + Object.keys(args.response).length && + Object.entries(approval.response).every( + ([key, value]) => args.response[key] === value, + ) + ); +} + +export function fileConsentContext(tools, toolResults, events, request) { + const domainTools = tools.filter((tool) => !tool.name.includes("ask_user")); + const current = domainTools.at(-1); + const live = domainTools.filter((tool) => tool.endMs === undefined); + const currentResult = toolResults.findLast( + (tool) => tool.toolCallId === current?.toolCallId, + )?.result?.structuredContent; + const pending = + currentResult?.status === "requires_interaction" + ? currentResult + : undefined; + if ( + !current || + !/processCommand|executeAction|continueAction/.test(current.name) || + live.length > 1 || + (live.length === 1 && live[0] !== current) || + (!pending && live.length === 0) + ) + return {}; + const action = + pending?.prompt?.action ?? + pendingFileAction(events.slice(current.backendEventOffset)); + const handlerAnswer = fileHandlerConfirmation( + pending?.prompt ?? { + type: "question", + message: request.question, + choices: request.choices, + }, + action, + ); + return { action, pending, handlerAnswer }; +} + export function isClarificationQuestion(id, question) { if (/\b(confirm|approve|proceed|allow)\b/i.test(question)) return false; const subject = { @@ -301,7 +421,7 @@ export function isClarificationQuestion(id, question) { A2: /\b(report|file)\b/i, A3: /\b(pull request|PR|number)\b/i, A4: /\b(item|entry|line)\b/i, - A5: /\bfile\b/i, + A5: /\b(file|filename|file name)\b/i, }[id]; return Boolean( subject?.test(question) && @@ -316,6 +436,7 @@ export async function sendWithClarification({ testCase, canClarify, clarify, + isClarification = isClarificationQuestion, }) { const start = performance.now(); const first = await session.sendAndWait({ prompt }, timeoutMs); @@ -323,7 +444,7 @@ export async function sendWithClarification({ if ( canClarify() && testCase.clarification && - isClarificationQuestion(testCase.id, text) + isClarification(testCase.id, text) ) { const remaining = timeoutMs - (performance.now() - start); if (remaining <= 0) diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs index fbac377851..fe6e661d69 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs @@ -2,20 +2,114 @@ // Licensed under the MIT License. import { fileFixture } from "./ghcp-eval-corpus.mjs"; +import { + isGhcpEvalReadOnlyAction, + isGhcpEvalRecoverableReadError, +} from "../../dispatcher/dispatcher/dist/execute/ghcpEvalPolicy.js"; -export function terminalExecutionFailure(toolName, result, success) { +export function recoverableBackendReadFailure(events) { + const actions = events.filter(({ event }) => event.startsWith("action.")); + const admitted = actions.filter(({ event }) => event === "action.admitted"); + const completed = actions.filter( + ({ event }) => event === "action.completed", + ); + return ( + admitted.length > 0 && + admitted.length === completed.length && + actions.length === admitted.length + completed.length && + actions.every(({ detail }) => + isGhcpEvalReadOnlyAction(detail.schemaName, detail.actionName), + ) && + completed.some(({ detail }) => detail.success === false) && + completed.every( + ({ detail }) => + detail.success === true || detail.recoverable === true, + ) + ); +} + +function recoverableTypeAgentReadFailure( + toolName, + result, + success, + error, + backendEvents, +) { + const structured = result?.structuredContent; + const failedRead = + structured?.status === "failed" && + structured.error?.code === "execution_failed"; + // Never infer safety from a missing payload or an uncertain transport. + return ( + success === true && + error === undefined && + (structured?.status === undefined || failedRead) && + isGhcpEvalRecoverableReadError( + [ + structured?.error?.code, + structured?.error?.message, + result?.content, + ] + .filter((value) => typeof value === "string") + .join("\n"), + ) && + (failedRead || + (toolName.includes("processCommand") && + /^Error:(?:\s|$)/.test(result?.content ?? ""))) && + recoverableBackendReadFailure(backendEvents) + ); +} + +export function terminalExecutionFailure( + toolName, + result, + success, + error, + backendEvents = [], +) { if ( - /^(?:functions[.-])?(?:powershell|view|edit|create|glob|rg|web_fetch)$/.test( + /^(?:functions[.-])?stop_powershell$/.test(toolName) || + toolName.includes("cancelAction") + ) + return true; + if ( + /^(?:functions[.-])?(?:powershell|read_powershell|view|edit|create|glob|rg|web_fetch)$/.test( toolName, ) ) - return success === false; - if (!/processCommand|executeAction|continueAction/.test(toolName)) + return ( + success === false && + !( + /^(?:functions[.-])?(?:view|glob|rg|web_fetch)$/.test( + toolName, + ) && + isGhcpEvalRecoverableReadError( + typeof error?.message === "string" + ? `${error.code ?? ""}\n${error.message}` + : undefined, + ) + ) + ); + if ( + !/processCommand|executeAction|continueAction|cancelAction/.test( + toolName, + ) + ) + return false; + if ( + recoverableTypeAgentReadFailure( + toolName, + result, + success, + error, + backendEvents, + ) + ) return false; return ( success === false || (toolName.includes("processCommand") && - /^Error:\s/.test(result?.content ?? "")) || + /^Error:(?:\s|$)/.test(result?.content ?? "")) || ["failed", "cancelled", "unavailable", "execution_uncertain"].includes( result?.structuredContent?.status, ) @@ -90,16 +184,20 @@ export function preliminaryGrade(result, evidence) { }; if ( !result.grade.fileStateMatchesOracle || - !result.grade.listStateUnchanged + !(result.category === "lists" + ? result.grade.listStateMatchesOracle + : result.grade.listStateUnchanged) ) return { outcome: "failed", reason: "independent_fixture_oracle_mismatch", }; if ( - result.caseId.startsWith("A") && + (result.caseId.startsWith("A") || result.caseId.startsWith("list-A")) && (!result.grade.clarificationRequested || - !result.grade.noPrematureFileMutation) + !result.grade.noPrematureFileMutation || + (result.category === "lists" && + !result.grade.noPrematureListMutation)) ) return { outcome: "failed", @@ -129,6 +227,11 @@ export function preliminaryGrade(result, evidence) { A2: ["charger", "adapter"], A5: ["destination", "mountain", "jacket", "required"], }[result.caseId]; + if (required?.some((term) => typeof term !== "string")) + return { + outcome: "pending_review", + reason: "independent_answer_evidence_unavailable", + }; const missing = required?.filter( (term) => !answer.toLowerCase().includes(term.toLowerCase()), diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-list-preflight.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-list-preflight.mjs new file mode 100644 index 0000000000..4e55aa8737 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-list-preflight.mjs @@ -0,0 +1,121 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import assert from "node:assert/strict"; +import { isDeepStrictEqual } from "node:util"; +import { readLists } from "./ghcp-eval-lists.mjs"; + +export const listRequiredContracts = [ + ...[ + "listLists", + "getList", + "createList", + "addItems", + "removeItems", + "clearList", + ].map((action) => ["list", action]), + ["github-cli", "issueView"], + ["powershell.powershell-files", "readFile"], +]; + +export function preflightListPolicy(store) { + return { + version: 1, + category: "lists", + store, + enabled: true, + rules: [ + { actionName: "listLists" }, + ...["getList", "createList", "clearList"].map((actionName) => ({ + actionName, + listName: "preflight", + })), + ...["addItems", "removeItems"].map((actionName) => ({ + actionName, + listName: "preflight", + items: ["milk", "eggs"], + })), + ], + externalReads: [ + { + schemaName: "github-cli", + actionName: "issueView", + repo: "microsoft/TypeAgent", + number: 2617, + }, + ], + }; +} + +export async function verifyListPreflight(client, scopeId, env, result) { + const invoke = async (actionName, parameters) => { + const action = { schemaName: "list", actionName, parameters }; + let response = await client.callTool( + { + name: "typeagent-executeAction", + arguments: { protocolVersion: 1, scopeId, ...action }, + }, + undefined, + { timeout: 60_000 }, + ); + const pending = response.structuredContent; + if ( + pending?.status === "requires_interaction" && + pending.scopeId === scopeId && + pending.prompt?.type === "confirmation" && + pending.prompt.action?.schemaName === "list" && + pending.prompt.action.actionName === actionName && + isDeepStrictEqual(pending.prompt.action.parameters, parameters) + ) { + response = await client.callTool( + { + name: "typeagent-continueAction", + arguments: { + protocolVersion: 1, + scopeId, + operationId: pending.operationId, + interactionId: pending.interactionId, + response: { type: "confirmation", approved: true }, + }, + }, + undefined, + { timeout: 60_000 }, + ); + } + assert.notEqual(response.isError, true); + assert.equal(response.structuredContent?.status, "completed"); + result.externalEvidence.push({ + schemaName: "list", + actionName, + capturedAt: new Date().toISOString(), + outcome: response.structuredContent, + }); + }; + await invoke("listLists", {}); + const stores = fs + .readdirSync(env.TYPEAGENT_USER_DATA_DIR, { recursive: true }) + .filter((name) => path.basename(name) === "lists.json") + .map((name) => path.join(env.TYPEAGENT_USER_DATA_DIR, name)); + assert.equal(stores.length, 1); + const store = stores[0]; + assert.deepEqual(readLists(store), {}); + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_LIST_POLICY, + JSON.stringify(preflightListPolicy(store)), + ); + await invoke("createList", { listName: "preflight" }); + assert.deepEqual(readLists(store), { preflight: [] }); + await invoke("addItems", { + listName: "preflight", + items: ["milk", "eggs"], + }); + assert.deepEqual(readLists(store), { preflight: ["eggs", "milk"] }); + await invoke("getList", { listName: "preflight" }); + await invoke("removeItems", { listName: "preflight", items: ["milk"] }); + assert.deepEqual(readLists(store), { preflight: ["eggs"] }); + await invoke("clearList", { listName: "preflight" }); + assert.deepEqual(readLists(store), { preflight: [] }); + return store; +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-lists.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-lists.mjs new file mode 100644 index 0000000000..3f69272bba --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-lists.mjs @@ -0,0 +1,273 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { + listFixture, + normalizeLists, + fileFixture, +} from "./ghcp-eval-corpus.mjs"; +import { fileStateMatches } from "./ghcp-eval-files.mjs"; + +export const listCorpusVersion = "lists-v1"; +export const listCandidates = [1, 2, 3, 4]; +export const listPreparation = + "Discover contracts for list inventory, reading, adding/removing items, creating and clearing lists, read-only GitHub issue details and reading files. Do not execute actions, inspect contents, establish preferred targets, or guess future requests."; + +export function buildListCorpus(files, repo, issue) { + return [ + ["S1", "Show my lists."], + ["S2", "Show my grocery list."], + ["S3", "Remove milk from my grocery list, preserving the other items."], + ["S4", "Add apples to my grocery list."], + ["S5", "Create a new empty list named book."], + ["M1", "Show my grocery list and my pantry list, labeling each."], + [ + "M2", + "Add tea and coffee to my office list, then remove notebook. Preserve its other items.", + ], + ["M3", "Empty my grocery list, then add bread and oranges to it."], + [ + "M4", + "Create a new list named picnic, then add blanket and water to it.", + ], + [ + "M5", + `Show issue ${issue} in ${repo}, then add "review issue ${issue}" to my errand list.`, + ], + ["R1", "Which items are on both my grocery list and my pantry list?"], + [ + "R2", + "Compare my packing and pantry lists. Which has more items, and by how many?", + ], + [ + "R3", + "Add the items from my travel list that are missing from my packing list to packing. Preserve existing items and tell me what you added.", + ], + [ + "R4", + `Read ${path.join(files, "trip.txt")}. If it says a jacket is required, add jacket to my packing list; otherwise leave the list unchanged. Tell me what you did.`, + ], + [ + "R5", + `Read issue ${issue} in ${repo} and add its exact title to my errand list, but only if that title is not already there.`, + ], + ["A1", "Add apples to my list.", "The grocery list."], + ["A2", "Show that list.", "The pantry list."], + ["A3", "Remove charger from one of my lists.", "The travel list."], + [ + "A4", + "Remove the item from my grocery list.", + "Remove milk, keeping the other items.", + ], + ["A5", "Empty my list, but keep the list itself.", "The office list."], + ].map(([id, prompt, clarification]) => ({ + id: `list-${id}`, + category: "lists", + cohort: id[0], + prompt, + ...(clarification ? { clarification } : {}), + })); +} + +export function expectedLists(id, issue = 2617, title) { + const lists = structuredClone(listFixture); + const add = (name, items) => { + lists[name] = [...new Set([...(lists[name] ?? []), ...items])]; + }; + if (["list-S4", "list-A1"].includes(id)) add("grocery", ["apples"]); + if (["list-S3", "list-A4"].includes(id)) + lists.grocery = lists.grocery.filter((item) => item !== "milk"); + if (id === "list-S5") lists.book = []; + if (id === "list-M2") lists.office = ["charger", "pen", "tea", "coffee"]; + if (id === "list-M3") lists.grocery = ["bread", "oranges"]; + if (id === "list-M4") lists.picnic = ["blanket", "water"]; + if (id === "list-M5") add("errand", [`review issue ${issue}`]); + if (id === "list-R3") add("packing", ["adapter"]); + if (id === "list-R4") add("packing", ["jacket"]); + if (id === "list-R5") { + if (!title) + throw new Error( + "List title oracle requires independent issue evidence", + ); + add("errand", [title]); + } + if (id === "list-A3") lists.travel = ["adapter"]; + if (id === "list-A5") lists.office = []; + return normalizeLists( + Object.entries(lists).map(([name, items]) => ({ name, items })), + ); +} + +export function readLists(store) { + const stat = fs.lstatSync(store); + if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1) + throw new Error("Unsafe evaluation list store"); + const lists = JSON.parse(fs.readFileSync(store, "utf8")); + if ( + !Array.isArray(lists) || + lists.some( + (entry) => + typeof entry.name !== "string" || + !Array.isArray(entry.items) || + entry.items.some((item) => typeof item !== "string"), + ) || + new Set(lists.map((entry) => entry.name)).size !== lists.length + ) + throw new Error("Invalid evaluation list snapshot"); + return normalizeLists(lists); +} + +export function resetLists(store) { + readLists(store); + fs.writeFileSync( + store, + JSON.stringify( + Object.entries(listFixture).map(([name, items]) => ({ + name, + items, + })), + ), + ); +} + +export function listsUnchanged(store) { + return ( + JSON.stringify(readLists(store)) === + JSON.stringify(expectedLists("list-S1")) + ); +} + +export function listFilePolicy(id) { + return { + version: 1, + readFiles: id === "list-R4" ? ["trip.txt"] : [], + writeFiles: [], + readsEnabled: id === "list-R4", + writesEnabled: false, + allowInventory: false, + prerequisites: {}, + }; +} + +export function listPolicy(id, store, clarified = false, title, issue = 2617) { + const rule = (actionName, listName, items, before) => ({ + actionName, + ...(listName ? { listName } : {}), + ...(items ? { items } : {}), + ...(before ? { before } : {}), + }); + const reads = { + S1: [], + S2: ["grocery"], + S3: ["grocery"], + S4: ["grocery"], + S5: ["book"], + M1: ["grocery", "pantry"], + M2: ["office"], + M3: ["grocery"], + M4: ["picnic"], + M5: ["errand"], + R1: ["grocery", "pantry"], + R2: ["packing", "pantry"], + R3: ["travel", "packing"], + R4: ["packing"], + R5: ["errand"], + A1: ["grocery"], + A2: ["pantry"], + A3: ["travel"], + A4: ["grocery"], + A5: ["office"], + }; + const key = id.replace(/^list-/, ""); + if (!id.startsWith("list-") || !Object.hasOwn(reads, key)) + throw new Error("Unknown list case"); + const rules = reads[key].map((name) => rule("getList", name)); + if (key === "S1") rules.push(rule("listLists")); + if (["S4", "A1"].includes(key)) + rules.push(rule("addItems", "grocery", ["apples"])); + if (["S3", "A4"].includes(key)) + rules.push(rule("removeItems", "grocery", ["milk"])); + if (key === "S5") + rules.push(rule("createList", "book", undefined, { book: null })); + if (key === "M2") + rules.push( + rule("addItems", "office", ["tea", "coffee"]), + rule("removeItems", "office", ["notebook"], { + office: [...listFixture.office, "tea", "coffee"], + }), + ); + if (key === "M3") + rules.push( + rule("clearList", "grocery", undefined, { + grocery: listFixture.grocery, + }), + ...[[], ["bread"], ["oranges"]].map((before) => + rule("addItems", "grocery", ["bread", "oranges"], { + grocery: before, + }), + ), + ); + if (key === "M4") + rules.push( + rule("createList", "picnic", undefined, { picnic: null }), + ...[[], ["blanket"], ["water"]].map((before) => + rule("addItems", "picnic", ["blanket", "water"], { + picnic: before, + }), + ), + ); + if (key === "M5") + rules.push(rule("addItems", "errand", [`review issue ${issue}`])); + if (key === "R3") rules.push(rule("addItems", "packing", ["adapter"])); + if (key === "R4") rules.push(rule("addItems", "packing", ["jacket"])); + if (key === "R5") { + if (!title) + throw new Error("List policy requires independent issue title"); + rules.push(rule("addItems", "errand", [title])); + } + if (key === "A3") rules.push(rule("removeItems", "travel", ["charger"])); + if (key === "A5") rules.push(rule("clearList", "office")); + return { + version: 1, + category: "lists", + store, + enabled: !key.startsWith("A") || clarified, + rules, + externalReads: ["M5", "R5"].includes(key) + ? [ + { + schemaName: "github-cli", + actionName: "issueView", + repo: "microsoft/TypeAgent", + number: issue, + }, + ] + : [], + }; +} + +export function listClarificationQuestion(id, question) { + if (/\b(confirm|approve|proceed|allow)\b/i.test(question)) return false; + const subject = id === "list-A4" ? /\b(item|entry)\b/i : /\blist\b/i; + return ( + subject.test(question) && + /\b(which|what|choose|specify|mean)\b/i.test(question) + ); +} + +export function gradeListTrial(result, store, files, title) { + const lists = readLists(store); + return { + listStateMatchesOracle: + JSON.stringify(lists) === + JSON.stringify(expectedLists(result.caseId, 2617, title)), + fileStateMatchesOracle: fileStateMatches(files, fileFixture), + noPrematureListMutation: + result.noPrematureListMutation === true && + (!result.caseId.startsWith("list-A") || + JSON.stringify(result.listsAtClarification) === + JSON.stringify(expectedLists("list-S1"))), + }; +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs index 79678220d1..e9863f994c 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs @@ -7,11 +7,23 @@ import path from "node:path"; import { fileURLToPath } from "node:url"; import { createHash } from "node:crypto"; import { - corpusVersion, fileFixture, filePolicy, + fileHandlerConfirmation, logicalFileContent, + protocolVersion, } from "./ghcp-eval-corpus.mjs"; +import { evaluationCategory } from "./ghcp-eval-categories.mjs"; +import { + listFilePolicy, + resetLists, + listsUnchanged, +} from "./ghcp-eval-lists.mjs"; +import { + listRequiredContracts, + preflightListPolicy, + verifyListPreflight, +} from "./ghcp-eval-list-preflight.mjs"; import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; @@ -28,8 +40,15 @@ const root = path.resolve( path.dirname(fileURLToPath(import.meta.url)), "../../..", ); -const [outputDirectory, configDirectory, ledgerPath, evidenceMode] = - process.argv.slice(2); +const [ + outputDirectory, + configDirectory, + ledgerPath, + evidenceMode, + categoryName = "common-files", +] = process.argv.slice(2); +const category = evaluationCategory(categoryName); +const listCategory = category.name === "lists"; if (!outputDirectory || !configDirectory || !ledgerPath) { throw new Error( "Usage: node ghcp-eval-preflight.mjs ", @@ -53,25 +72,44 @@ for (const [name, content] of Object.entries(fileFixture)) { fs.writeFileSync(path.join(fixtures, name), content); } env.TYPEAGENT_GHCP_EVAL_FIXTURES = fixtures; +env.TYPEAGENT_GHCP_EVAL_CATEGORY = category.name; +delete env.TYPEAGENT_GHCP_EVAL_LIST_POLICY; +delete env.TYPEAGENT_GHCP_EVAL_CORRELATION; +if (listCategory) { + env.TYPEAGENT_GHCP_EVAL_LIST_POLICY = path.resolve( + outputDirectory, + "list-policy.json", + ); + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_LIST_POLICY, + JSON.stringify(preflightListPolicy("")), + ); +} env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = path.resolve( outputDirectory, "file-policy.json", ); fs.writeFileSync( env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, - JSON.stringify({ - ...filePolicy("M3"), - allowInventory: true, - allowListInventory: true, - prerequisites: {}, - }), + JSON.stringify( + listCategory + ? listFilePolicy("list-R4") + : { + ...filePolicy("M3"), + allowInventory: true, + allowListInventory: true, + prerequisites: {}, + }, + ), ); fs.mkdirSync(env.TYPEAGENT_PLUGIN_DATA); stageCopilotPlugin(path.join(outputDirectory, "plugin")); const result = { kind: "catalog_preflight_not_eval", model: evalModel, - corpusVersion, + corpusVersion: category.corpusVersion, + protocolVersion, + category: category.name, status: "running", contracts: [], missing: [], @@ -87,6 +125,7 @@ const stdout = fs.openSync( const stderr = fs.openSync(log, "a"); let server; let client; +let verifiedListStore; try { server = startProcess( process.execPath, @@ -114,18 +153,20 @@ try { stderr: "inherit", }), ); - const required = [ - ["list", "listLists"], - ["github-cli", "prFiles"], - ["github-cli", "prChecks"], - ["github-cli", "issueView"], - ["powershell.powershell-files", "readFile"], - ["powershell.powershell-files", "listFiles"], - ["powershell.powershell-files", "writeFile"], - ["powershell.powershell-files", "copyFile"], - ["ipconfig", "displayFullConfigurationInformation"], - ["ipconfig", "displayDNSResolverCacheContents"], - ]; + const required = listCategory + ? listRequiredContracts + : [ + ["list", "listLists"], + ["github-cli", "prFiles"], + ["github-cli", "prChecks"], + ["github-cli", "issueView"], + ["powershell.powershell-files", "readFile"], + ["powershell.powershell-files", "listFiles"], + ["powershell.powershell-files", "writeFile"], + ["powershell.powershell-files", "copyFile"], + ["ipconfig", "displayFullConfigurationInformation"], + ["ipconfig", "displayDNSResolverCacheContents"], + ]; let scopeId; for (const [schemaName, actionName] of required) { let contract; @@ -167,13 +208,24 @@ try { } result.status = result.missing.length === 0 ? "passed" : "blocked"; if (result.status === "passed" && evidenceMode === "--external-evidence") { - const requests = [ - ["prFiles", 3058], - ["prChecks", 3058], - ["prFiles", 3067], - ["prChecks", 3067], - ["issueView", 2617], - ].map(([actionName, number]) => ({ + if (listCategory) + verifiedListStore = await verifyListPreflight( + client, + scopeId, + env, + result, + ); + const requests = ( + listCategory + ? [["issueView", 2617]] + : [ + ["prFiles", 3058], + ["prChecks", 3058], + ["prFiles", 3067], + ["prChecks", 3067], + ["issueView", 2617], + ] + ).map(([actionName, number]) => ({ schemaName: "github-cli", actionName, parameters: { @@ -184,44 +236,51 @@ try { : {}), }, })); - requests.push( - { schemaName: "list", actionName: "listLists", parameters: {} }, - { - schemaName: "powershell.powershell-files", - actionName: "listFiles", - parameters: { path: fixtures }, - }, - { + if (listCategory) + requests.push({ schemaName: "powershell.powershell-files", actionName: "readFile", - parameters: { path: path.join(fixtures, "report-a.txt") }, - }, - { - schemaName: "powershell.powershell-files", - actionName: "copyFile", - parameters: { - source: path.join(fixtures, "grocery.txt"), - destination: path.join(fixtures, "grocery-backup.txt"), + parameters: { path: path.join(fixtures, "trip.txt") }, + }); + else + requests.push( + { schemaName: "list", actionName: "listLists", parameters: {} }, + { + schemaName: "powershell.powershell-files", + actionName: "listFiles", + parameters: { path: fixtures }, }, - }, - { - schemaName: "powershell.powershell-files", - actionName: "writeFile", - parameters: { - path: path.join(fixtures, "grocery.txt"), - content: "apples", - append: true, + { + schemaName: "powershell.powershell-files", + actionName: "readFile", + parameters: { path: path.join(fixtures, "report-a.txt") }, }, - }, - ...[ - "displayFullConfigurationInformation", - "displayDNSResolverCacheContents", - ].map((actionName) => ({ - schemaName: "ipconfig", - actionName, - parameters: {}, - })), - ); + { + schemaName: "powershell.powershell-files", + actionName: "copyFile", + parameters: { + source: path.join(fixtures, "grocery.txt"), + destination: path.join(fixtures, "grocery-backup.txt"), + }, + }, + { + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { + path: path.join(fixtures, "grocery.txt"), + content: "apples", + append: true, + }, + }, + ...[ + "displayFullConfigurationInformation", + "displayDNSResolverCacheContents", + ].map((actionName) => ({ + schemaName: "ipconfig", + actionName, + parameters: {}, + })), + ); for (const action of requests) { let response = await client.callTool( { @@ -260,7 +319,34 @@ try { { timeout: 60_000 }, ); } + const question = response.structuredContent; + const handlerAnswer = fileHandlerConfirmation( + question?.prompt, + action, + ); + if ( + question?.status === "requires_interaction" && + question.scopeId === scopeId && + question.operationId === pending?.operationId && + handlerAnswer + ) { + response = await client.callTool( + { + name: "typeagent-continueAction", + arguments: { + protocolVersion: 1, + scopeId, + operationId: question.operationId, + interactionId: question.interactionId, + response: handlerAnswer, + }, + }, + undefined, + { timeout: 60_000 }, + ); + } result.externalEvidence.push({ + schemaName: action.schemaName, actionName: action.actionName, number: action.parameters.number, capturedAt: new Date().toISOString(), @@ -318,6 +404,18 @@ try { if (client) await client.close(); } finally { if (server) await stopProcess(server); + if (verifiedListStore) { + try { + resetLists(verifiedListStore); + if (!listsUnchanged(verifiedListStore)) + throw new Error("List preflight reset failed"); + result.listOperationsVerified = result.status === "passed"; + } catch (error) { + result.status = "failed"; + result.error = String(error); + process.exitCode = 1; + } + } fs.writeFileSync( path.join(outputDirectory, "result.json"), JSON.stringify(result, null, 2) + "\n", diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-report.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-report.mjs new file mode 100644 index 0000000000..9333c2c154 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-report.mjs @@ -0,0 +1,232 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import { createHash } from "node:crypto"; +import { pathToFileURL } from "node:url"; +import { protocolVersion } from "./ghcp-eval-corpus.mjs"; +import { + evaluationCategory, + assertCategoryResults, +} from "./ghcp-eval-categories.mjs"; +import { percentile, preliminaryGrade } from "./ghcp-eval-grade.mjs"; + +function latency(rows, select = (row) => row.e2eMs) { + const values = rows + .map(select) + .filter((value) => Number.isFinite(value) && value >= 0); + return { + population: rows.length, + measured: values.length, + missing: rows.length - values.length, + p50Ms: percentile(values, 0.5), + p90Ms: percentile(values, 0.9), + p95Ms: percentile(values, 0.95), + }; +} + +const key = (row) => `${row.repetition}:${row.caseId}:${row.candidate}`; +const pair = (row) => `${row.repetition}:${row.caseId}`; + +function reviewedRows(spec, results, reviews, category) { + if ( + spec.protocolVersion !== protocolVersion || + spec.category !== category.name || + !Number.isInteger(spec.repetitions) || + spec.repetitions < 1 || + spec.corpusVersion !== category.corpusVersion + ) + throw new Error("Cannot pool or relabel historical protocols"); + assertCategoryResults(results, spec.order, category); + const expected = new Set(spec.order.map(key)); + const expectedCount = 20 * category.candidates.length * spec.repetitions; + if ( + expected.size !== spec.order.length || + results.length !== spec.order.length || + spec.order.length !== expectedCount || + spec.order.some( + (row) => + row.category !== category.name || + !category.candidates.includes(row.candidate) || + !Number.isInteger(row.repetition) || + row.repetition < 0 || + row.repetition >= spec.repetitions || + !( + category.name === "lists" + ? /^list-[SMRA][1-5]$/ + : /^[SMRA][1-5]$/ + ).test(row.caseId), + ) + ) + throw new Error("A complete category schedule is required"); + const decisions = new Map(reviews.map((review) => [key(review), review])); + if ( + decisions.size !== reviews.length || + reviews.some( + (review) => + review.category !== category.name || + !expected.has(key(review)) || + !["success", "failed", "unknown"].includes(review.outcome), + ) + ) + throw new Error("Invalid or cross-category review"); + return results.map((result) => { + const review = decisions.get(key(result)); + const outcome = review?.outcome ?? "unknown"; + if ( + outcome === "success" && + (!review.reason?.trim() || + !Array.isArray(review.evidence) || + !review.evidence.length || + !review.evidence.every( + (value) => typeof value === "string" && value.trim(), + ) || + ["failed", "incomplete"].includes( + preliminaryGrade(result, {}).outcome, + )) + ) + throw new Error( + `Success contradicts trace/state guard or lacks review evidence: ${key(result)}`, + ); + return { ...result, outcome }; + }); +} + +function candidateSummary(rows, candidate, common) { + const own = rows.filter((row) => row.candidate === candidate); + const successes = own.filter((row) => row.outcome === "success"); + const preparation = candidate === 4 ? own : []; + const prepMs = preparation.every( + (row) => Number.isFinite(row.preparationMs) && row.preparationMs >= 0, + ) + ? preparation.reduce((sum, row) => sum + row.preparationMs, 0) + : null; + return { + candidate, + denominator: own.length, + successes: successes.length, + failed: own.filter((row) => row.outcome === "failed").length, + unknown: own.filter((row) => row.outcome === "unknown").length, + e2e: latency(own), + successfulE2e: latency(successes), + commonSuccessE2e: latency(own.filter((row) => common.has(pair(row)))), + preparation: { + ...latency(preparation, (row) => row.preparationMs), + totalMs: prepMs, + amortizedPerTrialMs: + own.length && prepMs !== null ? prepMs / own.length : null, + }, + includingPreparation: latency(own, (row) => + Number.isFinite(row.e2eMs) && + (candidate !== 4 || Number.isFinite(row.preparationMs)) + ? row.e2eMs + (row.preparationMs ?? 0) + : null, + ), + cohorts: Object.fromEntries( + ["S", "M", "R", "A"].map((cohort) => { + const cohortRows = own.filter((row) => + row.caseId.replace(/^list-/, "").startsWith(cohort), + ); + return [ + cohort, + { + denominator: cohortRows.length, + successes: cohortRows.filter( + (row) => row.outcome === "success", + ).length, + unknown: cohortRows.filter( + (row) => row.outcome === "unknown", + ).length, + failed: cohortRows.filter( + (row) => row.outcome === "failed", + ).length, + e2e: latency(cohortRows), + }, + ]; + }), + ), + }; +} + +function pairwiseLatency(rows, candidates) { + return candidates.flatMap((first, i) => + candidates.slice(i + 1).map((second) => { + const successful = rows.filter((row) => row.outcome === "success"); + const firstRows = successful.filter( + (row) => row.candidate === first, + ); + const secondByPair = new Map( + successful + .filter((row) => row.candidate === second) + .map((row) => [pair(row), row]), + ); + const paired = firstRows.filter((row) => + secondByPair.has(pair(row)), + ); + return { + candidates: [first, second], + pairs: paired.map(pair).sort(), + first: latency(paired), + second: latency( + paired.map((row) => secondByPair.get(pair(row))), + ), + }; + }), + ); +} + +export function summarizeCategory(spec, results, reviews) { + const category = evaluationCategory(spec.category); + const rows = reviewedRows(spec, results, reviews, category); + const common = new Set( + rows.filter((row) => row.outcome === "success").map(pair), + ); + for (const id of common) { + const successful = rows.filter( + (row) => pair(row) === id && row.outcome === "success", + ); + if ( + new Set(successful.map((row) => row.candidate)).size !== + category.candidates.length + ) + common.delete(id); + } + return { + protocolVersion, + category: category.name, + corpusVersion: category.corpusVersion, + denominator: rows.length, + commonSuccessPairs: [...common].sort(), + pairwiseCommonSuccess: pairwiseLatency(rows, category.candidates), + percentileMethod: + "nearest rank; null for no observed latency; missing timings never zero", + limitation: + "Exploratory repetitions, no powered winner; review faithfulness independently. Do not pool categories.", + candidates: category.candidates.map((candidate) => + candidateSummary(rows, candidate, common), + ), + }; +} + +if ( + process.argv[1] && + import.meta.url === pathToFileURL(process.argv[1]).href +) { + const [specFile, resultFile, reviewFile, output] = process.argv.slice(2); + if (!output) + throw new Error( + "Usage: ghcp-eval-report.mjs ", + ); + const sources = [specFile, resultFile, reviewFile].map((file) => + fs.readFileSync(file), + ); + const summary = summarizeCategory( + ...sources.map((buffer) => JSON.parse(buffer.toString("utf8"))), + ); + summary.sourceSha256 = sources.map((buffer) => + createHash("sha256").update(buffer).digest("hex"), + ); + fs.writeFileSync(output, JSON.stringify(summary, null, 2) + "\n", { + flag: "wx", + }); +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs index 8f40b18d70..417461a1aa 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs @@ -10,6 +10,13 @@ import { randomUUID, createHash } from "node:crypto"; import { execFileSync } from "node:child_process"; import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; +import { + candidateToolBoundary, + pendingInteractionGate, + callCorrelation, + toolEvidenceViews, + executionRouteViolation, +} from "./ghcp-eval-boundary.mjs"; import { snapshotFiles, fileStateMatches, @@ -17,22 +24,40 @@ import { restoreFiles, } from "./ghcp-eval-files.mjs"; import { ghcpEvalNativeFilePermission } from "../../dispatcher/dispatcher/dist/execute/ghcpEvalFiles.js"; +import { ghcpEvalListActionAllowed } from "../../dispatcher/dispatcher/dist/execute/ghcpEvalLists.js"; +import { + evaluationCategory, + categoryCorpus, + categorySchedule, + assertCategoryReadiness, + assertCategoryResults, +} from "./ghcp-eval-categories.mjs"; +import { + listPolicy, + listFilePolicy, + listClarificationQuestion, + readLists, + resetLists, + listsUnchanged, + gradeListTrial, +} from "./ghcp-eval-lists.mjs"; import { CopilotCreditBudget } from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; import { registerGhcpEvalArtifact, isGhcpEvalArtifact, } from "../../dispatcher/dispatcher/dist/execute/ghcpEvalArtifacts.js"; import { - balancedOrder, - buildCorpus, - corpusVersion, + assertFrozenSpecification, assertCorpusReadiness, filePolicy, + fileConsentContext, + consumeFixtureContinuation, fileFixture, fixtureConfirmationAllowed, isClarificationQuestion, listFixture, normalizeLists, + protocolVersion, shuffled, sendWithClarification, } from "./ghcp-eval-corpus.mjs"; @@ -61,14 +86,20 @@ const [ outputDirectory, configDirectory, ledgerPath, - selection = "1,2,3,4,5,6,7", + selectionText, phase = "pilot", evidencePath, - pilotCases = "S1", + pilotCasesText, batchStartText = "0", - batchSizeText = "7", + batchSizeText, repetitionsText = "1", + categoryName = "common-files", ] = process.argv.slice(2); +const category = evaluationCategory(categoryName); +const corpusVersion = category.corpusVersion; +const listCategory = category.name === "lists"; +const selection = selectionText ?? category.candidates.join(","); +const pilotCases = pilotCasesText ?? (listCategory ? "list-S1" : "S1"); if ( !cliPath || !template || @@ -83,7 +114,11 @@ if ( const fixtures = listFixture; const creditLedger = JSON.parse(fs.readFileSync(ledgerPath, "utf8")); validateEvalLedger(creditLedger); -assertCorpusReadiness( +function assertReadiness(readiness) { + assertCategoryReadiness(readiness, category); + if (!listCategory) assertCorpusReadiness(readiness); +} +assertReadiness( JSON.parse(fs.readFileSync(path.join(template, "result.json"), "utf8")), ); const { getCopilotPermissionDefault } = await import( @@ -93,6 +128,12 @@ const seededLists = Object.entries(fixtures).map(([name, items]) => ({ name, items, })); +const caseFilePolicy = (id, clarified = false) => + listCategory ? listFilePolicy(id) : filePolicy(id, clarified); +const clarificationQuestion = (id, question) => + listCategory + ? listClarificationQuestion(id, question) + : isClarificationQuestion(id, question); const port = 19024; const excludedSchemas = Object.keys( JSON.parse( @@ -148,6 +189,46 @@ const candidates = [ { id: 7, policy: "Native only" }, ]; +function caseConsentAllowed( + action, + testCase, + store, + workspace, + clarified, + evidence, + env, +) { + if ( + action?.schemaName === "powershell.powershell-files" && + action.actionName === "readFile" && + typeof action.parameters?.path === "string" && + isGhcpEvalArtifact( + action.parameters.path, + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, + ) + ) + return true; + if (!listCategory) + return fixtureConfirmationAllowed( + testCase.id, + action, + workspace, + evidence?.issueTitle, + ); + if (action?.schemaName === "list") + return ghcpEvalListActionAllowed( + action.actionName, + action.parameters, + listPolicy(testCase.id, store, clarified, evidence?.issueTitle), + ); + return ( + testCase.id === "list-R4" && + action?.schemaName === "powershell.powershell-files" && + action.actionName === "readFile" && + action.parameters?.path === path.join(workspace, "trip.txt") + ); +} + function findListStores(directory) { return fs .readdirSync(directory, { recursive: true }) @@ -155,16 +236,20 @@ function findListStores(directory) { .map((name) => path.join(directory, name)); } -function collectObservations(result, tracePath) { +function readBackendEvents(tracePath) { const trace = fs.existsSync(tracePath) ? fs.readFileSync(tracePath, "utf8") : ""; - result.typeagentEvents = trace.trim() + return trace.trim() ? trace .trim() .split("\n") .map((line) => JSON.parse(line)) : []; +} + +function collectObservations(result, tracePath) { + result.typeagentEvents = readBackendEvents(tracePath); result.fallback = result.candidate === 7 ? null @@ -251,20 +336,24 @@ function gradeCompletedTrial({ }) { const after = JSON.parse(fs.readFileSync(store, "utf8")); const finalFiles = snapshotFiles(workspace); - const correctNames = Object.keys(fileFixture).every((name) => - result.answer.includes(name), - ); + const correctNames = Object.keys( + listCategory ? listFixture : fileFixture, + ).every((name) => result.answer.includes(name)); result.grade = { - fileStateMatchesOracle: gradeFileState( - testCase.id, - finalFiles, - 2617, - evidence?.issueTitle, - ), + fileStateMatchesOracle: listCategory + ? fileStateMatches(finalFiles, fileFixture) + : gradeFileState( + testCase.id, + finalFiles, + 2617, + evidence?.issueTitle, + ), listStateUnchanged: JSON.stringify(normalizeLists(after)) === JSON.stringify(normalizeLists(seededLists)), - containsAllNames: testCase.id === "S1" ? correctNames : null, + containsAllNames: ["S1", "list-S1"].includes(testCase.id) + ? correctNames + : null, clarificationRequested: testCase.clarification ? clarificationGiven : null, @@ -274,6 +363,9 @@ function gradeCompletedTrial({ fileStateMatches(result.stateAtClarification, fileFixture) : null, requiresManualFaithfulnessCheck: true, + ...(listCategory + ? gradeListTrial(result, store, finalFiles, evidence?.issueTitle) + : {}), }; result.finalLists = normalizeLists(after); result.finalFiles = finalFiles; @@ -281,14 +373,16 @@ function gradeCompletedTrial({ if ( phase === "pilot" && result.candidate !== 7 && - ((testCase.id === "S1" && !correctNames) || + ((["S1", "list-S1"].includes(testCase.id) && !correctNames) || !result.grade.fileStateMatchesOracle || - !result.grade.listStateUnchanged) + !(listCategory + ? result.grade.listStateMatchesOracle + : result.grade.listStateUnchanged)) ) result.status = "pilot_needs_review"; } -function prepareTrial(candidate, directory, workspace, testCase) { +function prepareTrial(candidate, directory, workspace, testCase, evidence) { fs.mkdirSync(directory); const { env, mcp } = makeConfiguration( directory, @@ -305,17 +399,24 @@ function prepareTrial(candidate, directory, workspace, testCase) { env.TYPEAGENT_REASONING_TIMEOUT_MS = "90000"; env.DEBUG = "typeagent:request"; env.TYPEAGENT_GHCP_EVAL_FIXTURES = workspace; + env.TYPEAGENT_GHCP_EVAL_CATEGORY = category.name; + delete env.TYPEAGENT_GHCP_EVAL_LIST_POLICY; env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = path.resolve( directory, "file-policy.json", ); fs.writeFileSync( env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, - JSON.stringify(filePolicy(testCase.id)), + JSON.stringify(caseFilePolicy(testCase.id)), ); const sessionId = randomUUID(); env.TYPEAGENT_COPILOT_CREDIT_SESSION_SCOPE = sessionId; env.TYPEAGENT_GHCP_EVAL_TRACE = path.join(directory, "events.jsonl"); + env.TYPEAGENT_GHCP_EVAL_CORRELATION = path.join( + directory, + "call-correlation.json", + ); + fs.writeFileSync(env.TYPEAGENT_GHCP_EVAL_CORRELATION, "{}"); const temporaryRoot = path.resolve(directory, "sdk-temp"); fs.mkdirSync(temporaryRoot); env.TEMP = env.TMP = env.TMPDIR = temporaryRoot; @@ -350,7 +451,19 @@ function prepareTrial(candidate, directory, workspace, testCase) { const stores = findListStores(env.TYPEAGENT_USER_DATA_DIR); if (stores.length !== 1) throw new Error("Expected exactly one disposable list store"); - fs.writeFileSync(stores[0], JSON.stringify(seededLists)); + resetLists(stores[0]); + if (listCategory) { + env.TYPEAGENT_GHCP_EVAL_LIST_POLICY = path.resolve( + directory, + "list-policy.json", + ); + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_LIST_POLICY, + JSON.stringify( + listPolicy(testCase.id, stores[0], false, evidence?.issueTitle), + ), + ); + } const sessionDataPath = path.join( path.dirname(path.dirname(stores[0])), "data.json", @@ -414,6 +527,10 @@ function persistTrial({ result.totalIncludingSetupMs = performance.now() - started; try { result.finalFiles = snapshotFiles(workspace); + const stores = findListStores(env.TYPEAGENT_USER_DATA_DIR); + if (stores.length !== 1) + throw new Error("Expected one final list store"); + result.finalLists = readLists(stores[0]); } catch (error) { result.status = "harness_failed"; result.error = `Final fixture snapshot failed: ${String(error)}`; @@ -441,11 +558,13 @@ function persistTrial({ async function trial(candidate, directory, testCase, workspace, evidence) { const { env, config, stores, sessionDataPath, sessionData, sessionId } = - prepareTrial(candidate, directory, workspace, testCase); + prepareTrial(candidate, directory, workspace, testCase, evidence); const result = { phase, candidate: candidate.id, caseId: testCase.id, + category: category.name, + protocolVersion, corpusVersion, prompt: testCase.prompt, sessionId, @@ -459,7 +578,9 @@ async function trial(candidate, directory, testCase, workspace, evidence) { grade: null, permissions: [], toolResults: [], + recoverableToolFailures: [], noPrematureFileMutation: true, + noPrematureListMutation: true, }; let server; let client; @@ -472,18 +593,35 @@ async function trial(candidate, directory, testCase, workspace, evidence) { const network = ["S5", "M2"].includes(testCase.id) ? { toolResults: [] } : undefined; - const approvedInteractions = new Set(); + const approvedInteractions = new Map(); + const boundary = candidateToolBoundary(candidate, nativeTools); + const interactionGate = pendingInteractionGate(); + const consentResults = []; const clarify = (question, source) => { if (executionStopped || clarificationGiven) throw new Error("Clarification cannot replay stopped work"); result.stateAtClarification = snapshotFiles(workspace); + result.listsAtClarification = readLists(stores[0]); + if (!listsUnchanged(stores[0])) result.noPrematureListMutation = false; if (!fileStateMatches(result.stateAtClarification, fileFixture)) result.noPrematureFileMutation = false; clarificationGiven = true; fs.writeFileSync( env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, - JSON.stringify(filePolicy(testCase.id, true)), + JSON.stringify(caseFilePolicy(testCase.id, true)), ); + if (listCategory) + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_LIST_POLICY, + JSON.stringify( + listPolicy( + testCase.id, + stores[0], + true, + evidence?.issueTitle, + ), + ), + ); result.clarificationSource = source; result.clarificationQuestion = question; return testCase.clarification; @@ -560,15 +698,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { sessionLimits: { maxAiCredits: 60 }, workingDirectory: workspace, skipCustomInstructions: true, - availableTools: - candidate.id === 7 - ? nativeTools - : [ - "mcp:*", - ...(candidate.id >= 5 - ? nativeTools - : ["builtin:ask_user"]), - ], + availableTools: boundary.availableTools, ...(candidate.id === 7 ? {} : { mcpServers: { "typeagent-e2e": config } }), @@ -577,7 +707,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { : { systemMessage: { mode: "append", - content: `${candidate.policy}. Use only the exposed TypeAgent MCP interface. Preserve confirmation and clarification; never replay failed or uncertain effects through another route.`, + content: `${candidate.policy}. Use only the exposed TypeAgent MCP interface. Preserve confirmation and clarification. Ordinary read failures may be corrected within this route; denials, cancellation and uncertain effects never authorize replay or a permission bypass.`, }, }), onPermissionRequest: (request) => { @@ -585,7 +715,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { kind: request.kind, readOnly: request.readOnly, }); - const policy = filePolicy(testCase.id, clarificationGiven); + const policy = caseFilePolicy(testCase.id, clarificationGiven); if ( executionStopped || (preparation && request.kind === "write") @@ -633,12 +763,11 @@ async function trial(candidate, directory, testCase, workspace, evidence) { onUserInputRequest: (request) => { result.interactions.push(request.question); if (testCase.clarification && !clarificationGiven) { - if ( - !isClarificationQuestion(testCase.id, request.question) - ) { + if (!clarificationQuestion(testCase.id, request.question)) { result.routeViolations.push( "confirmation-or-unrelated-question-before-clarification", ); + executionStopped = true; throw new Error( "Clarification is required before effect confirmation.", ); @@ -648,72 +777,130 @@ async function trial(candidate, directory, testCase, workspace, evidence) { wasFreeform: true, }; } - const pending = result.toolResults.findLast( - (tool) => - tool.result?.structuredContent?.status === - "requires_interaction", - )?.result.structuredContent; - const action = pending?.prompt?.action; + const { action, pending, handlerAnswer } = fileConsentContext( + result.tools, + consentResults, + readBackendEvents(env.TYPEAGENT_GHCP_EVAL_TRACE), + request, + ); if ( !executionStopped && confirmationCount < 8 && - pending?.prompt?.type === "confirmation" && - (fixtureConfirmationAllowed( - testCase.id, + (pending?.prompt?.type === "confirmation" || + handlerAnswer) && + caseConsentAllowed( action, + testCase, + stores[0], workspace, - evidence?.issueTitle, - ) || - (action?.schemaName === "powershell.powershell-files" && - action.actionName === "readFile" && - typeof action.parameters?.path === "string" && - isGhcpEvalArtifact( - action.parameters.path, - env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, - ))) + clarificationGiven, + evidence, + env, + ) ) { - const yes = request.choices?.find((choice) => - /^(yes|approve|confirm|proceed|allow)\b/i.test(choice), - ); + const yes = handlerAnswer + ? "Run" + : request.choices?.find((choice) => + /^(yes|approve|confirm|proceed|allow)\b/i.test( + choice, + ), + ); confirmationCount++; - approvedInteractions.add(pending.interactionId); + if (pending) + approvedInteractions.set(pending.interactionId, { + operationId: pending.operationId, + scopeId: pending.scopeId, + response: handlerAnswer ?? { + type: "confirmation", + approved: true, + }, + }); return { answer: yes ?? "Yes", wasFreeform: yes === undefined, }; } + executionStopped = true; throw new Error( "No authorized scripted answer for this interaction.", ); }, hooks: { + onPreMcpToolCall: (input) => { + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_CORRELATION, + JSON.stringify( + callCorrelation(result, input, result.tools.length), + ), + ); + }, onPreToolUse: (input) => { + input = { + ...input, + toolName: boundary.canonical(input.toolName), + }; + if ( + listCategory && + testCase.clarification && + !clarificationGiven && + !listsUnchanged(stores[0]) + ) + result.noPrematureListMutation = false; if ( testCase.clarification && !clarificationGiven && !fileStateMatches(snapshotFiles(workspace), fileFixture) ) result.noPrematureFileMutation = false; + const continuation = + input.toolName.includes("continueAction"); const unauthorizedContinuation = - input.toolName.includes("continueAction") && - input.toolArgs?.response?.approved === true && - !approvedInteractions.has( - input.toolArgs?.interactionId, + continuation && + input.toolArgs?.response?.approved !== false && + !consumeFixtureContinuation( + approvedInteractions, + input.toolArgs, + executionStopped, ); + if (!/ask_user|continueAction/.test(input.toolName)) + approvedInteractions.clear(); + const wrongRoute = !boundary.check(input.toolName); + const pendingReason = interactionGate.reason( + input.toolName, + input.toolArgs, + ); const forbidden = + wrongRoute || + pendingReason || + (input.toolName !== "ask_user" && + boundary.hasActiveDomainCall()) || unauthorizedContinuation || - (executionStopped && - !/ask_user|cancelAction/.test(input.toolName)) || - (candidate.id === 4 && - ((!preparation && - input.toolName.includes("searchActions")) || - (preparation && - input.toolName.includes("executeAction")))); + executionRouteViolation( + input.toolName, + candidate, + preparation, + executionStopped, + ); + boundary.recordDecision( + input.toolName, + input.toolArgs, + !forbidden, + ); if (forbidden) { + executionStopped = true; result.routeViolations.push(input.toolName); + result.boundaryDenials ??= []; + result.boundaryDenials.push({ + tool: input.toolName, + reason: wrongRoute + ? "outside_candidate_allowlist" + : (pendingReason ?? + "execution_or_consent_guard"), + }); return { permissionDecision: "deny", permissionDecisionReason: + pendingReason ?? "Evaluation route/interaction policy denied this call; do not replay it.", }; } @@ -721,6 +908,16 @@ async function trial(candidate, directory, testCase, workspace, evidence) { }, }, }); + try { + result.toolBoundary = { + admitted: await boundary.initialize(session), + verifiedBeforePrompt: true, + auditedCalls: 0, + }; + } catch (error) { + result.harnessError = `Tool boundary preflight failed: ${String(error)}`; + throw error; + } session.on("assistant.usage", (event) => result.usage.push({ model: event.data.model, @@ -729,15 +926,23 @@ async function trial(candidate, directory, testCase, workspace, evidence) { preparation, }), ); - session.on("tool.execution_start", (event) => - result.tools.push({ - toolCallId: event.data.toolCallId, - name: event.data.toolName, - arguments: event.data.arguments, - preparation, - startMs: performance.now() - started, - }), - ); + session.on("tool.execution_start", (event) => { + try { + result.tools.push({ + toolCallId: event.data.toolCallId, + name: event.data.toolName, + arguments: event.data.arguments, + preparation, + startMs: performance.now() - started, + backendEventOffset: readBackendEvents( + env.TYPEAGENT_GHCP_EVAL_TRACE, + ).length, + }); + } catch (error) { + result.harnessError = `Backend trace failed: ${String(error)}`; + executionStopped = true; + } + }); session.on("tool.execution_complete", (event) => { try { const artifact = registerGhcpEvalArtifact( @@ -758,29 +963,56 @@ async function trial(candidate, directory, testCase, workspace, evidence) { (tool) => tool.toolCallId === event.data.toolCallId, ); if (tool) tool.endMs = performance.now() - started; - if ( - terminalExecutionFailure( - tool?.name ?? "", - event.data.result, - event.data.success, + try { + if (!tool) + throw new Error("Tool completion has no start record"); + const admitted = boundary.audit(tool.name, tool.arguments); + result.toolBoundary.auditedCalls++; + if (!admitted && event.data.success === true) + throw new Error( + "Denied outer tool reported success; enforcement unavailable", + ); + interactionGate.observe(event.data.result); + if ( + terminalExecutionFailure( + tool?.name ?? "", + event.data.result, + event.data.success, + event.data.error, + tool + ? readBackendEvents( + env.TYPEAGENT_GHCP_EVAL_TRACE, + ).slice(tool.backendEventOffset) + : [], + ) + ) + executionStopped = true; + else if ( + terminalExecutionFailure( + tool?.name ?? "", + event.data.result, + event.data.success, + ) ) - ) + result.recoverableToolFailures.push(event.data.toolCallId); + } catch (error) { + result.harnessError = `Backend trace failed: ${String(error)}`; executionStopped = true; - result.toolResults.push({ - toolCallId: event.data.toolCallId, - success: event.data.success, - result: - testCase.id === "S5" || testCase.id === "M2" - ? "[network evidence withheld]" - : event.data.result, - }); + } + const views = toolEvidenceViews(event.data, Boolean(network)); + result.toolResults.push(views.persisted); + consentResults.push(views.consent); + if (executionStopped) { + approvedInteractions.clear(); + interactionGate.clear(); + } }); result.status = "running"; if (preparation) { const preparationStart = performance.now(); await session.sendAndWait( { - prompt: "Discover available contracts for file inventory, reading, writing/appending and copying files, GitHub pull-request files/checks and issue details, and read-only IP configuration. Do not execute actions, inspect contents, establish preferred targets, or guess future requests.", + prompt: category.preparation, }, 90_000, ); @@ -799,10 +1031,17 @@ async function trial(candidate, directory, testCase, workspace, evidence) { testCase, canClarify: () => !clarificationGiven && !executionStopped, clarify, + isClarification: clarificationQuestion, }); result.e2eMs = performance.now() - measuredStart; result.finalResponseAt = new Date().toISOString(); result.answer = answer?.data.content ?? ""; + interactionGate.assertSettled(); + if (result.toolBoundary.auditedCalls !== result.tools.length) { + result.harnessError = + "Incomplete tool boundary audit at final response"; + throw new Error(result.harnessError); + } gradeCompletedTrial({ result, testCase, @@ -853,18 +1092,19 @@ async function trial(candidate, directory, testCase, workspace, evidence) { const repetitions = phase === "pilot" ? 1 : Number(repetitionsText); const batchStart = Number(batchStartText); -const batchSize = phase === "pilot" ? 7 : Number(batchSizeText); +const width = category.candidates.length; +const batchSize = phase === "pilot" ? width : Number(batchSizeText ?? width); if ( !Number.isInteger(batchStart) || batchStart < 0 || !Number.isInteger(batchSize) || batchSize < 1 || - batchSize > 7 + batchSize > width ) - throw new Error("Each batch must contain between one and seven trials"); -if (phase === "measured" && (batchStart % 7 !== 0 || batchSize !== 7)) + throw new Error("Batch size exceeds the active category candidate count"); +if (phase === "measured" && (batchStart % width !== 0 || batchSize !== width)) throw new Error( - "Measured batches must preserve all seven candidates for one paired case", + "Measured batches must preserve all category candidates for one paired case", ); fs.mkdirSync(outputDirectory, { recursive: true }); const resultsPath = path.join(outputDirectory, "results.json"); @@ -877,7 +1117,8 @@ if (results.length !== batchStart) ); const selected = selection.split(",").map(Number); if ( - selected.some((id) => !candidates.some((candidate) => candidate.id === id)) + selected.some((id) => !category.candidates.includes(id)) || + new Set(selected).size !== selected.length ) { throw new Error("Unknown pilot candidate"); } @@ -885,12 +1126,19 @@ if (!["pilot", "measured"].includes(phase)) throw new Error("Unknown run phase"); const workspace = path.resolve(outputDirectory, "workspace"); fs.mkdirSync(workspace, { recursive: true }); -const corpus = buildCorpus(workspace, "microsoft/TypeAgent", 3058, 3067, 2617); +const corpus = categoryCorpus( + category, + workspace, + "microsoft/TypeAgent", + 3058, + 3067, + 2617, +); const evidence = evidencePath ? JSON.parse(fs.readFileSync(evidencePath, "utf8")) : undefined; if (evidence?.readinessFile) - assertCorpusReadiness( + assertReadiness( JSON.parse( fs.readFileSync( path.resolve( @@ -904,15 +1152,15 @@ if (evidence?.readinessFile) if ( phase === "measured" && (!evidence?.issueTitle || - selected.length !== 7 || - new Set(selected).size !== 7 || + selected.length !== width || + new Set(selected).size !== width || !evidence.readinessFile) ) { throw new Error( - "Measured runs require independent issue evidence and all seven candidates", + "Measured runs require independent issue evidence and all category candidates", ); } -if (evidence?.readinessFile) +if (evidence?.readinessFile && !listCategory) evidence.prOracles = externalOracle( JSON.parse( fs.readFileSync( @@ -930,19 +1178,24 @@ const cases = ? corpus.filter(({ id }) => pilotCases.split(",").includes(id)) : shuffled(corpus, seed); if (cases.length === 0) throw new Error("No cases selected"); -const order = balancedOrder( +const { order, ...applicability } = categorySchedule( + category, cases, phase === "pilot" ? selected : shuffled(selected, seed), repetitions, ); +assertCategoryResults(results, order, category); const specification = JSON.stringify( { - protocolVersion: 5, + protocolVersion, + category: category.name, corpusVersion, - applicability: - "All twenty common-file cases apply to all seven candidates; no list tasks or native N/A slots.", fixtures: fileFixture, + listFixtures: listFixture, + preparationPrompt: category.preparation, + outerToolBoundary: + "Exact source-qualified allowlist, initialized runtime metadata before prompt, deny hook, per-call enforcement audit. Missing evidence fails closed.", runnerSha256: createHash("sha256") .update(fs.readFileSync(fileURLToPath(import.meta.url))) .digest("hex"), @@ -956,7 +1209,13 @@ const specification = seed, order, cases, - candidates, + candidates: candidates.filter(({ id }) => + category.candidates.includes(id), + ), + applicability: { + ...applicability, + policy: `Only ${category.candidates.join(",")} are eligible in ${category.name}; derive separate denominators from this category schedule.`, + }, model: evalModel, reasoningEffort: "high", concurrency: 1, @@ -973,12 +1232,12 @@ const specification = overflowPolicy: "Trial-private temp artifacts registered from SDK completion notices; canonical direct child, regular unlinked file, SHA256 rechecked before registered reads.", clarificationPolicy: - "One scripted corpus answer through callback or final text, same 90-second end-to-end deadline; no continuation after execution failure.", + "One scripted corpus answer through callback or final text, same 90-second end-to-end deadline; no continuation after terminal failure.", sessionCreditSoftLimit: 60, ledgerPath, templateDirectory: path.resolve(template), fixtureReset: - "Copy catalog-only state, exclude stale locks; keep inactive lists unchanged, restore seven ordinary text files and remove the previous trial backup.", + "Copy category preflight state, exclude stale locks; restore exact seven-list seed and seven ordinary files, remove previous backup; validate both domains independently.", gradingStatus: "independent fixture oracles; explicit final-answer review required", nativeTools, @@ -993,27 +1252,29 @@ const specification = "typeagent-macros", "typeagent-skills", ], - safety: "Normal confirmation retained; no replay after failed/denied/cancelled/uncertain execution, including internal error-triggered retries. Translation fallback toolset retained.", + safety: "Ordinary read I/O errors require affirmative safe evidence for recovery within existing routes, permissions and budgets. Denied/cancelled/uncertain execution and missing error details remain terminal. Translation fallback toolset retained.", }, null, 2, ) + "\n"; const specificationPath = path.join(outputDirectory, "specification.json"); -if ( - fs.existsSync(specificationPath) && - fs.readFileSync(specificationPath, "utf8") !== specification -) - throw new Error("Frozen run specification changed; start a distinct run"); +assertFrozenSpecification( + fs.existsSync(specificationPath) + ? fs.readFileSync(specificationPath, "utf8") + : undefined, + specification, +); fs.writeFileSync(specificationPath, specification); for (const entry of order.slice(batchStart, batchStart + batchSize)) { const candidate = candidates.find(({ id }) => id === entry.candidate); const testCase = cases.find(({ id }) => id === entry.caseId); + const directory = path.join( + outputDirectory, + `${entry.repetition}-${entry.caseId}-candidate-${candidate.id}`, + ); const result = await trial( candidate, - path.join( - outputDirectory, - `${entry.repetition}-${entry.caseId}-candidate-${candidate.id}`, - ), + directory, testCase, workspace, evidence, diff --git a/ts/packages/copilot-plugin-eval/scripts/test/boundary.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/boundary.spec.mjs new file mode 100644 index 0000000000..15c6aa093f --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/test/boundary.spec.mjs @@ -0,0 +1,379 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { CopilotSession } from "@github/copilot-sdk"; +import { + candidateToolBoundary, + pendingInteractionGate, + callCorrelation, + toolEvidenceViews, + executionRouteViolation, +} from "../ghcp-eval-boundary.mjs"; +import { + isClarificationQuestion, + fileConsentContext, +} from "../ghcp-eval-corpus.mjs"; + +const metadata = (tools) => ({ + rpc: { + tools: { + initializeAndValidate: async () => {}, + getCurrentMetadata: async () => ({ tools }), + }, + }, +}); +const nl = { id: 1, tools: ["typeagent-processCommand"] }; +const inventory = [ + { name: "ask_user" }, + { + name: "typeagent-e2e-typeagent-processCommand", + namespacedName: "functions.typeagent-e2e-typeagent-processCommand", + mcpServerName: "typeagent-e2e", + mcpToolName: "typeagent-processCommand", + }, +]; +const structured = { + id: 3, + tools: [ + "typeagent-searchActions", + "typeagent-executeAction", + "typeagent-continueAction", + "typeagent-cancelAction", + ], +}; + +test("terminal stops and C4 phase routing never imply replay or renewed discovery", () => { + const c4 = { ...structured, id: 4 }; + assert.equal( + executionRouteViolation( + "typeagent-e2e-typeagent-executeAction", + c4, + true, + false, + ), + "reuse_preparation_contract", + ); + assert.equal( + executionRouteViolation( + "typeagent-e2e-typeagent-searchActions", + c4, + false, + false, + ), + "reuse_preparation_contract", + ); + assert.equal( + executionRouteViolation( + "typeagent-e2e-typeagent-searchActions", + structured, + false, + false, + ), + undefined, + ); + assert.equal( + executionRouteViolation( + "typeagent-e2e-typeagent-executeAction", + c4, + false, + true, + ), + "terminal_execution_stop", + ); + assert.equal( + executionRouteViolation( + "typeagent-e2e-typeagent-cancelAction", + c4, + false, + true, + ), + undefined, + ); +}); + +test("A5 accepts measured filename questions but never consents or unrelated questions", () => { + for (const question of [ + "Which filename inside `workspace` should I read?", + "What is the exact filename in that `workspace` directory that you want me to read?", + "Which file name do you mean?", + "Which file should I read?", + ]) + assert.equal(isClarificationQuestion("A5", question), true, question); + for (const question of [ + "Confirm which filename to read?", + "Approve reading that file?", + "Run or Cancel?", + "What is your name?", + "Which profile should I load?", + "Read trip.txt?", + "Allow access to that filename?", + ]) + assert.equal(isClarificationQuestion("A5", question), false, question); +}); + +test("every candidate has a closed source-qualified allowlist, never wildcard MCP", async () => { + for (const candidate of [ + nl, + { ...nl, id: 2 }, + structured, + { ...structured, id: 4 }, + { id: 5, tools: [...nl.tools, ...structured.tools] }, + { id: 6, tools: [...nl.tools, ...structured.tools] }, + { id: 7 }, + ]) { + const boundary = candidateToolBoundary(candidate, [ + "builtin:ask_user", + "builtin:view", + ]); + assert.equal( + boundary.availableTools.some((name) => name.includes("*")), + false, + ); + const tools = [ + { name: "ask_user" }, + ...(candidate.id >= 5 ? [{ name: "view" }] : []), + ...(candidate.tools ?? []).map((tool) => ({ + name: `typeagent-e2e-${tool}`, + mcpServerName: "typeagent-e2e", + mcpToolName: tool, + })), + ]; + assert.equal(boundary.check("ask_user"), false); + await boundary.initialize(metadata(tools)); + assert.equal(boundary.check("ask_user"), true); + for (const offRoute of [ + "web_search", + "github-mcp-server-get_file_contents", + "unknown-typeagent-processCommand", + "task", + "evil_ask_user", + "typeagent-e2e-typeagent-deleteEverything", + ]) + assert.equal(boundary.check(offRoute), false); + assert.equal(boundary.check("view"), candidate.id >= 5); + } + for (const tools of [ + null, + [], + [...inventory, { name: "web_search" }], + [{ name: "ask_user" }], + [ + ...inventory, + { + name: "view", + mcpServerName: "github-mcp-server", + mcpToolName: "get_file_contents", + }, + ], + ]) + await assert.rejects(() => + candidateToolBoundary(nl, []).initialize(metadata(tools)), + ); + await assert.rejects(() => + candidateToolBoundary(nl, []).initialize({ rpc: {} }), + ); +}); + +test("installed SDK pre-tool hook returns denial before mocked effects and audit requires matching evidence", async () => { + const boundary = candidateToolBoundary(nl, []); + await boundary.initialize(metadata(inventory)); + const session = new CopilotSession("offline", { + sendRequest: () => { + throw new Error("No runtime calls permitted"); + }, + }); + session.registerHooks({ + onPreToolUse: (input) => { + const allowed = boundary.check(input.toolName); + boundary.recordDecision(input.toolName, input.toolArgs, allowed); + return { permissionDecision: allowed ? "allow" : "deny" }; + }, + }); + let effects = 0; + for (const toolName of [ + "web_search", + "github-mcp-server-get_file_contents", + inventory[1].name, + ]) { + const answer = await session._handleHooksInvoke("preToolUse", { + sessionId: "offline", + timestamp: new Date().toISOString(), + workingDirectory: "fixture", + toolName, + toolArgs: { query: "synthetic" }, + }); + if (answer.permissionDecision !== "deny") effects++; + assert.equal( + boundary.audit(toolName, { query: "synthetic" }), + toolName === inventory[1].name, + ); + } + assert.equal(effects, 1); + assert.throws(() => boundary.audit("web_search", {}), /Missing pre-tool/); + boundary.recordDecision(inventory[1].name, { a: 1, b: 2 }, true); + assert.equal(boundary.hasActiveDomainCall(), true); + assert.throws( + () => boundary.audit(inventory[1].name, { a: 2 }), + /Missing pre-tool/, + ); + assert.equal(boundary.audit(inventory[1].name, { b: 2, a: 1 }), true); + assert.equal(boundary.hasActiveDomainCall(), false); + boundary.recordDecision(inventory[1].namespacedName, {}, true); + assert.equal(boundary.audit(inventory[1].name, {}), true); + session._markDisconnected(); +}); + +test("pending contract prohibits a second execute, stale/wrong handles and unfinished final answer", () => { + const gate = pendingInteractionGate(); + const handles = { + scopeId: "scope", + operationId: "operation", + interactionId: "interaction", + }; + gate.observe({ + structuredContent: { status: "requires_interaction", ...handles }, + }); + assert.match( + gate.reason("typeagent-e2e-typeagent-executeAction", {}), + /pending interaction/, + ); + assert.equal(gate.reason("ask_user", {}), undefined); + for (const key of Object.keys(handles)) + assert.match( + gate.reason("typeagent-e2e-typeagent-continueAction", { + ...handles, + [key]: "wrong", + }), + /handles/, + ); + assert.equal( + gate.reason("typeagent-e2e-typeagent-continueAction", handles), + undefined, + ); + assert.equal( + gate.reason("typeagent-e2e-typeagent-cancelAction", handles), + undefined, + ); + assert.throws(() => gate.assertSettled(), /unresolved/); + gate.observe({ structuredContent: { status: "completed" } }); + assert.doesNotThrow(() => gate.assertSettled()); + assert.throws( + () => + gate.observe({ + structuredContent: { status: "requires_interaction" }, + }), + /missing/, + ); + gate.clear(); +}); + +test("private evidence redaction cannot erase active confirmation context", () => { + const action = { + schemaName: "powershell.powershell-files", + actionName: "readFile", + parameters: { path: "private-artifact" }, + }; + const event = { + toolCallId: "call", + success: true, + result: { + structuredContent: { + status: "requires_interaction", + scopeId: "s", + operationId: "o", + interactionId: "i", + prompt: { type: "confirmation", action }, + output: ["private-network-value"], + }, + }, + }; + const views = toolEvidenceViews(event, true); + assert.doesNotMatch( + JSON.stringify(views.persisted), + /private-network-value|private-artifact/, + ); + const tools = [ + { + toolCallId: "call", + name: "typeagent-e2e-typeagent-executeAction", + endMs: 1, + }, + ]; + assert.equal( + fileConsentContext(tools, [views.persisted], [], {}).action, + undefined, + ); + assert.deepEqual( + fileConsentContext(tools, [views.consent], [], {}).action, + action, + ); + const correlation = callCorrelation( + { caseId: "S5", candidate: 3, sessionId: "private-session" }, + { + toolCallId: "private-call", + arguments: { + scopeId: "private-scope", + operationId: "private-operation", + interactionId: "private-interaction", + content: "private-text", + }, + }, + 4, + ); + assert.doesNotMatch(JSON.stringify(correlation), /private-/); + assert.equal(correlation.scopeSha256.length, 64); + assert.equal(correlation.callSequence, 4); +}); + +test("SDK backend-only output or empty final presentation never becomes a successful answer", async () => { + for (const emptyMessage of [false, true]) { + let session; + session = new CopilotSession("offline", { + sendRequest: async () => { + session._dispatchEvent({ + type: "tool.execution_complete", + data: { result: { content: "complete backend result" } }, + }); + if (emptyMessage) + session._dispatchEvent({ + type: "assistant.message", + data: { content: "" }, + }); + session._dispatchEvent({ type: "session.idle", data: {} }); + return {}; + }, + }); + const answer = await session.sendAndWait("synthetic", 1000); + assert.equal(answer?.data.content ?? "", ""); + session._markDisconnected(); + } +}); + +test("SDK final-message delivery preserves content without promoting backend output to a final answer", async () => { + let session; + const connection = { + sendRequest: async (method) => { + assert.equal(method, "session.send"); + session._dispatchEvent({ + type: "assistant.message", + data: { content: "Intermediate" }, + }); + session._dispatchEvent({ + type: "tool.execution_complete", + data: { result: { content: "backend output" } }, + }); + session._dispatchEvent({ + type: "assistant.message", + data: { content: "Faithful final presentation" }, + }); + session._dispatchEvent({ type: "session.idle", data: {} }); + return {}; + }, + }; + session = new CopilotSession("offline", connection); + const result = await session.sendAndWait({ prompt: "synthetic" }, 1000); + assert.equal(result.data.content, "Faithful final presentation"); + session._markDisconnected(); +}); diff --git a/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs index 96c64f4fbf..1028b05878 100644 --- a/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs @@ -15,6 +15,11 @@ import { fixtureConfirmationAllowed, buildCorpus, balancedOrder, + fileHandlerConfirmation, + fileConsentContext, + pendingFileAction, + consumeFixtureContinuation, + isClarificationQuestion, } from "../ghcp-eval-corpus.mjs"; import { fileStateMatches, @@ -43,6 +48,196 @@ test("all twenty file cases are applicable to all seven candidates", () => { assert.equal(balancedOrder(cases.slice(0, 2), [1, 7], 1).length, 4); }); +test("handler consent needs an exact question and unambiguous admitted file action", () => { + const action = { + schemaName: "powershell.powershell-files", + actionName: "copyFile", + parameters: { + source: "grocery.txt", + destination: "grocery-backup.txt", + }, + }; + const admitted = { event: "action.admitted", detail: action }; + const prompt = { + type: "question", + message: "Copy the requested file or directory?", + choices: ["Run", "Cancel"], + defaultId: 1, + }; + assert.equal(pendingFileAction([admitted]), action); + assert.deepEqual(fileHandlerConfirmation(prompt, action), { + type: "question", + selected: 0, + }); + for (const events of [ + [], + [admitted, admitted], + [admitted, { event: "action.completed", detail: action }], + [admitted, { event: "action.denied", detail: action }], + [admitted, { event: "action.failed", detail: action }], + [ + admitted, + { event: "action.completed", detail: { actionName: "writeFile" } }, + ], + ]) + assert.equal(pendingFileAction(events), undefined); + for (const bad of [ + { ...prompt, choices: ["Cancel", "Run"] }, + { ...prompt, message: "Delete the requested file or directory?" }, + { ...prompt, defaultId: 0 }, + { ...prompt, type: "confirmation" }, + ]) + assert.equal(fileHandlerConfirmation(bad, action), undefined); + assert.equal(fileHandlerConfirmation(prompt, undefined), undefined); + assert.equal( + fileHandlerConfirmation(prompt, { ...action, actionName: "writeFile" }), + undefined, + ); + for (const id of ["A1", "A4"]) { + assert.equal(isClarificationQuestion(id, prompt.message), false); + assert.equal(filePolicy(id).writesEnabled, false); + } +}); + +test("handler continuations are single-use and bound to operation, scope and response", () => { + const args = { + interactionId: "i", + operationId: "o", + scopeId: "s", + response: { type: "question", selected: 0 }, + }; + const approval = () => + new Map([ + ["i", { operationId: "o", scopeId: "s", response: args.response }], + ]); + const approvals = approval(); + assert.equal(consumeFixtureContinuation(approvals, args, false), true); + assert.equal(consumeFixtureContinuation(approvals, args, false), false); + for (const other of [ + { ...args, interactionId: "old" }, + { ...args, operationId: "other" }, + { ...args, scopeId: "other" }, + { ...args, response: { type: "question", selected: 1 } }, + { ...args, response: { type: "confirmation", approved: true } }, + ]) + assert.equal( + consumeFixtureContinuation(approval(), other, false), + false, + ); + assert.equal(consumeFixtureContinuation(approval(), args, true), false); + assert.equal(consumeFixtureContinuation(new Map(), args, false), false); + assert.equal( + consumeFixtureContinuation( + approval(), + { + ...args, + response: { selected: 0, type: "question" }, + }, + false, + ), + true, + ); + const confirmation = new Map([ + [ + "i", + { + operationId: "o", + scopeId: "s", + response: { type: "confirmation", approved: true }, + }, + ], + ]); + assert.equal( + consumeFixtureContinuation( + confirmation, + { + ...args, + response: { approved: true, type: "confirmation" }, + }, + false, + ), + true, + ); + assert.equal( + consumeFixtureContinuation( + approval(), + { + ...args, + response: { ...args.response, extra: true }, + }, + false, + ), + false, + ); +}); + +test("file consent cannot reuse stale action traces or overlapping tool contexts", () => { + const tool = { + name: "typeagent-processCommand", + toolCallId: "t", + backendEventOffset: 0, + }; + const action = { + schemaName: "powershell.powershell-files", + actionName: "copyFile", + parameters: {}, + }; + const events = [{ event: "action.admitted", detail: action }]; + const request = { + question: "Copy the requested file or directory?", + choices: ["Run", "Cancel"], + }; + assert.equal( + fileConsentContext([tool], [], events, request).action, + action, + ); + for (const tools of [ + [], + [{ ...tool, endMs: 1 }], + [{ ...tool, backendEventOffset: 1 }], + [tool, { ...tool, toolCallId: "other" }], + [ + { ...tool, endMs: 1 }, + { name: "typeagent-searchActions", toolCallId: "search", endMs: 2 }, + ], + ]) + assert.equal( + fileConsentContext(tools, [], events, request).handlerAnswer, + undefined, + ); + const pending = { + status: "requires_interaction", + prompt: { + type: "question", + message: request.question, + choices: request.choices, + }, + operationId: "o", + scopeId: "s", + interactionId: "i", + }; + const results = [ + { toolCallId: "t", result: { structuredContent: pending } }, + ]; + const context = fileConsentContext( + [{ ...tool, endMs: 1 }], + results, + events, + request, + ); + assert.equal(context.pending, pending); + assert.deepEqual(context.handlerAnswer, { type: "question", selected: 0 }); + assert.equal( + fileConsentContext( + [{ ...tool, endMs: 1 }], + [{ ...results[0], toolCallId: "old" }], + events, + request, + ).handlerAnswer, + undefined, + ); +}); + test("old or incomplete preflights cannot run the new corpus", () => { const readiness = { corpusVersion, diff --git a/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs index a78a1285d0..63e941fb29 100644 --- a/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs @@ -9,17 +9,22 @@ import { externalOracle, intervalUnionMs, percentile, + preliminaryGrade, + recoverableBackendReadFailure, terminalExecutionFailure, } from "../ghcp-eval-grade.mjs"; import { balancedOrder, + assertFrozenSpecification, buildCorpus, + buildTrialSchedule, expectedFiles, fileFixture, corpusVersion, fixtureConfirmationAllowed, isClarificationQuestion, normalizeLists, + protocolVersion, shuffled, sendWithClarification, } from "../ghcp-eval-corpus.mjs"; @@ -36,6 +41,441 @@ test("evaluation pins Luna 5.6 and rejects another model before paid work", () = } }); const corpus = buildCorpus(fixtures, "owner/repo", 10, 20, 30); +const readFailureEvents = [ + { + event: "action.admitted", + detail: { schemaName: "github-cli", actionName: "prFiles" }, + }, + { + event: "action.completed", + detail: { + schemaName: "github-cli", + actionName: "prFiles", + success: false, + recoverable: true, + }, + }, +]; + +test("protocol seven preserves all twenty common-file cases for all seven candidates", () => { + assert.equal(protocolVersion, 7); + const schedule = buildTrialSchedule(corpus, [1, 2, 3, 4, 5, 6, 7], 1); + const { order } = schedule; + assert.equal(schedule.totalSlots, 140); + assert.equal(schedule.scheduledTrials, 140); + assert.deepEqual(schedule.applicableCounts, { + 1: 20, + 2: 20, + 3: 20, + 4: 20, + 5: 20, + 6: 20, + 7: 20, + }); + assert.ok(order.every(({ applicable }) => applicable)); + for (let candidate = 1; candidate <= 7; candidate++) + assert.deepEqual( + order + .filter((entry) => entry.candidate === candidate) + .map(({ caseId }) => caseId) + .sort(), + corpus.map(({ id }) => id).sort(), + ); +}); + +test("frozen schedules derive pilot/repetition counts and cannot resume older protocols or changed order", () => { + const schedule = buildTrialSchedule(corpus, [1, 2, 3, 4, 5, 6, 7], 2); + assert.equal(schedule.scheduledTrials, 280); + assert.equal(schedule.applicableCounts[7], 40); + const pilot = buildTrialSchedule(corpus.slice(0, 2), [1, 7], 1); + assert.equal(pilot.totalSlots, 4); + assert.equal(pilot.scheduledTrials, 4); + const frozen = JSON.stringify({ protocolVersion, corpusVersion, ...pilot }); + assert.doesNotThrow(() => assertFrozenSpecification(undefined, frozen)); + assert.doesNotThrow(() => assertFrozenSpecification(frozen, frozen)); + assert.throws( + () => + assertFrozenSpecification( + JSON.stringify({ protocolVersion: 4, ...pilot }), + frozen, + ), + /Frozen run specification changed/, + ); + assert.throws( + () => + assertFrozenSpecification( + JSON.stringify({ + protocolVersion, + corpusVersion, + ...pilot, + order: [...pilot.order].reverse(), + }), + frozen, + ), + /Frozen run specification changed/, + ); + assert.throws( + () => + assertFrozenSpecification( + JSON.stringify({ + protocolVersion, + corpusVersion: "old-files", + ...pilot, + }), + frozen, + ), + /Frozen run specification changed/, + ); +}); + +test("known native read I/O failures allow recovery, never opaque shell errors or denied reads", () => { + for (const tool of ["view", "glob", "rg", "web_fetch", "functions.view"]) { + assert.equal( + terminalExecutionFailure(tool, undefined, false, { + message: "ENOENT: missing file", + }), + false, + ); + for (const message of [ + "", + "unknown error", + "permission denied: ENOENT", + "ENOENT after cancellation: cancelled", + "uncertain delivery: ECONNRESET", + ]) + assert.equal( + terminalExecutionFailure(tool, undefined, false, { message }), + true, + ); + } + for (const tool of ["powershell", "edit", "create", "functions.edit"]) + assert.equal( + terminalExecutionFailure(tool, undefined, false, { + message: "ENOENT", + }), + true, + ); + for (const code of [ + "ERR_ACCESS_DENIED", + "permissionDenied", + "cancelled", + "execution_uncertain", + ]) { + assert.equal( + terminalExecutionFailure("view", undefined, false, { + code, + message: "ENOENT", + }), + true, + ); + } +}); + +test("TypeAgent recovery requires a complete read-only trace and affirmative error details", () => { + assert.equal(recoverableBackendReadFailure(readFailureEvents), true); + for (const actionName of [ + "readFile", + "listFiles", + "writeFile", + "copyFile", + ]) { + const events = readFailureEvents.map((entry) => ({ + ...entry, + detail: { + ...entry.detail, + schemaName: "powershell.powershell-files", + actionName, + }, + })); + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "Error: ENOENT" }, + true, + undefined, + events, + ), + actionName === "writeFile" || actionName === "copyFile", + ); + } + for (const events of [ + [], + readFailureEvents.slice(0, 1), + [...readFailureEvents, { event: "action.denied", detail: {} }], + [...readFailureEvents, { event: "action.failed", detail: {} }], + readFailureEvents.map((event) => ({ + ...event, + detail: { + ...event.detail, + schemaName: "list", + actionName: "addItems", + }, + })), + readFailureEvents.map((event) => ({ + ...event, + detail: { ...event.detail, recoverable: undefined }, + })), + ]) { + assert.equal(recoverableBackendReadFailure(events), false); + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "Error: ECONNRESET" }, + true, + undefined, + events, + ), + true, + ); + } + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "Error: ECONNRESET" }, + true, + undefined, + readFailureEvents, + ), + false, + ); + for (const status of [ + "failed", + "cancelled", + "unavailable", + "execution_uncertain", + ]) { + const result = { + structuredContent: { + status, + error: { code: "execution_failed", message: "ECONNRESET" }, + }, + }; + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + result, + true, + undefined, + readFailureEvents, + ), + status !== "failed", + ); + } + for (const content of [ + "Error: permission denied: ECONNRESET", + "Error: uncertain delivery", + "Error:", + ]) { + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content }, + true, + undefined, + readFailureEvents, + ), + true, + ); + } + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + undefined, + false, + undefined, + readFailureEvents, + ), + true, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + { + structuredContent: { + status: "failed", + error: { code: "invalid_scope", message: "ECONNRESET" }, + }, + }, + true, + undefined, + readFailureEvents, + ), + true, + ); +}); + +test("safe recovery can reach clarification within the same deadline but is not itself success", async () => { + let calls = 0; + let stopped = false; + const answer = await sendWithClarification({ + session: { + sendAndWait: async () => { + calls++; + stopped ||= terminalExecutionFailure("view", undefined, false, { + message: "ENOENT", + }); + return { + data: { + content: calls === 1 ? "Which file?" : "trip contents", + }, + }; + }, + }, + prompt: corpus.find(({ id }) => id === "A5").prompt, + testCase: corpus.find(({ id }) => id === "A5"), + timeoutMs: 1000, + canClarify: () => !stopped, + clarify: () => "trip.txt.", + }); + assert.equal(calls, 2); + assert.equal(answer.data.content, "trip contents"); + assert.equal( + preliminaryGrade( + { + status: "completed_ungraded", + caseId: "S2", + routeViolations: [], + grade: { + fileStateMatchesOracle: true, + listStateUnchanged: true, + }, + answer: "not the file contents", + }, + {}, + ).outcome, + "pending_review", + ); +}); + +test("contradictory TypeAgent denial and cancellation evidence overrides recoverable content", () => { + for (const status of [ + "cancelled", + "execution_uncertain", + "unavailable", + "requires_interaction", + ]) { + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "Error: ENOENT", structuredContent: { status } }, + true, + undefined, + readFailureEvents, + ), + true, + ); + } + const failed = { + status: "failed", + error: { code: "execution_failed", message: "ENOENT" }, + }; + for (const content of [ + "Error: permission denied", + "Error: cancelled", + "Error: uncertain delivery", + ]) { + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + { content, structuredContent: failed }, + true, + undefined, + readFailureEvents, + ), + true, + ); + } + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "Error: ENOENT", structuredContent: failed }, + true, + { code: "permissionDenied" }, + readFailureEvents, + ), + true, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-cancelAction", + { structuredContent: failed }, + true, + undefined, + readFailureEvents, + ), + true, + ); +}); + +test("A4 requires unresolved-item clarification and preserves every other entry and file", () => { + const testCase = corpus.find(({ id }) => id === "A4"); + assert.equal( + testCase.prompt, + `Remove an item from ${path.join(fixtures, "grocery.txt")}, preserving the other entries.`, + ); + assert.equal(testCase.clarification, "Remove eggs."); + assert.equal( + isClarificationQuestion("A4", "Which item should I remove?"), + true, + ); + assert.equal( + isClarificationQuestion("A4", "Confirm removing milk?"), + false, + ); + assert.equal( + isClarificationQuestion("A4", "Which list should I clear?"), + false, + ); + const expected = expectedFiles("A4", 30); + assert.equal(expected["grocery.txt"], "milk\nrice\n"); + assert.deepEqual( + { ...expected, "grocery.txt": fileFixture["grocery.txt"] }, + fileFixture, + ); + const action = { + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { + path: path.join(fixtures, "grocery.txt"), + content: "milk\nrice\n", + }, + }; + assert.equal(fixtureConfirmationAllowed("A4", action, fixtures), true); + for (const other of [ + { ...action, parameters: { ...action.parameters, content: "" } }, + { + ...action, + parameters: { ...action.parameters, content: "eggs\nrice\n" }, + }, + { ...action, parameters: { ...action.parameters, content: "rice\n" } }, + { + ...action, + parameters: { + ...action.parameters, + path: path.join(fixtures, "pantry.txt"), + }, + }, + ]) + assert.equal(fixtureConfirmationAllowed("A4", other, fixtures), false); + const result = { + status: "completed_ungraded", + caseId: "A4", + routeViolations: [], + answer: "Removed eggs.", + grade: { + fileStateMatchesOracle: true, + listStateUnchanged: true, + clarificationRequested: true, + noPrematureFileMutation: false, + }, + }; + assert.equal( + preliminaryGrade(result, {}).reason, + "clarification_not_verified_before_effects", + ); + result.grade.noPrematureFileMutation = true; + assert.equal(preliminaryGrade(result, {}).outcome, "pending_review"); +}); test("legacy list confirmations are not approved in the file corpus", () => { const action = { schemaName: "list", @@ -189,6 +629,29 @@ test("failed native execution cannot trigger a scripted continuation", async () }); assert.equal(calls, 1); }); +test("explicit cancellation and uncertain shell follow-ups remain terminal", () => { + assert.equal( + terminalExecutionFailure( + "typeagent-cancelAction", + { + structuredContent: { status: "cancelled" }, + }, + true, + ), + true, + ); + assert.equal(terminalExecutionFailure("stop_powershell", {}, true), true); + assert.equal( + terminalExecutionFailure("functions.stop_powershell", undefined, false), + true, + ); + assert.equal( + terminalExecutionFailure("read_powershell", undefined, false, { + message: "ENOENT", + }), + true, + ); +}); test("independent PR file evidence must be complete", () => { const snapshot = { status: "passed", diff --git a/ts/packages/copilot-plugin-eval/scripts/test/list-category.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/list-category.spec.mjs new file mode 100644 index 0000000000..299d1ce1da --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/test/list-category.spec.mjs @@ -0,0 +1,681 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { test } from "node:test"; +import { + buildCorpus, + protocolVersion, + listFixture, + consumeFixtureContinuation, + fileConsentContext, + sendWithClarification, +} from "../ghcp-eval-corpus.mjs"; +import { + evaluationCategory, + categoryCorpus, + categorySchedule, + assertCategoryReadiness, + assertCategoryResults, +} from "../ghcp-eval-categories.mjs"; +import { + expectedLists, + readLists, + resetLists, + listsUnchanged, + listPolicy, + listFilePolicy, + listClarificationQuestion, + gradeListTrial, +} from "../ghcp-eval-lists.mjs"; +import { restoreFiles, snapshotFiles } from "../ghcp-eval-files.mjs"; +import { + preliminaryGrade, + terminalExecutionFailure, +} from "../ghcp-eval-grade.mjs"; +import { summarizeCategory } from "../ghcp-eval-report.mjs"; +import { + verifyListPreflight, + preflightListPolicy, +} from "../ghcp-eval-list-preflight.mjs"; +import { + ghcpEvalListActionAllowed, + ghcpEvalListExternalReadAllowed, +} from "../../../dispatcher/dispatcher/dist/execute/ghcpEvalLists.js"; + +const category = evaluationCategory("lists"); +const corpus = categoryCorpus( + category, + "fixtures", + "microsoft/TypeAgent", + 3058, + 3067, + 2617, +); +const order = categorySchedule(category, corpus, category.candidates, 1).order; +const metadata = { + protocolVersion, + category: category.name, + corpusVersion: category.corpusVersion, +}; +const writeLists = (store, lists) => + fs.writeFileSync( + store, + JSON.stringify( + Object.entries(lists).map(([name, items]) => ({ name, items })), + ), + ); + +function fixture(t) { + const directory = fs.mkdtempSync(path.join(os.tmpdir(), "eval-lists-")); + t.after(() => fs.rmSync(directory, { recursive: true, force: true })); + const store = path.join(directory, "lists.json"); + writeLists(store, listFixture); + const files = path.join(directory, "files"); + fs.mkdirSync(files); + restoreFiles(files); + return { directory, store, files }; +} + +test("common defaults and exact 20x7 / 20x4 categories retain separate IDs", () => { + const common = evaluationCategory(); + const commonCases = categoryCorpus( + common, + "fixtures", + "microsoft/TypeAgent", + 3058, + 3067, + 2617, + ); + assert.deepEqual( + commonCases, + buildCorpus("fixtures", "microsoft/TypeAgent", 3058, 3067, 2617), + ); + assert.equal( + categorySchedule(common, commonCases, common.candidates, 1).order + .length, + 140, + ); + assert.equal(order.length, 80); + assert.equal(new Set(corpus.map((row) => row.id)).size, 20); + for (const cohort of ["S", "M", "R", "A"]) + assert.equal(corpus.filter((row) => row.cohort === cohort).length, 5); + assert.deepEqual( + [...new Set(order.map((row) => row.candidate))].sort(), + [1, 2, 3, 4], + ); + for (const ids of [[1, 5], [6], [7], [1, 1]]) + assert.throws( + () => categorySchedule(category, corpus, ids, 1), + /another evaluation category/, + ); + assert.throws( + () => categorySchedule(category, commonCases, [1], 1), + /another evaluation category/, + ); + assert.throws( + () => categorySchedule(common, corpus, [1], 1), + /another evaluation category/, + ); + assert.doesNotMatch( + category.preparation, + /milk|eggs|apples|2617|grocery|pantry|adapter/, + ); +}); + +test("category/version readiness and resume reject frozen, mixed or reordered evidence", () => { + const readiness = { + ...metadata, + status: "passed", + listOperationsVerified: true, + }; + assert.doesNotThrow(() => assertCategoryReadiness(readiness, category)); + for (const override of [ + { protocolVersion: 5 }, + { category: "common-files" }, + { listOperationsVerified: false }, + { corpusVersion: "old" }, + ]) + assert.throws(() => + assertCategoryReadiness({ ...readiness, ...override }, category), + ); + const prefix = order.slice(0, 4).map((row) => ({ ...row, ...metadata })); + assert.doesNotThrow(() => assertCategoryResults(prefix, order, category)); + for (const override of [ + { protocolVersion: 5 }, + { category: "common-files" }, + { caseId: "S1" }, + { candidate: 7 }, + { repetition: 1 }, + ]) + assert.throws(() => + assertCategoryResults( + [{ ...prefix[0], ...override }], + order, + category, + ), + ); +}); + +test("all list oracles require exact lists and unchanged independent files", (t) => { + const { store, files } = fixture(t); + for (const row of corpus) { + resetLists(store); + const before = readLists(store); + writeLists(store, expectedLists(row.id, 2617, "Independent title")); + const result = { + caseId: row.id, + noPrematureListMutation: true, + listsAtClarification: before, + }; + const grade = gradeListTrial( + result, + store, + snapshotFiles(files), + "Independent title", + ); + assert.deepEqual( + grade, + { + listStateMatchesOracle: true, + fileStateMatchesOracle: true, + noPrematureListMutation: true, + }, + row.id, + ); + const corrupted = readLists(store); + corrupted.unrelated = ["unexpected"]; + writeLists(store, corrupted); + assert.equal( + gradeListTrial( + result, + store, + snapshotFiles(files), + "Independent title", + ).listStateMatchesOracle, + false, + ); + } + resetLists(store); + assert.equal(listsUnchanged(store), true); + fs.writeFileSync(path.join(files, "trip.txt"), "changed"); + assert.equal( + gradeListTrial( + { caseId: "list-S1", noPrematureListMutation: true }, + store, + snapshotFiles(files), + ).fileStateMatchesOracle, + false, + ); + fs.writeFileSync( + store, + '[{"name":"x","items":[]},{"name":"x","items":[]}]', + ); + assert.throws(() => readLists(store), /Invalid evaluation list snapshot/); +}); + +test("list policy enforces exact targets/items, ordering, category file scope and independent issue oracle", (t) => { + const { store } = fixture(t); + const allowed = (id, action, args, clarified = true) => + ghcpEvalListActionAllowed( + action, + args, + listPolicy(id, store, clarified, "Independent title"), + ); + assert.equal( + allowed("list-S4", "addItems", { + listName: "grocery", + items: ["apples"], + }), + true, + ); + for (const args of [ + { listName: "pantry", items: ["apples"] }, + { listName: "grocery", items: ["bananas"] }, + { listName: "grocery", items: [] }, + { listName: "grocery", items: ["apples"], extra: true }, + ]) + assert.equal(allowed("list-S4", "addItems", args), false); + assert.equal( + allowed("list-M3", "addItems", { + listName: "grocery", + items: ["bread", "oranges"], + }), + false, + ); + assert.equal( + allowed("list-M3", "clearList", { listName: "grocery" }), + true, + ); + writeLists(store, { ...listFixture, grocery: [] }); + assert.equal( + allowed("list-M3", "addItems", { + listName: "grocery", + items: ["bread", "oranges"], + }), + true, + ); + assert.equal( + allowed("list-M3", "clearList", { listName: "grocery" }), + false, + ); + assert.equal( + allowed("list-M4", "addItems", { + listName: "picnic", + items: ["water"], + }), + false, + ); + assert.equal( + allowed("list-M4", "createList", { listName: "picnic" }), + true, + ); + assert.equal( + allowed("list-M2", "removeItems", { + listName: "office", + items: ["notebook"], + }), + false, + ); + writeLists(store, { + ...listFixture, + office: [...listFixture.office, "tea", "coffee"], + picnic: [], + }); + assert.equal( + allowed("list-M2", "removeItems", { + listName: "office", + items: ["notebook"], + }), + true, + ); + assert.equal( + allowed("list-M4", "addItems", { + listName: "picnic", + items: ["water"], + }), + true, + ); + assert.throws( + () => listPolicy("list-R5", store), + /independent issue title/, + ); + for (const row of corpus) { + const policy = listFilePolicy(row.id); + assert.deepEqual(policy.writeFiles, []); + assert.deepEqual( + policy.readFiles, + row.id === "list-R4" ? ["trip.txt"] : [], + ); + assert.equal(policy.allowInventory, false); + const permit = ghcpEvalListExternalReadAllowed( + "github-cli", + "issueView", + { repo: "microsoft/TypeAgent", number: 2617 }, + listPolicy(row.id, store, true, "title"), + ); + assert.equal(permit, ["list-M5", "list-R5"].includes(row.id)); + } +}); + +test("all ambiguity cases deny list effects until genuine referent clarification", async (t) => { + const { store, files } = fixture(t); + for (const row of corpus.filter((row) => row.clarification)) { + const policy = listPolicy(row.id, store); + for (const rule of policy.rules) + assert.equal( + ghcpEvalListActionAllowed( + rule.actionName, + { + listName: rule.listName, + ...(rule.items ? { items: rule.items } : {}), + }, + policy, + ), + false, + ); + assert.equal( + listClarificationQuestion(row.id, "Run or Cancel?"), + false, + ); + assert.equal( + listClarificationQuestion(row.id, "Confirm which list to clear?"), + false, + ); + } + const result = { + caseId: "list-A4", + noPrematureListMutation: false, + listsAtClarification: expectedLists("list-S1"), + }; + writeLists(store, expectedLists("list-A4")); + assert.equal( + gradeListTrial(result, store, snapshotFiles(files)) + .noPrematureListMutation, + false, + ); + let turns = 0; + let clarified = false; + const answer = await sendWithClarification({ + session: { + sendAndWait: async () => ({ + data: { + content: + ++turns === 1 + ? "Which item should I remove?" + : "Removed milk.", + }, + }), + }, + prompt: "Remove the item.", + timeoutMs: 1000, + testCase: corpus.find((row) => row.id === "list-A4"), + canClarify: () => !clarified, + clarify: () => { + clarified = true; + return "milk"; + }, + isClarification: listClarificationQuestion, + }); + assert.equal(answer.data.content, "Removed milk."); + assert.equal(turns, 2); +}); + +test("structured list consent uses current backend context and single-use scoped continuation", () => { + const action = { + schemaName: "list", + actionName: "clearList", + parameters: { listName: "office" }, + }; + const pending = { + status: "requires_interaction", + scopeId: "s", + operationId: "o", + interactionId: "i", + prompt: { type: "confirmation", action }, + }; + const tools = [ + { + name: "typeagent-executeAction", + toolCallId: "t", + endMs: 1, + backendEventOffset: 0, + }, + ]; + const results = [ + { toolCallId: "t", result: { structuredContent: pending } }, + ]; + assert.deepEqual(fileConsentContext(tools, results, [], {}).action, action); + assert.deepEqual(fileConsentContext([], results, [], {}), {}); + const args = { + scopeId: "s", + operationId: "o", + interactionId: "i", + response: { type: "confirmation", approved: true }, + }; + const make = () => + new Map([ + ["i", { scopeId: "s", operationId: "o", response: args.response }], + ]); + const approvals = make(); + assert.equal(consumeFixtureContinuation(approvals, args, false), true); + assert.equal(consumeFixtureContinuation(approvals, args, false), false); + assert.equal(consumeFixtureContinuation(make(), args, true), false); + for (const change of [ + { scopeId: "other" }, + { operationId: "stale" }, + { interactionId: "missing" }, + { response: { type: "question", selected: 0 } }, + ]) + assert.equal( + consumeFixtureContinuation(make(), { ...args, ...change }, false), + false, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + { structuredContent: { status: "execution_uncertain" } }, + true, + ), + true, + ); + const events = [ + { + event: "action.admitted", + detail: { schemaName: "list", actionName: "getList" }, + }, + { + event: "action.completed", + detail: { + schemaName: "list", + actionName: "getList", + success: false, + recoverable: true, + }, + }, + ]; + const failure = { + structuredContent: { + status: "failed", + error: { + code: "execution_failed", + message: "ENOENT: list missing", + }, + }, + }; + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + failure, + true, + undefined, + events, + ), + false, + ); + failure.structuredContent.error.message = "permission denied: ENOENT"; + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + failure, + true, + undefined, + events, + ), + true, + ); +}); + +test("category reports preserve denominators, unknowns, latency and preparation populations", () => { + const spec = { ...metadata, repetitions: 1, order }; + const results = order.map((row, i) => ({ + ...row, + ...metadata, + status: "completed_ungraded", + routeViolations: [], + answer: "Reviewed", + grade: { + fileStateMatchesOracle: true, + listStateMatchesOracle: true, + clarificationRequested: true, + noPrematureFileMutation: true, + noPrematureListMutation: true, + }, + e2eMs: i + 1, + preparationMs: row.candidate === 4 ? 100 : null, + })); + const reviews = results.map((row) => ({ + ...row, + outcome: "success", + reason: "Independent fixture and final-answer review", + evidence: ["fixture", "answer"], + })); + const report = summarizeCategory(spec, results, reviews); + assert.equal(report.denominator, 80); + assert.equal(report.commonSuccessPairs.length, 20); + assert.equal(report.pairwiseCommonSuccess.length, 6); + assert.equal(report.pairwiseCommonSuccess[0].pairs.length, 20); + const c1Times = results + .filter((row) => row.candidate === 1) + .map((row) => row.e2eMs) + .sort((a, b) => a - b); + assert.equal(report.candidates[0].e2e.p50Ms, c1Times[9]); + assert.equal(report.candidates[0].e2e.p90Ms, c1Times[17]); + assert.equal(report.candidates[0].e2e.p95Ms, c1Times[18]); + for (const row of report.candidates) { + assert.equal(row.denominator, 20); + assert.equal(row.successfulE2e.population, 20); + assert.equal(row.commonSuccessE2e.population, 20); + assert.equal(row.cohorts.A.denominator, 5); + assert.equal(row.preparation.totalMs, row.candidate === 4 ? 2000 : 0); + } + const unknown = summarizeCategory(spec, results, []); + assert.equal(unknown.candidates[0].unknown, 20); + assert.equal(unknown.commonSuccessPairs.length, 0); + assert.equal(unknown.candidates[0].successfulE2e.p50Ms, null); + const missing = structuredClone(results); + missing.find((row) => row.candidate === 4).preparationMs = null; + missing.find((row) => row.candidate === 1).e2eMs = null; + const missingReport = summarizeCategory(spec, missing, reviews); + assert.equal(missingReport.candidates[3].preparation.missing, 1); + assert.equal(missingReport.candidates[3].preparation.totalMs, null); + assert.equal(missingReport.candidates[0].e2e.missing, 1); + const denied = structuredClone(results); + denied[0].terminalExecutionFailure = true; + assert.throws( + () => summarizeCategory(spec, denied, reviews), + /Success contradicts/, + ); + assert.throws( + () => + summarizeCategory( + { ...spec, protocolVersion: 5 }, + results, + reviews, + ), + /historical/, + ); + assert.throws( + () => summarizeCategory(spec, results.slice(1), reviews), + /schedule/, + ); + const malformed = structuredClone(results); + malformed[0].category = "common-files"; + assert.throws( + () => summarizeCategory(spec, malformed, reviews), + /schedule/, + ); + assert.equal( + preliminaryGrade({ ...results[0], status: "timed_out" }, {}).outcome, + "incomplete", + ); +}); + +test("common category report independently retains 140 rows and missing external answer evidence remains pending", () => { + const common = evaluationCategory(); + const cases = categoryCorpus( + common, + "fixture", + "microsoft/TypeAgent", + 3058, + 3067, + 2617, + ); + const order = categorySchedule(common, cases, common.candidates, 1).order; + const metadata = { + protocolVersion, + category: common.name, + corpusVersion: common.corpusVersion, + }; + const rows = order.map((row) => ({ + ...metadata, + ...row, + status: "completed_ungraded", + routeViolations: [], + answer: "Reviewed answer", + e2eMs: 10, + grade: { + fileStateMatchesOracle: true, + listStateUnchanged: true, + clarificationRequested: true, + noPrematureFileMutation: true, + }, + preparationMs: row.candidate === 4 ? 2 : null, + })); + const reviews = rows.map((row) => ({ + ...row, + outcome: "success", + reason: "Synthetic review", + evidence: ["synthetic"], + })); + const report = summarizeCategory( + { ...metadata, order, repetitions: 1 }, + rows, + reviews, + ); + assert.equal(report.denominator, 140); + assert.equal(report.candidates.length, 7); + assert.equal(report.pairwiseCommonSuccess.length, 21); + assert.equal( + preliminaryGrade( + rows.find((row) => row.caseId === "R5"), + {}, + ).reason, + "independent_answer_evidence_unavailable", + ); +}); + +test("offline list preflight validates six actual operation outcomes and denies unexpected confirmation context", async (t) => { + const { directory } = fixture(t); + const data = path.join(directory, "data"); + fs.mkdirSync(data); + const store = path.join(data, "lists.json"); + const policyFile = path.join(directory, "policy.json"); + fs.writeFileSync(policyFile, JSON.stringify(preflightListPolicy(""))); + writeLists(store, {}); + const invoke = async ({ name, arguments: args }) => { + assert.equal(name, "typeagent-executeAction"); + const lists = readLists(store); + const p = args.parameters; + if (args.actionName === "createList") lists[p.listName] = []; + if (args.actionName === "addItems") lists[p.listName].push(...p.items); + if (args.actionName === "removeItems") + lists[p.listName] = lists[p.listName].filter( + (item) => !p.items.includes(item), + ); + if (args.actionName === "clearList") lists[p.listName] = []; + writeLists(store, lists); + return { structuredContent: { status: "completed" } }; + }; + const env = { + TYPEAGENT_USER_DATA_DIR: data, + TYPEAGENT_GHCP_EVAL_LIST_POLICY: policyFile, + }; + const result = { externalEvidence: [] }; + assert.equal( + await verifyListPreflight({ callTool: invoke }, "scope", env, result), + store, + ); + assert.equal(result.externalEvidence.length, 6); + resetLists(store); + assert.equal(listsUnchanged(store), true); + await assert.rejects(() => + verifyListPreflight( + { + callTool: async () => ({ + structuredContent: { + status: "requires_interaction", + scopeId: "wrong", + prompt: { type: "confirmation" }, + }, + }), + }, + "scope", + env, + { externalEvidence: [] }, + ), + ); +}); diff --git a/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts b/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts index 80da2ae11a..65a3f9d3d9 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts @@ -3,6 +3,8 @@ import { assertGhcpEvalAction, + isGhcpEvalReadOnlyAction, + isGhcpEvalRecoverableReadError, markGhcpEvalExecutionFailure, recordGhcpEvalEvent, } from "./ghcpEvalPolicy.js"; @@ -545,11 +547,15 @@ export async function executeAction( schemaName, ); + const recoverable = + isGhcpEvalReadOnlyAction(schemaName, action.actionName) && + isGhcpEvalRecoverableReadError(outcome.result.error); if (outcome.result.error !== undefined) - markGhcpEvalExecutionFailure(); + markGhcpEvalExecutionFailure(recoverable); recordGhcpEvalEvent("action.completed", { ...eventData, success: outcome.result.error === undefined, + recoverable, elapsedMs: Date.now() - actionStartedAt, }); logActionCompleted(systemContext.logger, { diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalDiagnostics.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalDiagnostics.ts new file mode 100644 index 0000000000..f30610a7db --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalDiagnostics.ts @@ -0,0 +1,149 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { createHash } from "node:crypto"; +import { + isGhcpEvalFixtureFile, + readGhcpEvalFilePolicy, +} from "./ghcpEvalFiles.js"; + +const digest = (value: string) => + createHash("sha256").update(value).digest("hex"); +const fixtureNames = new Set([ + "report-a.txt", + "report-b.txt", + "trip.txt", + "grocery.txt", + "pantry.txt", + "packing.txt", + "errands.txt", + "grocery-backup.txt", +]); + +function pathDiagnostic(value: unknown, root: string) { + if (typeof value !== "string") return { kind: "missing_or_non_string" }; + const name = path.basename(value); + const knownName = fixtureNames.has(name); + return { + kind: path.isAbsolute(value) ? "absolute" : "relative", + requestedSha256: digest(value), + characters: value.length, + fixtureName: knownName ? name : null, + directFixtureFile: + knownName && isGhcpEvalFixtureFile(value, root, [name], true), + // Never persist external paths or prose accidentally bound as a path. + sanitizedRequestedPath: knownName + ? `\\${name}` + : "", + }; +} + +function fileDenialReason( + action: string, + args: Record, + root: string, +) { + const policy = readGhcpEvalFilePolicy(); + if (!policy) return "missing_file_policy"; + if (!["readFile", "writeFile", "copyFile", "listFiles"].includes(action)) + return "file_action_not_in_contract"; + if (action === "listFiles") + return "inventory_scope_or_recursion_not_allowed"; + const fields = action === "copyFile" ? ["source", "destination"] : ["path"]; + for (const field of fields) { + const value = args[field]; + if (typeof value !== "string") return `${field}_missing_or_non_string`; + if (!path.isAbsolute(value)) return `${field}_not_absolute`; + } + if (action === "readFile" && !policy.readsEnabled) + return "read_before_clarification_or_not_in_case"; + if (action !== "readFile" && !policy.writesEnabled) + return "write_before_clarification_or_not_in_case"; + if (action === "copyFile" && !policy.allowCopy) return "copy_not_in_case"; + if (args.recurse === true) return "recursive_effect_not_allowed"; + if ( + action === "copyFile" && + policy.allowCopy && + !isGhcpEvalFixtureFile(String(args.source), root, [ + policy.allowCopy.source, + ]) + ) + return "copy_source_outside_case_or_noncanonical_file"; + const target = String(action === "copyFile" ? args.destination : args.path); + const names = action === "readFile" ? policy.readFiles : policy.writeFiles; + if (!isGhcpEvalFixtureFile(target, root, names, action !== "readFile")) + return "target_outside_case_or_noncanonical_file"; + if (action === "writeFile" && typeof args.content !== "string") + return "content_missing_or_non_string"; + return "prerequisite_state_mismatch"; +} + +function correlation() { + const file = process.env.TYPEAGENT_GHCP_EVAL_CORRELATION; + if (!file) return { availability: "not_supplied" }; + const input = JSON.parse(fs.readFileSync(file, "utf8")); + const hashes = Object.fromEntries( + ["session", "call", "scope", "operation", "interaction"].map((key) => [ + `${key}Sha256`, + typeof input[`${key}Sha256`] === "string" && + /^[a-f0-9]{64}$/.test(input[`${key}Sha256`]) + ? input[`${key}Sha256`] + : null, + ]), + ); + return { + ...hashes, + caseId: /^(list-)?[SMRA][1-5]$/.test(input.caseId) + ? input.caseId + : null, + candidate: Number.isInteger(input.candidate) ? input.candidate : null, + callSequence: Number.isInteger(input.callSequence) + ? input.callSequence + : null, + }; +} + +export function ghcpEvalDenialDiagnostic( + schemaName: string, + actionName: string, + parameters: unknown, + root: string, +) { + const args = + parameters !== null && typeof parameters === "object" + ? (parameters as Record) + : {}; + const fileAction = schemaName === "powershell.powershell-files"; + return { + schemaName, + actionName, + category: + process.env.TYPEAGENT_GHCP_EVAL_CATEGORY === "lists" + ? "lists" + : "common-files", + correlation: correlation(), + reason: fileAction + ? fileDenialReason(actionName, args, root) + : "action_outside_active_category_policy", + canonicalScope: { + label: "", + sha256: digest(fs.realpathSync(root)), + }, + paths: fileAction + ? Object.fromEntries( + ["path", "source", "destination"].map((field) => [ + field, + pathDiagnostic(args[field], root), + ]), + ) + : undefined, + content: + fileAction && typeof args.content === "string" + ? { characters: args.content.length } + : undefined, + recurse: args.recurse === true, + append: args.append === true, + }; +} diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalLists.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalLists.ts new file mode 100644 index 0000000000..456cf9ad3e --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalLists.ts @@ -0,0 +1,182 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; + +export type GhcpEvalListRule = { + actionName: string; + listName?: string; + items?: string[]; + before?: Record; +}; + +export type GhcpEvalListPolicy = { + version: 1; + category: "lists"; + store: string; + enabled: boolean; + rules: GhcpEvalListRule[]; + externalReads?: { + schemaName: string; + actionName: string; + repo: string; + number: number; + }[]; +}; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function strings(value: unknown): value is string[] { + return ( + Array.isArray(value) && value.every((item) => typeof item === "string") + ); +} + +function validRule(value: unknown): boolean { + if (!isRecord(value)) return false; + const { actionName, listName, items, before } = value; + if (actionName === "listLists") + return ( + listName === undefined && + items === undefined && + before === undefined + ); + if (typeof listName !== "string" || !listName) return false; + const itemAction = + actionName === "addItems" || actionName === "removeItems"; + if ( + !( + itemAction || + ["getList", "createList", "clearList"].includes(String(actionName)) + ) + ) + return false; + if ( + itemAction ? !strings(items) || items.length === 0 : items !== undefined + ) + return false; + return ( + before === undefined || + (isRecord(before) && + Object.values(before).every( + (value) => value === null || strings(value), + )) + ); +} + +export function readGhcpEvalListPolicy( + file = process.env.TYPEAGENT_GHCP_EVAL_LIST_POLICY, +): GhcpEvalListPolicy | undefined { + if (!file) return undefined; + const policy = JSON.parse(fs.readFileSync(file, "utf8")); + if ( + !isRecord(policy) || + policy.version !== 1 || + policy.category !== "lists" || + typeof policy.store !== "string" || + typeof policy.enabled !== "boolean" || + !Array.isArray(policy.rules) || + !policy.rules.every(validRule) + ) + throw new Error("Invalid GHCP eval list policy"); + if ( + policy.externalReads !== undefined && + (!Array.isArray(policy.externalReads) || + policy.externalReads.some( + (read) => + !isRecord(read) || + read.schemaName !== "github-cli" || + read.actionName !== "issueView" || + typeof read.repo !== "string" || + !Number.isSafeInteger(read.number) || + Number(read.number) <= 0, + )) + ) + throw new Error("Invalid GHCP eval external read policy"); + return policy as GhcpEvalListPolicy; +} + +/** The launcher supplies exact per-case operations, never a general list grant. */ +export function ghcpEvalListActionAllowed( + actionName: string, + parameters: unknown, + policy: GhcpEvalListPolicy, +): boolean { + if ( + !policy.enabled || + typeof parameters !== "object" || + parameters === null || + Array.isArray(parameters) + ) + return false; + const args = parameters as Record; + return policy.rules.some((rule) => { + if (rule.actionName !== actionName) return false; + const keys = rule.items + ? ["listName", "items"] + : rule.listName + ? ["listName"] + : []; + if ( + Object.keys(args).length !== keys.length || + keys.some((key) => !(key in args)) || + args.listName !== rule.listName + ) + return false; + const allowedItems = rule.items; + if ( + allowedItems && + (!Array.isArray(args.items) || + args.items.length === 0 || + !args.items.every( + (item) => + typeof item === "string" && allowedItems.includes(item), + )) + ) + return false; + if (!rule.before) return true; + const stat = fs.lstatSync(policy.store); + if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1) + throw new Error("Unsafe GHCP eval list store"); + const lists: { name: string; items: string[] }[] = JSON.parse( + fs.readFileSync(policy.store, "utf8"), + ); + return Object.entries(rule.before).every(([name, expected]) => { + const list = lists.find((entry) => entry.name === name); + return expected === null + ? list === undefined + : list !== undefined && + JSON.stringify([...list.items].sort()) === + JSON.stringify([...expected].sort()); + }); + }); +} + +export function ghcpEvalListExternalReadAllowed( + schemaName: string, + actionName: string, + parameters: unknown, + policy: GhcpEvalListPolicy, +): boolean { + if ( + !policy.enabled || + typeof parameters !== "object" || + parameters === null + ) + return false; + const args = parameters as Record; + return ( + schemaName === "github-cli" && + actionName === "issueView" && + Object.keys(args).every((key) => key === "repo" || key === "number") && + policy.externalReads?.some( + (read) => + read.schemaName === schemaName && + read.actionName === actionName && + read.repo === args.repo && + read.number === args.number, + ) === true + ); +} diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts index 4f3abc05b6..acf6236320 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts @@ -4,15 +4,21 @@ import fs from "node:fs"; import path from "node:path"; import { isGhcpEvalArtifact } from "./ghcpEvalArtifacts.js"; +import { ghcpEvalDenialDiagnostic } from "./ghcpEvalDiagnostics.js"; import { ghcpEvalFileActionAllowed, readGhcpEvalFilePolicy, } from "./ghcpEvalFiles.js"; +import { + ghcpEvalListActionAllowed, + readGhcpEvalListPolicy, + ghcpEvalListExternalReadAllowed, +} from "./ghcpEvalLists.js"; let executionFailureObserved = false; -export function markGhcpEvalExecutionFailure(): void { - if (process.env.TYPEAGENT_GHCP_EVAL_FIXTURES !== undefined) +export function markGhcpEvalExecutionFailure(recoverable = false): void { + if (process.env.TYPEAGENT_GHCP_EVAL_FIXTURES !== undefined && !recoverable) executionFailureObserved = true; } @@ -49,6 +55,55 @@ const reads = new Map>([ ], ]); +export function isGhcpEvalReadOnlyAction( + schemaName: string, + actionName: string, +): boolean { + return ( + reads.get(schemaName)?.has(actionName) === true || + (schemaName === "powershell.powershell-files" && + (actionName === "readFile" || actionName === "listFiles")) || + (schemaName === "list" && + (actionName === "getList" || actionName === "listLists")) + ); +} + +// An ordinary I/O failure is positive evidence, unlike an absent SDK error. +// Authorization, cancellation and uncertain delivery always take precedence. +export function isGhcpEvalRecoverableReadError(error: unknown): boolean { + return ( + typeof error === "string" && + !/(deni(?:ed|al)|unauthoriz|forbidden|permission|policy|sandbox|reject|EACCES|EPERM|\b401\b|\b403\b|cancel|uncertain|abort)/i.test( + error, + ) && + /\b(ENOENT|ENOTDIR|EISDIR|ETIMEDOUT|ECONNRESET|EAI_AGAIN)\b/.test(error) + ); +} + +function categoryActionAllowed( + schemaName: string, + actionName: string, + parameters: unknown, +): boolean { + if (process.env.TYPEAGENT_GHCP_EVAL_CATEGORY === "lists") { + const policy = readGhcpEvalListPolicy(); + if (!policy) return false; + return schemaName === "list" + ? ghcpEvalListActionAllowed(actionName, parameters, policy) + : ghcpEvalListExternalReadAllowed( + schemaName, + actionName, + parameters, + policy, + ); + } + return schemaName === "list" + ? !process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY || + (actionName === "listLists" && + readGhcpEvalFilePolicy()?.allowListInventory === true) + : reads.get(schemaName)?.has(actionName) === true; +} + /** Apply only to an explicitly isolated evaluation server, never normal sessions. */ export function assertGhcpEvalAction( schemaName: string, @@ -61,14 +116,11 @@ export function assertGhcpEvalAction( throw new Error( "GHCP eval stopped execution after a failed or cancelled action", ); + const listCategory = process.env.TYPEAGENT_GHCP_EVAL_CATEGORY === "lists"; if ( - (schemaName === "list" && - (!process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY || - (actionName === "listLists" && - readGhcpEvalFilePolicy()?.allowListInventory === true))) || + categoryActionAllowed(schemaName, actionName, parameters) || schemaName === "dispatcher" || - schemaName.startsWith("dispatcher.") || - reads.get(schemaName)?.has(actionName) + schemaName.startsWith("dispatcher.") ) { return; } @@ -94,7 +146,7 @@ export function assertGhcpEvalAction( typeof parameters.path === "string" ) { if (isGhcpEvalArtifact(parameters.path)) return; - if (!policy) { + if (!policy && !listCategory) { const requested = fs .realpathSync(parameters.path) .toLowerCase(); @@ -109,7 +161,16 @@ export function assertGhcpEvalAction( } } } - recordGhcpEvalEvent("action.denied", { schemaName, actionName }); + if (process.env.TYPEAGENT_GHCP_EVAL_TRACE) + recordGhcpEvalEvent( + "action.denied", + ghcpEvalDenialDiagnostic( + schemaName, + actionName, + parameters, + fixtureRoot, + ), + ); markGhcpEvalExecutionFailure(); throw new Error( `GHCP eval policy denied ${schemaName}.${actionName} before execution`, diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalLists.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalLists.spec.ts new file mode 100644 index 0000000000..e253f012ca --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalLists.spec.ts @@ -0,0 +1,270 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { + readGhcpEvalListPolicy, + ghcpEvalListActionAllowed, +} from "../src/execute/ghcpEvalLists.js"; +import { + assertGhcpEvalAction, + isGhcpEvalReadOnlyAction, + isGhcpEvalRecoverableReadError, +} from "../src/execute/ghcpEvalPolicy.js"; +import { ghcpEvalDenialDiagnostic } from "../src/execute/ghcpEvalDiagnostics.js"; + +describe("category-specific evaluation list admission", () => { + let directory: string; + let saved: NodeJS.ProcessEnv; + beforeEach(() => { + saved = { ...process.env }; + directory = fs.mkdtempSync(path.join(os.tmpdir(), "list-policy-")); + delete process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + delete process.env.TYPEAGENT_GHCP_EVAL_TRACE; + process.env.TYPEAGENT_GHCP_EVAL_CATEGORY = "lists"; + process.env.TYPEAGENT_GHCP_EVAL_LIST_POLICY = path.join( + directory, + "policy.json", + ); + process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = path.join( + directory, + "files.json", + ); + fs.writeFileSync( + process.env.TYPEAGENT_GHCP_EVAL_LIST_POLICY, + JSON.stringify({ + version: 1, + category: "lists", + store: path.join(directory, "lists.json"), + enabled: true, + rules: [ + { + actionName: "addItems", + listName: "grocery", + items: ["apples"], + }, + ], + externalReads: [ + { + schemaName: "github-cli", + actionName: "issueView", + repo: "microsoft/TypeAgent", + number: 2617, + }, + ], + }), + ); + fs.writeFileSync( + process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, + JSON.stringify({ + version: 1, + readFiles: [], + writeFiles: [], + readsEnabled: false, + writesEnabled: false, + allowInventory: false, + prerequisites: {}, + }), + ); + }); + afterEach(() => { + process.env = saved; + fs.rmSync(directory, { recursive: true, force: true }); + }); + it("admits only the active case, not global list/file/external access", () => { + expect(() => + assertGhcpEvalAction( + "list", + "addItems", + { listName: "grocery", items: ["apples"] }, + directory, + ), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction( + "github-cli", + "issueView", + { repo: "microsoft/TypeAgent", number: 2617 }, + directory, + ), + ).not.toThrow(); + for (const [schema, action, parameters] of [ + ["list", "clearList", { listName: "grocery" }], + ["list", "addItems", { listName: "pantry", items: ["apples"] }], + [ + "github-cli", + "issueView", + { repo: "microsoft/TypeAgent", number: 1 }, + ], + ["github-cli", "prFiles", {}], + ["ipconfig", "displayFullConfigurationInformation", {}], + [ + "powershell.powershell-files", + "writeFile", + { path: path.join(directory, "trip.txt"), content: "x" }, + ], + ] as const) + expect(() => + assertGhcpEvalAction(schema, action, parameters, directory), + ).toThrow("before execution"); + }); + it("never widens common files or normal sessions", () => { + process.env.TYPEAGENT_GHCP_EVAL_CATEGORY = "common-files"; + expect(() => + assertGhcpEvalAction( + "list", + "addItems", + { listName: "grocery", items: ["apples"] }, + directory, + ), + ).toThrow("before execution"); + expect(() => + assertGhcpEvalAction("list", "clearList", {}), + ).not.toThrow(); + }); + it("missing or malformed list policies fail closed without legacy read fallback", () => { + delete process.env.TYPEAGENT_GHCP_EVAL_LIST_POLICY; + expect(() => + assertGhcpEvalAction("list", "addItems", {}, directory), + ).toThrow("before execution"); + delete process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY; + fs.writeFileSync(path.join(directory, "trip.txt"), "private"); + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "readFile", + { path: path.join(directory, "trip.txt") }, + directory, + ), + ).toThrow("before execution"); + const file = path.join(directory, "bad.json"); + for (const rules of [ + [{ actionName: "deleteList", listName: "grocery" }], + [{ actionName: "addItems", listName: "grocery", items: [3] }], + [null], + ]) { + fs.writeFileSync( + file, + JSON.stringify({ + version: 1, + category: "lists", + store: "", + enabled: true, + rules, + }), + ); + expect(() => readGhcpEvalListPolicy(file)).toThrow( + "Invalid GHCP eval list policy", + ); + } + }); + it("disabled ambiguity policy rejects even exact approved item arguments", () => { + const policy = readGhcpEvalListPolicy()!; + expect( + ghcpEvalListActionAllowed( + "addItems", + { listName: "grocery", items: ["apples"] }, + { ...policy, enabled: false }, + ), + ).toBe(false); + }); + it("safe recovery remains read-only and denial or uncertainty wins", () => { + expect(isGhcpEvalReadOnlyAction("list", "getList")).toBe(true); + expect(isGhcpEvalReadOnlyAction("list", "listLists")).toBe(true); + for (const action of [ + "addItems", + "removeItems", + "clearList", + "createList", + ]) + expect(isGhcpEvalReadOnlyAction("list", action)).toBe(false); + expect( + isGhcpEvalRecoverableReadError("ENOENT: list snapshot missing"), + ).toBe(true); + for (const error of [ + undefined, + "permission denied: ENOENT", + "uncertain: ETIMEDOUT", + "cancelled: ECONNRESET", + ]) + expect(isGhcpEvalRecoverableReadError(error)).toBe(false); + }); + it("denial diagnostics preserve scope and correlation without leaking path prose or content", () => { + process.env.TYPEAGENT_GHCP_EVAL_CORRELATION = path.join( + directory, + "correlation.json", + ); + const id = "a".repeat(64); + fs.writeFileSync( + process.env.TYPEAGENT_GHCP_EVAL_CORRELATION, + JSON.stringify({ + caseId: "list-R4", + candidate: 3, + callSequence: 2, + callSha256: id, + scopeSha256: id, + operationSha256: id, + interactionSha256: id, + }), + ); + const diagnostic = ghcpEvalDenialDiagnostic( + "powershell.powershell-files", + "readFile", + { + path: "full contents of files: C:\\private-user\\secrets.txt", + content: "private credential value", + token: "private credential value", + }, + directory, + ); + expect(diagnostic.reason).toBe("path_not_absolute"); + expect(diagnostic.correlation).toMatchObject({ + callSha256: id, + scopeSha256: id, + caseId: "list-R4", + }); + expect(diagnostic.canonicalScope.sha256).toHaveLength(64); + expect(JSON.stringify(diagnostic)).not.toMatch( + /private-user|secrets|credential|full contents/, + ); + expect(JSON.stringify(diagnostic)).not.toContain(directory); + expect( + ghcpEvalDenialDiagnostic( + "powershell.powershell-files", + "readFile", + {}, + directory, + ).reason, + ).toBe("path_missing_or_non_string"); + expect( + ghcpEvalDenialDiagnostic( + "powershell.powershell-files", + "readFile", + { + path: path.join(directory, "trip.txt"), + }, + directory, + ).reason, + ).toBe("read_before_clarification_or_not_in_case"); + process.env.TYPEAGENT_GHCP_EVAL_TRACE = path.join( + directory, + "trace.jsonl", + ); + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "readFile", + {}, + directory, + ), + ).toThrow("before execution"); + const trace = JSON.parse( + fs.readFileSync(process.env.TYPEAGENT_GHCP_EVAL_TRACE, "utf8"), + ); + expect(trace.event).toBe("action.denied"); + expect(trace.detail.reason).toBe("path_missing_or_non_string"); + expect(trace.detail.category).toBe("lists"); + }); +}); diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts index 76e9080ed9..79ac99f607 100644 --- a/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts @@ -7,10 +7,60 @@ import path from "node:path"; import { assertGhcpEvalAction, ghcpEvalExecutionStopped, + isGhcpEvalReadOnlyAction, + isGhcpEvalRecoverableReadError, markGhcpEvalExecutionFailure, } from "../src/execute/ghcpEvalPolicy.js"; describe("isolated GHCP evaluation action policy", () => { + it("allows affirmative read failures without clearing a prior terminal stop", () => { + const original = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + try { + process.env.TYPEAGENT_GHCP_EVAL_FIXTURES = "fixture"; + const recoverable = + isGhcpEvalReadOnlyAction("github-cli", "prFiles") && + isGhcpEvalRecoverableReadError("ECONNRESET: socket closed"); + expect(recoverable).toBe(true); + markGhcpEvalExecutionFailure(recoverable); + expect(ghcpEvalExecutionStopped()).toBe(false); + expect(() => + assertGhcpEvalAction("github-cli", "prChecks", {}), + ).not.toThrow(); + markGhcpEvalExecutionFailure(); + markGhcpEvalExecutionFailure(recoverable); + expect(ghcpEvalExecutionStopped()).toBe(true); + } finally { + if (original === undefined) + delete process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + else process.env.TYPEAGENT_GHCP_EVAL_FIXTURES = original; + } + }); + it("does not infer safe recovery from mutations, missing errors, denial or cancellation", () => { + for (const action of ["readFile", "listFiles"]) + expect( + isGhcpEvalReadOnlyAction("powershell.powershell-files", action), + ).toBe(true); + for (const action of ["writeFile", "copyFile"]) + expect( + isGhcpEvalReadOnlyAction("powershell.powershell-files", action), + ).toBe(false); + expect(isGhcpEvalReadOnlyAction("list", "clearList")).toBe(false); + expect(isGhcpEvalReadOnlyAction("ipconfig", "releaseAddress")).toBe( + false, + ); + for (const error of [ + undefined, + "", + "unknown failure", + "permission denied: ENOENT", + "cancelled after ECONNRESET", + "execution_uncertain: ETIMEDOUT", + ]) + expect(isGhcpEvalRecoverableReadError(error)).toBe(false); + expect(isGhcpEvalRecoverableReadError("ENOENT: missing file")).toBe( + true, + ); + }); it("stops subsequent execution after failure only in the isolated eval process", () => { const original = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; try {