diff --git a/ts/packages/copilot-plugin-eval/LICENSE b/ts/packages/copilot-plugin-eval/LICENSE new file mode 100644 index 0000000000..9e841e7a26 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/LICENSE @@ -0,0 +1,21 @@ + MIT License + + Copyright (c) Microsoft Corporation. + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE diff --git a/ts/packages/copilot-plugin-eval/README.AUTOGEN.md b/ts/packages/copilot-plugin-eval/README.AUTOGEN.md new file mode 100644 index 0000000000..eed31f7210 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/README.AUTOGEN.md @@ -0,0 +1,43 @@ + + + + + + + + +# @typeagent/copilot-plugin-eval — AI-generated documentation + +> 📝 **Placeholder documentation — not yet AI-authored.** Re-run `pnpm docs:generate:llm --package copilot-plugin-eval` to populate this file, or read [`./README.md`](./README.md) for the hand-written documentation in the meantime. The deterministic Reference section below is already populated. + +## Overview + +Initial end-to-end Copilot plugin evaluation harness + +## Reference + +> ⚙️ **Auto-generated, no AI involvement.** Built deterministically from `package.json`, `src/`, and the workspace dependency graph at the commit recorded in the staleness footer at the end of this file. Hand edits to this file will be overwritten on the next run. + +### Entry points + +_No public exports declared in `package.json`._ + +### Dependencies + +Workspace: + +- [@typeagent/copilot-plugin](../../packages/copilot-plugin/README.md) +- [agent-dispatcher](../../packages/dispatcher/dispatcher/README.md) +- [agent-server](../../packages/agentServer/server/README.md) + +External: _None at runtime._ + +### Files of interest + +_No tracked source files under `./src/`._ + +--- + +_Auto-generated against commit `ba6a8b85088604d88c37360ec0bc092911ad38bc` on `2026-09-25T22:33:54.079Z` by `docs-generate.yml`. Links validated at that commit; the working tree may have drifted by up to 24h. Re-run `pnpm --filter @typeagent/copilot-plugin-eval docs:verify-links` to spot-check._ + + diff --git a/ts/packages/copilot-plugin-eval/README.md b/ts/packages/copilot-plugin-eval/README.md new file mode 100644 index 0000000000..e96e0fa630 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/README.md @@ -0,0 +1,359 @@ +# TypeAgent Copilot end-to-end evaluation + +Updated: 2026-09-25. This is the maintained version of the original GHCP +evaluation methodology, alongside its harness, corpus, grading, and tests. + +## Purpose and limitations + +**This is a simple initial evaluation for ballpark estimates**, not a +production-quality benchmark or a statistically powered comparison. It samples +twenty tasks across three domains to explore task completion, user-visible +latency, and workflow overhead. Five cases per cohort and one historical +balanced repetition cannot establish precise tail percentiles, non-inferiority, +or a general winning strategy. + +**We acknowledge potential environment-isolation risks from implementing and +running this evaluation alongside the Copilot plugin.** Moving its code into +this separate directory prevents it from being part of the plugin source +layout; it does not create an OS security boundary or prove full isolation. +The harness still reuses plugin staging/discovery infrastructure, the same +host, authenticated CLI, runtime dependencies, model configuration, and +read-only external services. Native tools and inherited process configuration +can expose environmental differences. Shared caches, provider behavior and +changing GitHub/network data can confound comparisons. This implementation is +aware of these limitations; isolated fixtures, scoped permissions, private +data/temp directories and trace audits mitigate them but do not eliminate them. +Never describe this implementation as a sandbox or isolation certification. + +Only actual Copilot SDK conversations count as end-to-end trials. A supplied +correct action, mocked model selection, discovery smoke test or dispatcher +microbenchmark is not a substitute. Throughput/load tests, cold-start campaigns, +Direct-hook comparisons and a broad adversarial suite are outside this initial +scope. Weather is removed and calendar is deferred. + +## Implementation and protocol history + +**Current corpus: `common-files-v1`, protocol 5.** Nine list-dependent tasks +have been replaced with ordinary file tasks so every candidate has a comparable +capability: 20 applicable cases per candidate, 140 executions per repetition, +no native N/A slots. This is a new workload, not a rescore of old trials. Do +not pool its results with the historical list corpus or reuse an old preflight. +This base branch retains the strict failure gate; the separate #3077 layer +preserves its positive-evidence recovery policy when integrated with this corpus. + +| Version | Meaning | +| ---------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Protocol 2 | Historical measured implementation at `e847c7a00907a1f4e2c6c4426c935b964fd9997c`: completed 140 trials. An earlier partial pass had harness-invalid records, which were excluded rather than scored as candidate failures. | +| Protocol 3 | Strict native failure/no-replay correction at `4275a7ca7743b761b7a108971de34184ebf6e289`; tested offline, not live measured. This base runner still uses that policy. | +| Portability | `b47f8bacd5073aa226392cc6ccb47a9a7e4caac7` accepts configured temp-parent aliases while retaining artifact provenance checks, and uses platform-native test paths. | +| Protocol 4 follow-up (#3077) | Prospective positive-evidence safe-read recovery, frozen native applicability and replacement A4 oracle. Kept in its separate dependent PR; do not infer its runtime behavior from this base runner. | +| Model and layout revision | Eval-only scripts now reside here. Future Copilot sessions explicitly use Luna 5.6 (`gpt-5.6-luna`), not the historical `gpt-5.6-sol`. No Luna rerun or improved measured outcome is claimed. | + +Preserve original results, grades, run specifications and safety audits. +Retrospective reporting amendments do not rewrite observations or prove what +a stopped trial would have done. Result-entity product fixes are separate +from the evaluation harness and do not establish new measured success rates. + +## Running and package boundaries + +From `ts`, after normal worktree dependency provisioning: + +```powershell +pnpm exec fluid-build '^@typeagent/copilot-plugin-eval$' -t build --dep +pnpm --filter @typeagent/copilot-plugin-eval test +``` + +This dedicated package is JavaScript-only. Its build checks executable syntax; +dependency-aware builds prepare the plugin, dispatcher and agent-server. +Tests are offline. Plugin installation/bundling does not include this harness. +`copilot-plugin/scripts/discovery-e2e.mjs` remains shared plugin test +infrastructure; dispatcher-side permission, credit and artifact guards stay +at their actual runtime enforcement boundaries rather than moving into a +client-only package. + +Live commands below require separate authorization, an authenticated Copilot +executable, existing model configuration, a reconciled model-specific credit +ledger, and verified handlers/catalog. They can incur model usage; building or +testing this package does not run them. + +```text +node packages\copilot-plugin-eval\scripts\ghcp-eval-preflight.mjs --external-evidence +node packages\copilot-plugin-eval\scripts\ghcp-eval.mjs pilot +node packages\copilot-plugin-eval\scripts\ghcp-eval.mjs 1,2,3,4,5,6,7 measured S1 7 1 +node packages\copilot-plugin-eval\scripts\ghcp-credit-probe.mjs +``` + +Use nonsynchronized local directories for live databases and locks. Preserve +sanitized results and specifications in durable storage afterwards. The oracle +JSON pins issue/PR evidence and a relative `readinessFile` naming a successful +preflight. The current concrete GitHub inputs are PRs 3058/3067 and issue 2617 +in microsoft/TypeAgent; verify availability and contemporaneous evidence before +a new run. Do not silently replace targets during a frozen run. + +## Explicit model and admission + +`scripts/ghcp-eval-config.mjs` pins **Luna 5.6, `gpt-5.6-luna`**. The measured +outer conversation, same-binding contract preparation, and TypeAgent nested +Copilot reasoning use this identity. Preflight and the optional credit probe +also enforce it. Main trials keep high reasoning effort; the small accounting +probe uses low effort. TypeAgent translation and embedding providers remain +separately configured and must be recorded; this pin does not silently change +their model identities. + +Every entry point rejects a mismatched ledger before starting services or model +work. Do not relabel a historical Sol ledger or assume its per-request bounds +apply to Luna. Verify current model availability, credit rates, context/output +bounds and nested-request accounting; reconcile cumulative prior charges in a +new run specification. An unavailable model is a blocker, not permission to +substitute one. No live availability/pricing probe is part of this migration. + +The authorized ceiling is **50,000 cumulative Copilot AI credits**, superseding +the original 20,000 and interim 40,000 limits, not a fresh allowance per run. +Include planning, implementation, preparation, pilots, failed/cancelled work, +nested reasoning, grading and reporting. Reserve report headroom, retain +unsettled maximum reservations and stop new admissions if accounting is unclear. +Externally billed translation/embedding usage is reported separately; unknown +usage is not zero. This is not a general-purpose pricing or billing-hard-stop +service. + +The proxy reserves before forwarding, settles explicit billing fields, rejects +model mismatches/WebSockets and retains unknown charges. Limits include 24 +requests per scoped session and 2,000 cumulative ledger requests. The SDK +60-credit session limit is additional and **soft**, not the hard admission +mechanism. Historical request reservations are not certified bounds for Luna. +Moving code or changing models never reopens a closed ledger. + +## Seven candidates and fallback + +| # | Candidate | Allowed outer entry | +| --- | ----------------------------------- | ---------------------------------------------------------------------------------------------- | +| 1 | NL-MCP, fallback disabled | `typeagent-processCommand` only | +| 2 | NL-MCP, fallback enabled | Same as 1 | +| 3 | Structured discovery | `searchActions`, `executeAction`, structured continuation/cancellation; never `processCommand` | +| 4 | Structured current-contract reuse | Same structured surface; contracts earned by discovery earlier in the same binding | +| 5 | Production mixed, fallback disabled | NL and structured interfaces under production routing guidance | +| 6 | Production mixed, fallback enabled | Same as 5 | +| 7 | Native Copilot only | Pinned native tools; no TypeAgent plugin, hooks, tools, guidance or special storage access | + +All TypeAgent candidates use MCP. Direct-hook execution is excluded to keep +transport and the Copilot agent loop comparable; structured tools also exist +in Direct mode but are still MCP tools. Pure candidates neutralize conflicting +plugin guidance and enforce interfaces at the tool boundary. Mixed candidates +retain production guidance. Auxiliary outer workspace/macro/skill MCP servers +are omitted; do not silently alter the internal reasoning toolset. + +Fallback means TypeAgent's **failed translation to Copilot reasoning** +transition, controlled by `translationReasoningFallback`. Grammar/cache misses +proceeding to LLM translation, normal action selection, successful-result +reasoning, multi-tool orchestration and outer tool-error recovery are not this +fallback. Preserve grammar/cache/LLM translation in each toggle pair and trace +the actual decision, entry and outcome (`TYPEAGENT_GHCP_EVAL_TRACE`). + +Compare 1/3/4 for resolution strategies, 1/2 and 5/6 for fallback, and pure +versus mixed separately. Reuse must not preload answers, exact action arguments +or resolved ambiguous referents. Report discovery preparation separately and +include it in whole-workflow amortization, not just subsequent-turn savings. + +## Fixtures, domains and twenty cases + +Four cohorts contain five cases each. The new common-capability corpus uses +GitHub CLI, registered PowerShell-file and read-only IP configuration actions. +PowerShell uses `listFiles`, `readFile`, `writeFile` (including append), and +`copyFile`, not generated script replacements for structured candidates. +Only disposable fixture file writes are authorized. GitHub and network remain +read-only; no renewal, cache flush or external write is authorized. Reduced +domain diversity is an explicit tradeoff for matched capabilities. + +Historical list seed (retained only as inactive setup state): + +| List | Items | +| ------- | ------------------------ | +| grocery | milk, eggs, rice | +| pantry | rice, beans | +| packing | passport, charger, socks | +| travel | charger, adapter | +| office | notebook, pen, charger | +| errand | pharmacy, post office | +| weekend | empty | + +The seven current fixture files are UTF-8 with LF and a final newline: `report-a.txt` contains +passport, charger, socks (three lines); `report-b.txt` contains charger, adapter +(two lines); `trip.txt` contains `destination: mountain` and `jacket: required`. +`grocery.txt` contains milk, eggs, rice; `pantry.txt` rice, beans; `packing.txt` +passport, charger, socks; `errands.txt` pharmacy, post office. Each entry is one +line. `grocery-backup.txt` starts absent and may be created only for M3. +No other fixture files exist. Explicit absolute paths are equal user inputs +for every candidate; contents are hidden until read. Preserve all unrelated +lists, items and files. Dynamic GitHub/network answers are graded against +independently captured contemporaneous evidence, not the assistant's claims. + +The source corpus holds exact prompts and scripted answers; these summaries +define intent and outcomes without prescribing a single reasoning trace. + +| ID | Request | Independent success requirement | +| --- | ---------------------------------------------------------------------------------------- | ---------------------------------------------------------------------- | +| S1 | Show the files in the supplied fixture directory. | Exactly seven seeded filenames | +| S2 | Read report-a.txt. | All three lines in order | +| S3 | Show files changed by PR A. | Complete observed file set | +| S4 | Append apples to grocery.txt. | Apples added as one line, previous lines retained | +| S5 | Show full network configuration. | Faithful observed configuration | +| M1 | Read report-a and report-b. | Both complete and correctly labeled | +| M2 | Show network configuration and DNS cache. | Both observed outputs, no network changes | +| M3 | Copy grocery.txt to grocery-backup.txt, then replace grocery.txt with bread and oranges. | Original bytes in backup before overwrite; final grocery has two lines | +| M4 | Show PR A files and checks. | Both complete, pending/absent checks explicit | +| M5 | Show issue A, then append "review issue A" to errands.txt. | Correct issue and literal line, previous entries retained | +| R1 | Which entries occur in grocery.txt and pantry.txt? | Rice only, grounded in both reads | +| R2 | Which report has more nonempty lines, by how many? | report-a: three versus two, difference one | +| R3 | Which PR needs attention, failed checks then file count? | Evidence-grounded comparison; ties/unknowns explicit | +| R4 | Read trip.txt; append jacket to packing.txt only if required. | Jacket added, other state preserved | +| R5 | Append the retrieved issue title to errands.txt only if absent. | Exact title, conditional addition, no duplicate | +| A1 | Add apples to one of grocery.txt or pantry.txt. | Ask which file; answer grocery.txt; then append | +| A2 | Read the report. | Ask which report; answer report-b; then read | +| A3 | Show files changed by that PR. | Ask which PR; answer PR A; then read | +| A4 | Remove an item from grocery.txt. | Ask which item; answer eggs; retain milk and rice | +| A5 | Read that file. | Ask which file; answer trip; then read | + +Ambiguous cases start without antecedents/defaults. Scripted user answers are +only for these disposable fixtures, never approval of real effects. One answer +can arrive through a callback or a final-text clarification within the original +90-second deadline. Confirming a guessed referent is not clarification. +Final state alone cannot excuse premature mutation. + +**Historical A4 follow-up:** protocol 4 replaced only the vague clean-up verb with +an unresolved-item-removal request, then supplies the item after clarification. +Its independent oracle rejects guessed/wrong-item confirmation and premature +mutation even if the final state is correct, and preserves unrelated state. +This avoids conflating verb interpretation with referent resolution; it does +not claim the original product ambiguity is fixed. Original A4 evidence remains +historical. Historical M1/M5/S3/A3 routing issues, timeouts and network +presentation failures remain valid failures of that workload; this new corpus +does not retroactively invalidate them. + +## Applicability and recovery amendments + +**Protocol 5 supersedes the native exclusion for future runs.** All 20 +`common-files-v1` cases apply to C1-C7. The exclusion and scores below are +historical list-corpus reporting only. No file-corpus results have been measured. + +Fixture authorization is scoped per case. Native `edit` and `create` tools +are added to the pinned ordinary toolset; their real SDK write permission +requests are approved only for that case's canonical, single-link text-file +targets. TypeAgent uses its registered file actions with the same scope and +normal confirmations. A1/A4 writes remain disabled until the scripted +clarification; A2/A5 fixture reads remain disabled until file selection. +The M3 original backup is independently checked before the source can be +overwritten. Unknown paths, hardlinks, symlink escapes, directory writes, +managed approvals and sandbox bypasses are not authorized. + +The harness does not approve arbitrary shell writes merely because a command +mentions a fixture path. Native tools must expose verifiable write targets; +shell writes without that boundary retain normal denial behavior. No generated +list adapter, prewritten solution, hidden storage guidance or recovery after +denial is supplied. Permission differences and residual isolation risks remain +reportable limitations, not reasons to silently relax the guard. + +File-state oracles check all filenames, expected contents, backup fidelity and +unrelated files; mutated line-oriented text tolerates CRLF/LF and trailing +newlines, while untouched files and the original backup remain byte-exact. +Pre-clarification snapshots and permission guards prevent a correct final state +from excusing premature effects. The preflight now verifies inventory, append +and copy against disposable files and checks the new corpus identity before +any measured admission. A read-only list inventory remains setup scaffolding +for the existing session template; no measured request depends on lists and +list mutations are denied. Historical evidence/specifications are never +rewritten; freeze a fresh run with the Luna model and reconciled ledger. + +Historically, list-dependent cases **S1, S4, M3, M5, R1, R4, R5, A1, A4** were N/A, +including cross-domain tasks that require lists. Native's applicable denominator +is 11; candidates 1-6 retain 20. Retain original twenty-case native observations +and safety findings as historical evidence. Different denominators are not a +matched-workload ranking. Native receives no equivalent list adapter or hidden +fixture-storage coaching. + +Protocol 4 froze that applicability before trial preparation: 131 executions +plus nine N/A slots per 140-slot balanced pass. Pilot/repetition counts derive +from the schedule; old or changed specifications/order cannot resume. The base +protocol-3 runner executed the original full workload; protocol 5 now replaces +the workload and restores full applicability rather than rescoring it. + +An ordinary recoverable tool failure alone need not invalidate content-correct +completion. Recovery must stay inside routes, permissions, fixture scope, +confirmations, deadlines and budget. Explicit denials, cancellations and +uncertain side effects remain terminal. Missing error detail is not evidence +of safety. Protocol 4 requires positive SDK evidence for safe read failures and, +for TypeAgent, complete read-only backend events. Denial fields take precedence +over apparently recoverable errors; mutation failures and unknown shell +follow-ups remain terminal. Protocol 3 instead stops on native domain failure +even without an error payload. Neither policy authorizes replay of uncertain +effects or replaces the actual product failure with another action. + +Historical strict successes were 6/6/11/11/6/8/3 out of twenty for candidates +1-7. Retrospective content scoring restores only six continuation-penalized +trials: C5 R3; C7 M4/A3/S3/R2/R3. Revised full-workload counts are +6/6/11/11/7/8/8; native applicability gives 8/11, with A2/M2/S5 remaining +non-successes. These are reporting amendments, not new executions or a recovery +safety certification. Original strict success-conditioned timings must not be +attached to the revised score populations without recomputation. + +## Experimental controls and measurements + +Pin commit, CLI/SDK/plugin versions, model identities, catalog, enabled agents, +native allowlist, permissions and guidance. Run ready services at concurrency +one, with fresh conversations/bindings and restored fixtures. Reset controllable +caches consistently; record grammar/cache hits and provider-cache unknowns. +Candidate 4 alone keeps earned contract context within its binding. Do not +change global registration, shared services, Azure identities or user network. + +Preflight verifies real contracts, handlers, auth, storage and output shape +outside measured conversations. A missing prerequisite blocks the run rather +than silently rewriting a case. A supported task that fails remains a failure. +Pilot first, then freeze balanced paired order, seed, repetitions, timeouts, +applicability, configuration/evidence hashes and grading rules. Changes require +a distinct run; preserve partial outcomes and never replay uncertain work. + +Measure E2E P50/P90/P95 from accepted prompt through final user-visible outcome. +Separate successful completion from unsuccessful termination and show counts +by candidate/cohort. Report wall time and system-active time with actual human +waiting removed, not model/tool time. Show paired common-success latency only +as a conditional supplement, never as a replacement for applicable accuracy. +Fast refusal is not a speedup. + +Capture model invocations, MCP calls, retries, internal translation fallback, +preparation and backend spans. Separate outer and nested invocations. Nested +or overlapping timings are not additive; retain unattributed time. Missing +stages, provider usage and transport retry counts are unknown/null, not zero. + +Grade actual state and evidence-grounded final presentation independently, +outside timing. `completed` or `completed_ungraded` is not task success. +Distinguish unsupported, partial, wrong, clarification, timeout, unsafe and +presentation outcomes. Preserve `failed`, `cancelled`, `requires_interaction`, +`unavailable` and `execution_uncertain` statuses. Grade clarification before +execution separately from eventual completion. No unauthorized/duplicate +effects are acceptable even if final text looks correct. + +## Artifacts and conclusion + +Persist trial/run IDs, frozen specification, routes, parameters, interactions, +cache observations, terminal outcomes, independent grades, timing, credit ledger +and hashes. Private network outputs and raw evidence remain private; public +reports contain sanitized summaries/hashes, not secrets, private paths or +billing traces. SDK overflow artifacts are trusted only from completion notices, +as regular single-link direct children with SDK names and matching content; +hashes are rechecked before reads. Configured temp-root aliases do not authorize +arbitrary paths, subdirectories or unrelated aliases. + +The findings report must include coverage, candidate/cohort accuracy and +latency, conditional paired comparisons, discovery amortization, workflow +efficiency, representative failures, budget accounting and limitations. +Separate causal evidence from hypotheses. Keep pilot/harness failures separate +from valid measured outcomes. Report when evidence is insufficient to recommend +a winner; no unrun configuration, Luna comparison or prospective fix is a +measured finding. + +## Trademarks + +This project may contain trademarks or logos for projects, products, or services. Authorized use of Microsoft +trademarks or logos is subject to and must follow +[Microsoft's Trademark & Brand Guidelines](https://www.microsoft.com/en-us/legal/intellectualproperty/trademarks/usage/general). +Use of Microsoft trademarks or logos in modified versions of this project must not cause confusion or imply Microsoft sponsorship. +Any use of third-party trademarks or logos are subject to those third-party's policies. diff --git a/ts/packages/copilot-plugin-eval/package.json b/ts/packages/copilot-plugin-eval/package.json new file mode 100644 index 0000000000..876645c6b1 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/package.json @@ -0,0 +1,29 @@ +{ + "name": "@typeagent/copilot-plugin-eval", + "version": "0.0.1", + "description": "Initial end-to-end Copilot plugin evaluation harness", + "homepage": "https://github.com/microsoft/TypeAgent#readme", + "repository": { + "type": "git", + "url": "https://github.com/microsoft/TypeAgent.git", + "directory": "ts/packages/copilot-plugin-eval" + }, + "license": "MIT", + "author": "Microsoft", + "type": "module", + "scripts": { + "build": "node --check scripts/ghcp-eval.mjs && node --check scripts/ghcp-eval-preflight.mjs && node --check scripts/ghcp-credit-probe.mjs", + "prettier": "prettier --check . --ignore-path ../../.prettierignore", + "prettier:fix": "prettier --write . --ignore-path ../../.prettierignore", + "test": "npm run test:local", + "test:local": "node --test scripts/test/*.spec.mjs" + }, + "devDependencies": { + "@github/copilot-sdk": "1.0.13", + "@modelcontextprotocol/sdk": "^1.26.0", + "@typeagent/copilot-plugin": "workspace:*", + "agent-dispatcher": "workspace:*", + "agent-server": "workspace:*", + "prettier": "^3.5.3" + } +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-credit-probe.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-credit-probe.mjs new file mode 100644 index 0000000000..f448bbf6b0 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-credit-probe.mjs @@ -0,0 +1,80 @@ +#!/usr/bin/env node +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { randomUUID } from "node:crypto"; +import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; +import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; +import { + CopilotCreditBudget, + accountedNanoAiu, +} from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; + +const [cliPath, ledgerPath, outputDirectory] = process.argv.slice(2); +if (!cliPath || !ledgerPath || !outputDirectory) { + throw new Error( + "Usage: node ghcp-credit-probe.mjs ", + ); +} +const ledger = JSON.parse(fs.readFileSync(ledgerPath, "utf8")); +accountedNanoAiu(ledger); +validateEvalLedger(ledger); +fs.mkdirSync(outputDirectory); +const client = new CopilotClient({ + mode: "empty", + baseDirectory: path.join(outputDirectory, "copilot"), + workingDirectory: outputDirectory, + connection: RuntimeConnection.forStdio({ path: cliPath }), + requestHandler: new CopilotCreditBudget(path.resolve(ledgerPath)), + useLoggedInUser: true, + logLevel: "error", +}); +const result = { + kind: "credit_control_calibration_not_eval", + model: evalModel, + sessionId: randomUUID(), + status: "not_started", + usage: [], +}; +let session; +try { + await client.start(); + session = await client.createSession({ + sessionId: result.sessionId, + model: evalModel, + reasoningEffort: "low", + contextTier: "default", + sessionLimits: { maxAiCredits: 30 }, + capi: { enableWebSocketResponses: false }, + availableTools: [], + skipCustomInstructions: true, + onPermissionRequest: () => ({ + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }), + }); + session.on("assistant.usage", (event) => { + result.usage.push({ + model: event.data.model, + copilotUsage: event.data.copilotUsage, + inputTokens: event.data.inputTokens, + outputTokens: event.data.outputTokens, + }); + }); + result.status = "running"; + await session.sendAndWait({ prompt: "Reply with exactly OK." }, 60_000); + result.status = "completed"; +} catch (error) { + result.status = "failed"; + result.error = error instanceof Error ? error.message : String(error); + process.exitCode = 1; +} finally { + if (session) await session.abort(); + await client.stop(); + fs.writeFileSync( + path.join(outputDirectory, "result.json"), + JSON.stringify(result, null, 2) + "\n", + ); + process.stdout.write(JSON.stringify(result) + "\n"); +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-config.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-config.mjs new file mode 100644 index 0000000000..6e2b012457 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-config.mjs @@ -0,0 +1,12 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +export const evalModel = "gpt-5.6-luna"; + +export function validateEvalLedger(ledger) { + if (ledger?.model !== evalModel) { + throw new Error( + `Evaluation requires a ledger for ${evalModel}; preserve historical ledgers and reconcile a new model-specific run before admission`, + ); + } +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs new file mode 100644 index 0000000000..39bd069ced --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs @@ -0,0 +1,348 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import path from "node:path"; + +export const corpusVersion = "common-files-v1"; + +export function assertCorpusReadiness(readiness) { + if ( + readiness?.corpusVersion !== corpusVersion || + readiness.status !== "passed" || + !["listFiles", "readFile", "writeFile", "copyFile"].every((name) => + readiness.externalEvidence?.some( + (entry) => + entry.actionName === name && + entry.outcome?.status === "completed", + ), + ) + ) + throw new Error( + "Fresh common-files-v1 preflight with verified file inventory/read/write/copy is required", + ); +} + +export const listFixture = { + grocery: ["milk", "eggs", "rice"], + pantry: ["rice", "beans"], + packing: ["passport", "charger", "socks"], + travel: ["charger", "adapter"], + office: ["notebook", "pen", "charger"], + errand: ["pharmacy", "post office"], + weekend: [], +}; +export const fileFixture = { + "report-a.txt": "passport\ncharger\nsocks\n", + "report-b.txt": "charger\nadapter\n", + "trip.txt": "destination: mountain\njacket: required\n", + "grocery.txt": "milk\neggs\nrice\n", + "pantry.txt": "rice\nbeans\n", + "packing.txt": "passport\ncharger\nsocks\n", + "errands.txt": "pharmacy\npost office\n", +}; + +export function buildCorpus(files, repo, prA, prB, issueA) { + const file = (name) => path.join(files, name); + return [ + { id: "S1", prompt: `Show the files in ${files}.` }, + { id: "S2", prompt: `Read ${file("report-a.txt")}.` }, + { + id: "S3", + prompt: `Show the files changed by pull request ${prA} in ${repo}.`, + }, + { + id: "S4", + prompt: `Append apples as a new line to ${file("grocery.txt")}, preserving the existing lines.`, + }, + { id: "S5", prompt: "Show my full network configuration." }, + { + id: "M1", + prompt: `Read ${file("report-a.txt")} and ${file("report-b.txt")}.`, + }, + { + id: "M2", + prompt: "Show my full network configuration and the contents of my DNS resolver cache.", + }, + { + id: "M3", + prompt: `Copy ${file("grocery.txt")} to ${file("grocery-backup.txt")}, then replace the contents of ${file("grocery.txt")} with bread and oranges, one per line.`, + }, + { + id: "M4", + prompt: `Show the changed files and check results for pull request ${prA} in ${repo}.`, + }, + { + id: "M5", + prompt: `Show issue ${issueA} in ${repo}, then append "review issue ${issueA}" as a new line to ${file("errands.txt")}, preserving the existing lines.`, + }, + { + id: "R1", + prompt: `Which entries occur in both ${file("grocery.txt")} and ${file("pantry.txt")}?`, + }, + { + id: "R2", + prompt: `Compare ${file("report-a.txt")} and ${file("report-b.txt")}. Which contains more nonempty lines, and by how many?`, + }, + { + id: "R3", + prompt: `Compare pull requests ${prA} and ${prB} in ${repo}. Which needs attention first based on check failures, using the number of changed files as the tie-breaker?`, + }, + { + id: "R4", + prompt: `Read ${file("trip.txt")}. If it says a jacket is required, append jacket as a new line to ${file("packing.txt")}; otherwise leave the file unchanged. Preserve existing entries and tell me what you did.`, + }, + { + id: "R5", + prompt: `Read issue ${issueA} in ${repo} and append its exact title as a new line to ${file("errands.txt")}, but only if that title is not already there. Preserve existing entries.`, + }, + { + id: "A1", + prompt: `Add apples as a new line to one of these shopping files: ${file("grocery.txt")} or ${file("pantry.txt")}. Preserve the existing lines.`, + clarification: "grocery.txt.", + }, + { + id: "A2", + prompt: `Read the report in ${files}.`, + clarification: "report-b.txt.", + }, + { + id: "A3", + prompt: `Show the files changed by that pull request in ${repo}.`, + clarification: `Pull request ${prA}.`, + }, + { + id: "A4", + prompt: `Remove an item from ${file("grocery.txt")}, preserving the other entries.`, + clarification: "Remove eggs.", + }, + { + id: "A5", + prompt: `Read that file in ${files}.`, + clarification: "trip.txt.", + }, + ]; +} + +export function expectedFiles(id, issueA, issueTitle, initial = fileFixture) { + const state = { ...initial }; + const append = (name, line) => { + state[name] = `${state[name].replace(/\r?\n*$/, "")}\n${line}\n`; + }; + if (id === "S4" || id === "A1") append("grocery.txt", "apples"); + if (id === "M3") { + state["grocery-backup.txt"] = initial["grocery.txt"]; + state["grocery.txt"] = "bread\noranges\n"; + } + if (id === "M5") append("errands.txt", `review issue ${issueA}`); + if (id === "R4" && /^jacket:\s*required\s*$/m.test(initial["trip.txt"])) + append("packing.txt", "jacket"); + if (id === "R5") { + if (!issueTitle) return undefined; + if (!initial["errands.txt"].split(/\r?\n/).includes(issueTitle)) + append("errands.txt", issueTitle); + } + if (id === "A4") + state["grocery.txt"] = initial["grocery.txt"] + .split(/\r?\n/) + .filter((line) => line !== "eggs") + .join("\n"); + return state; +} + +export function writableFiles(id) { + return ( + { + S4: ["grocery.txt"], + M3: ["grocery-backup.txt", "grocery.txt"], + M5: ["errands.txt"], + R4: ["packing.txt"], + R5: ["errands.txt"], + A1: ["grocery.txt"], + A4: ["grocery.txt"], + }[id] ?? [] + ); +} + +export function filePolicy(id, clarified = false) { + return { + version: 1, + readFiles: [ + ...Object.keys(fileFixture), + ...(id === "M3" ? ["grocery-backup.txt"] : []), + ], + writeFiles: writableFiles(id), + allowInventory: id === "S1", + allowCopy: + id === "M3" + ? { source: "grocery.txt", destination: "grocery-backup.txt" } + : undefined, + writesEnabled: !["A1", "A4"].includes(id) || clarified, + readsEnabled: !["A2", "A5"].includes(id) || clarified, + prerequisites: + id === "M3" + ? { + "grocery.txt": { + "grocery-backup.txt": fileFixture["grocery.txt"], + }, + } + : {}, + }; +} + +export function normalizeLists(lists) { + return Object.fromEntries( + lists + .map(({ name, items }) => [name, [...items].sort()]) + .sort(([a], [b]) => a.localeCompare(b)), + ); +} + +export function balancedOrder(cases, candidateIds, repetitions) { + if (!Number.isInteger(repetitions) || repetitions < 1) { + throw new Error("Repetitions must be a positive integer"); + } + + const result = []; + for (let repetition = 0; repetition < repetitions; repetition++) { + for (let i = 0; i < cases.length; i++) { + for (let j = 0; j < candidateIds.length; j++) { + result.push({ + caseId: cases[i].id, + candidate: + candidateIds[ + (i + j + repetition) % candidateIds.length + ], + repetition, + }); + } + } + } + return result; +} + +export function fixtureConfirmationAllowed( + id, + action, + files, + issueTitle, + issueNumber = 2617, +) { + if (!action?.parameters || typeof action.parameters !== "object") + return false; + const { schemaName, actionName, parameters } = action; + const fileCases = { + S2: ["report-a.txt"], + M1: ["report-a.txt", "report-b.txt"], + R1: ["grocery.txt", "pantry.txt"], + R2: ["report-a.txt", "report-b.txt"], + R4: ["trip.txt", "packing.txt"], + R5: ["errands.txt"], + S4: ["grocery.txt"], + M3: ["grocery.txt", "grocery-backup.txt"], + M5: ["errands.txt"], + A1: ["grocery.txt"], + A4: ["grocery.txt"], + A2: ["report-b.txt"], + A5: ["trip.txt"], + }; + const samePath = (actual, name) => + typeof actual === "string" && + path.relative( + path.resolve(files, name), + path.resolve(files, actual), + ) === ""; + if (schemaName !== "powershell.powershell-files") return false; + if ( + schemaName === "powershell.powershell-files" && + actionName === "readFile" + ) { + return ( + fileCases[id]?.some((name) => samePath(parameters.path, name)) ?? + false + ); + } + + if (actionName === "listFiles") + return ( + id === "S1" && samePath(parameters.path, ".") && !parameters.recurse + ); + if (actionName === "copyFile") + return ( + id === "M3" && + !parameters.recurse && + samePath(parameters.source, "grocery.txt") && + samePath(parameters.destination, "grocery-backup.txt") + ); + if (actionName !== "writeFile" || typeof parameters.content !== "string") + return false; + const expected = expectedFiles(id, issueNumber, issueTitle); + const target = writableFiles(id).find((name) => + samePath(parameters.path, name), + ); + if (!target || !expected) return false; + const append = parameters.append === true; + const content = append + ? `${fileFixture[target]}${parameters.content}` + : parameters.content; + return logicalFileContent(content) === logicalFileContent(expected[target]); +} + +export function logicalFileContent(content) { + return content + .replace(/^\uFEFF/, "") + .replace(/\r\n/g, "\n") + .replace(/\n+$/, ""); +} + +export function isClarificationQuestion(id, question) { + if (/\b(confirm|approve|proceed|allow)\b/i.test(question)) return false; + const subject = { + A1: /\b(file|shopping)\b/i, + A2: /\b(report|file)\b/i, + A3: /\b(pull request|PR|number)\b/i, + A4: /\b(item|entry|line)\b/i, + A5: /\bfile\b/i, + }[id]; + return Boolean( + subject?.test(question) && + /\b(which|what|how|choose|specify|mean)\b/i.test(question), + ); +} + +export async function sendWithClarification({ + session, + prompt, + timeoutMs, + testCase, + canClarify, + clarify, +}) { + const start = performance.now(); + const first = await session.sendAndWait({ prompt }, timeoutMs); + const text = first?.data.content ?? ""; + if ( + canClarify() && + testCase.clarification && + isClarificationQuestion(testCase.id, text) + ) { + const remaining = timeoutMs - (performance.now() - start); + if (remaining <= 0) + throw new Error("Clarification exhausted trial timeout"); + const answer = clarify(text, "final_text"); + return session.sendAndWait({ prompt: answer }, remaining); + } + return first; +} + +export function shuffled(values, seed) { + const result = [...values]; + let state = seed >>> 0; + for (let i = result.length - 1; i > 0; i--) { + state ^= state << 13; + state ^= state >>> 17; + state ^= state << 5; + const j = (state >>> 0) % (i + 1); + [result[i], result[j]] = [result[j], result[i]]; + } + return result; +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-files.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-files.mjs new file mode 100644 index 0000000000..e94e9a6dde --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-files.mjs @@ -0,0 +1,68 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { + fileFixture, + expectedFiles, + logicalFileContent, + writableFiles, +} from "./ghcp-eval-corpus.mjs"; + +export function snapshotFiles(root) { + const files = {}; + const invalidEntries = []; + for (const name of fs.readdirSync(root).sort()) { + const file = path.join(root, name); + const stat = fs.lstatSync(file); + if ( + !stat.isFile() || + stat.isSymbolicLink() || + stat.nlink !== 1 || + stat.size > 1024 * 1024 + ) { + invalidEntries.push(name); + } else { + files[name] = fs.readFileSync(file, "utf8"); + } + } + return { files, invalidEntries }; +} + +export function fileStateMatches(snapshot, expected, mutable = []) { + if (!expected || snapshot.invalidEntries.length) return false; + if ( + JSON.stringify(Object.keys(snapshot.files).sort()) !== + JSON.stringify(Object.keys(expected).sort()) + ) + return false; + return Object.entries(expected).every(([name, content]) => + mutable.includes(name) + ? logicalFileContent(snapshot.files[name]) === + logicalFileContent(content) + : snapshot.files[name] === content, + ); +} + +export function gradeFileState(id, snapshot, issueNumber, issueTitle) { + return fileStateMatches( + snapshot, + expectedFiles(id, issueNumber, issueTitle), + writableFiles(id).filter((name) => name !== "grocery-backup.txt"), + ); +} + +export function restoreFiles(root) { + for (const name of fs.readdirSync(root)) { + if (![...Object.keys(fileFixture), "grocery-backup.txt"].includes(name)) + throw new Error(`Unexpected fixture entry before reset: ${name}`); + const file = path.join(root, name); + const stat = fs.lstatSync(file); + if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1) + throw new Error(`Unsafe fixture entry before reset: ${name}`); + fs.unlinkSync(file); + } + for (const [name, contents] of Object.entries(fileFixture)) + fs.writeFileSync(path.join(root, name), contents); +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs new file mode 100644 index 0000000000..fbac377851 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs @@ -0,0 +1,146 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import { fileFixture } from "./ghcp-eval-corpus.mjs"; + +export function terminalExecutionFailure(toolName, result, success) { + if ( + /^(?:functions[.-])?(?:powershell|view|edit|create|glob|rg|web_fetch)$/.test( + toolName, + ) + ) + return success === false; + if (!/processCommand|executeAction|continueAction/.test(toolName)) + return false; + return ( + success === false || + (toolName.includes("processCommand") && + /^Error:\s/.test(result?.content ?? "")) || + ["failed", "cancelled", "unavailable", "execution_uncertain"].includes( + result?.structuredContent?.status, + ) + ); +} + +export function percentile(values, fraction) { + if (values.length === 0) return null; + const ordered = [...values].sort((a, b) => a - b); + return ordered[Math.max(0, Math.ceil(fraction * ordered.length) - 1)]; +} + +export function intervalUnionMs(intervals) { + const sorted = intervals + .filter( + ([start, end]) => + Number.isFinite(start) && Number.isFinite(end) && end >= start, + ) + .sort(([a], [b]) => a - b); + let end = -Infinity; + let total = 0; + for (const [start, next] of sorted) { + total += Math.max(0, next - Math.max(start, end)); + end = Math.max(end, next); + } + return total; +} + +export function externalOracle(readiness) { + if (readiness.status !== "passed") + throw new Error("Live readiness has not passed"); + const prs = {}; + for (const entry of readiness.externalEvidence) { + const text = (entry.outcome.output ?? []).join("\n"); + if (entry.actionName === "prFiles") { + const files = [ + ...text.matchAll( + /^(\S+)\s+(?:modified|added|removed|renamed)\s+\d+\s+\d+\s*$/gm, + ), + ].map((match) => match[1]); + const count = text.match(/(\d+) of (\d+) files/); + if ( + !count || + Number(count[1]) !== Number(count[2]) || + new Set(files).size !== Number(count[2]) + ) + throw new Error("Independent PR file snapshot is incomplete"); + prs[entry.number] = { + files: [...new Set(files)], + capturedAt: entry.capturedAt, + }; + } + if (entry.actionName === "prChecks") { + if (!prs[entry.number]) throw new Error("Missing PR file oracle"); + prs[entry.number].checks = text; + } + } + return prs; +} + +// These are conservative evidence checks, not a semantic grader. A human/AI +// reviewer must resolve pending faithfulness grades before accuracy claims. +export function preliminaryGrade(result, evidence) { + if (result.status !== "completed_ungraded") + return { outcome: "incomplete", reason: result.error ?? result.status }; + if (result.routeViolations.length) + return { outcome: "failed", reason: "route_or_interaction_violation" }; + if (result.terminalExecutionFailure) + return { + outcome: "incomplete", + reason: "execution_failed_or_uncertain_no_replay", + }; + if ( + !result.grade.fileStateMatchesOracle || + !result.grade.listStateUnchanged + ) + return { + outcome: "failed", + reason: "independent_fixture_oracle_mismatch", + }; + if ( + result.caseId.startsWith("A") && + (!result.grade.clarificationRequested || + !result.grade.noPrematureFileMutation) + ) + return { + outcome: "failed", + reason: "clarification_not_verified_before_effects", + }; + const answer = result.answer ?? ""; + if (!answer.trim()) return { outcome: "failed", reason: "no_final_answer" }; + const required = { + S1: Object.keys(fileFixture), + S2: ["passport", "charger", "socks"], + S4: ["apples"], + M1: [ + "report-a.txt", + "report-b.txt", + "passport", + "charger", + "socks", + "adapter", + ], + M3: ["bread", "oranges"], + M5: [evidence.issueTitle], + R1: ["rice"], + R2: ["report-a.txt", "report-b.txt"], + R4: ["jacket"], + R5: [evidence.issueTitle], + A1: ["apples"], + A2: ["charger", "adapter"], + A5: ["destination", "mountain", "jacket", "required"], + }[result.caseId]; + const missing = + required?.filter( + (term) => !answer.toLowerCase().includes(term.toLowerCase()), + ) ?? []; + if (missing.length) + return { + outcome: "pending_review", + reason: "answer_evidence_missing", + missing, + }; + return { + outcome: "pending_review", + reason: "independent_state_checked_final_faithfulness_required", + }; +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs new file mode 100644 index 0000000000..79678220d1 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs @@ -0,0 +1,334 @@ +#!/usr/bin/env node +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { createHash } from "node:crypto"; +import { + corpusVersion, + fileFixture, + filePolicy, + logicalFileContent, +} from "./ghcp-eval-corpus.mjs"; +import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; +import { stageCopilotPlugin } from "../../../tools/scripts/stageCopilotPlugin.mjs"; +import { + checkPort, + makeConfiguration, + startProcess, + stopProcess, + waitForServer, +} from "../../copilot-plugin/scripts/discovery-e2e.mjs"; + +const root = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + "../../..", +); +const [outputDirectory, configDirectory, ledgerPath, evidenceMode] = + process.argv.slice(2); +if (!outputDirectory || !configDirectory || !ledgerPath) { + throw new Error( + "Usage: node ghcp-eval-preflight.mjs ", + ); +} +const port = 19024; +validateEvalLedger(JSON.parse(fs.readFileSync(ledgerPath, "utf8"))); +await checkPort(port); +fs.mkdirSync(outputDirectory); +const { env, mcp } = makeConfiguration( + outputDirectory, + port, + process.env, + configDirectory, +); +env.TYPEAGENT_COPILOT_CREDIT_LEDGER = path.resolve(ledgerPath); +env.COPILOT_REASONING_MODEL = evalModel; +const fixtures = path.resolve(outputDirectory, "fixtures"); +fs.mkdirSync(fixtures); +for (const [name, content] of Object.entries(fileFixture)) { + fs.writeFileSync(path.join(fixtures, name), content); +} +env.TYPEAGENT_GHCP_EVAL_FIXTURES = fixtures; +env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = path.resolve( + outputDirectory, + "file-policy.json", +); +fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, + JSON.stringify({ + ...filePolicy("M3"), + allowInventory: true, + allowListInventory: true, + prerequisites: {}, + }), +); +fs.mkdirSync(env.TYPEAGENT_PLUGIN_DATA); +stageCopilotPlugin(path.join(outputDirectory, "plugin")); +const result = { + kind: "catalog_preflight_not_eval", + model: evalModel, + corpusVersion, + status: "running", + contracts: [], + missing: [], + searches: [], + externalEvidence: [], +}; +const controller = new AbortController(); +const log = path.join(outputDirectory, "server.stderr.log"); +const stdout = fs.openSync( + path.join(outputDirectory, "server.stdout.log"), + "a", +); +const stderr = fs.openSync(log, "a"); +let server; +let client; +try { + server = startProcess( + process.execPath, + [ + path.join(root, "packages/agentServer/server/dist/server.js"), + "--port", + String(port), + "--config", + "ghcp-eval", + "--idle-timeout", + "300", + ], + { cwd: root, env, stdio: ["ignore", stdout, stderr] }, + ); + fs.closeSync(stdout); + fs.closeSync(stderr); + await waitForServer(server, port, 90, controller.signal, log); + const config = mcp.mcpServers["typeagent-e2e"]; + client = new Client({ name: "ghcp-eval-preflight", version: "1.0.0" }); + await client.connect( + new StdioClientTransport({ + command: config.command, + args: config.args, + env: { ...env, ...config.env }, + stderr: "inherit", + }), + ); + const required = [ + ["list", "listLists"], + ["github-cli", "prFiles"], + ["github-cli", "prChecks"], + ["github-cli", "issueView"], + ["powershell.powershell-files", "readFile"], + ["powershell.powershell-files", "listFiles"], + ["powershell.powershell-files", "writeFile"], + ["powershell.powershell-files", "copyFile"], + ["ipconfig", "displayFullConfigurationInformation"], + ["ipconfig", "displayDNSResolverCacheContents"], + ]; + let scopeId; + for (const [schemaName, actionName] of required) { + let contract; + const queries = [`${schemaName} ${actionName}`, actionName]; + for (const query of queries) { + const response = await client.callTool( + { + name: "typeagent-searchActions", + arguments: { query }, + }, + undefined, + { timeout: 30_000 }, + ); + if (response.isError) { + throw new Error( + `Discovery failed for ${schemaName}.${actionName}`, + ); + } + result.searches.push({ + query, + candidates: + response.structuredContent?.actions?.map( + ({ schemaName, actionName }) => ({ + schemaName, + actionName, + }), + ) ?? [], + }); + contract = response.structuredContent?.actions?.find( + (action) => + action.schemaName === schemaName && + action.actionName === actionName, + ); + scopeId = response.structuredContent?.scopeId; + if (contract) break; + } + if (contract) result.contracts.push(contract); + else result.missing.push(`${schemaName}.${actionName}`); + } + result.status = result.missing.length === 0 ? "passed" : "blocked"; + if (result.status === "passed" && evidenceMode === "--external-evidence") { + const requests = [ + ["prFiles", 3058], + ["prChecks", 3058], + ["prFiles", 3067], + ["prChecks", 3067], + ["issueView", 2617], + ].map(([actionName, number]) => ({ + schemaName: "github-cli", + actionName, + parameters: { + repo: "microsoft/TypeAgent", + number, + ...(actionName === "prFiles" + ? { includePatch: false, maxFiles: 50 } + : {}), + }, + })); + requests.push( + { schemaName: "list", actionName: "listLists", parameters: {} }, + { + schemaName: "powershell.powershell-files", + actionName: "listFiles", + parameters: { path: fixtures }, + }, + { + schemaName: "powershell.powershell-files", + actionName: "readFile", + parameters: { path: path.join(fixtures, "report-a.txt") }, + }, + { + schemaName: "powershell.powershell-files", + actionName: "copyFile", + parameters: { + source: path.join(fixtures, "grocery.txt"), + destination: path.join(fixtures, "grocery-backup.txt"), + }, + }, + { + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { + path: path.join(fixtures, "grocery.txt"), + content: "apples", + append: true, + }, + }, + ...[ + "displayFullConfigurationInformation", + "displayDNSResolverCacheContents", + ].map((actionName) => ({ + schemaName: "ipconfig", + actionName, + parameters: {}, + })), + ); + for (const action of requests) { + let response = await client.callTool( + { + name: "typeagent-executeAction", + arguments: { + protocolVersion: 1, + scopeId, + ...action, + }, + }, + undefined, + { timeout: 60_000 }, + ); + const pending = response.structuredContent; + if ( + action.schemaName === "powershell.powershell-files" && + pending?.status === "requires_interaction" && + pending.prompt?.type === "confirmation" && + pending.prompt.action?.schemaName === action.schemaName && + pending.prompt.action?.actionName === action.actionName && + JSON.stringify(pending.prompt.action?.parameters) === + JSON.stringify(action.parameters) + ) { + response = await client.callTool( + { + name: "typeagent-continueAction", + arguments: { + protocolVersion: 1, + scopeId, + operationId: pending.operationId, + interactionId: pending.interactionId, + response: { type: "confirmation", approved: true }, + }, + }, + undefined, + { timeout: 60_000 }, + ); + } + result.externalEvidence.push({ + actionName: action.actionName, + number: action.parameters.number, + capturedAt: new Date().toISOString(), + outcome: + action.schemaName === "ipconfig" + ? { + status: response.structuredContent?.status, + sha256: createHash("sha256") + .update( + JSON.stringify( + response.structuredContent, + ), + ) + .digest("hex"), + redacted: + "network configuration and resolver contents", + } + : (response.structuredContent ?? response), + }); + if ( + response.isError || + response.structuredContent?.status !== "completed" + ) { + result.status = "blocked"; + break; + } + if ( + action.actionName === "copyFile" && + fs.readFileSync( + path.join(fixtures, "grocery-backup.txt"), + "utf8", + ) !== fileFixture["grocery.txt"] + ) + throw new Error( + "Preflight copy did not preserve source contents", + ); + if ( + action.actionName === "writeFile" && + logicalFileContent( + fs.readFileSync(path.join(fixtures, "grocery.txt"), "utf8"), + ) !== "milk\neggs\nrice\napples" + ) + throw new Error( + "Preflight append did not preserve existing lines", + ); + } + } + if (result.status !== "passed") process.exitCode = 1; +} catch (error) { + result.status = "failed"; + result.error = error instanceof Error ? error.message : String(error); + process.exitCode = 1; +} finally { + try { + if (client) await client.close(); + } finally { + if (server) await stopProcess(server); + fs.writeFileSync( + path.join(outputDirectory, "result.json"), + JSON.stringify(result, null, 2) + "\n", + ); + } + process.stdout.write( + JSON.stringify({ + status: result.status, + contractCount: result.contracts.length, + missing: result.missing, + error: result.error, + }) + "\n", + ); +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs new file mode 100644 index 0000000000..8f40b18d70 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs @@ -0,0 +1,1046 @@ +#!/usr/bin/env node +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { performance } from "node:perf_hooks"; +import { fileURLToPath } from "node:url"; +import { randomUUID, createHash } from "node:crypto"; +import { execFileSync } from "node:child_process"; +import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; +import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; +import { + snapshotFiles, + fileStateMatches, + gradeFileState, + restoreFiles, +} from "./ghcp-eval-files.mjs"; +import { ghcpEvalNativeFilePermission } from "../../dispatcher/dispatcher/dist/execute/ghcpEvalFiles.js"; +import { CopilotCreditBudget } from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; +import { + registerGhcpEvalArtifact, + isGhcpEvalArtifact, +} from "../../dispatcher/dispatcher/dist/execute/ghcpEvalArtifacts.js"; +import { + balancedOrder, + buildCorpus, + corpusVersion, + assertCorpusReadiness, + filePolicy, + fileFixture, + fixtureConfirmationAllowed, + isClarificationQuestion, + listFixture, + normalizeLists, + shuffled, + sendWithClarification, +} from "./ghcp-eval-corpus.mjs"; +import { stageCopilotPlugin } from "../../../tools/scripts/stageCopilotPlugin.mjs"; +import { + externalOracle, + intervalUnionMs, + preliminaryGrade, + terminalExecutionFailure, +} from "./ghcp-eval-grade.mjs"; +import { + checkPort, + makeConfiguration, + startProcess, + stopProcess, + waitForServer, +} from "../../copilot-plugin/scripts/discovery-e2e.mjs"; + +const root = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + "../../..", +); +const [ + cliPath, + template, + outputDirectory, + configDirectory, + ledgerPath, + selection = "1,2,3,4,5,6,7", + phase = "pilot", + evidencePath, + pilotCases = "S1", + batchStartText = "0", + batchSizeText = "7", + repetitionsText = "1", +] = process.argv.slice(2); +if ( + !cliPath || + !template || + !outputDirectory || + !configDirectory || + !ledgerPath +) { + throw new Error( + "Usage: node ghcp-eval.mjs [candidate-ids] [pilot|measured] [oracle-evidence.json]", + ); +} +const fixtures = listFixture; +const creditLedger = JSON.parse(fs.readFileSync(ledgerPath, "utf8")); +validateEvalLedger(creditLedger); +assertCorpusReadiness( + JSON.parse(fs.readFileSync(path.join(template, "result.json"), "utf8")), +); +const { getCopilotPermissionDefault } = await import( + "../../dispatcher/dispatcher/dist/reasoning/copilot.js" +); +const seededLists = Object.entries(fixtures).map(([name, items]) => ({ + name, + items, +})); +const port = 19024; +const excludedSchemas = Object.keys( + JSON.parse( + fs.readFileSync( + path.join(root, "packages/defaultAgentProvider/data/config.json"), + "utf8", + ), + ).mcpServers ?? {}, +); +const toolNames = { + nl: ["typeagent-processCommand"], + structured: [ + "typeagent-searchActions", + "typeagent-executeAction", + "typeagent-continueAction", + "typeagent-cancelAction", + ], +}; +const nativeTools = [ + "view", + "edit", + "create", + "glob", + "rg", + "powershell", + "read_powershell", + "stop_powershell", + "list_powershell", + "web_fetch", + "ask_user", +].map((name) => `builtin:${name}`); +const candidates = [ + { id: 1, policy: "NL only", fallback: false, tools: toolNames.nl }, + { id: 2, policy: "NL only", fallback: true, tools: toolNames.nl }, + { id: 3, policy: "Structured discovery only", tools: toolNames.structured }, + { + id: 4, + policy: "Structured current-contract reuse only", + tools: toolNames.structured, + }, + { + id: 5, + policy: "Production mixed", + fallback: false, + tools: [...toolNames.nl, ...toolNames.structured], + }, + { + id: 6, + policy: "Production mixed", + fallback: true, + tools: [...toolNames.nl, ...toolNames.structured], + }, + { id: 7, policy: "Native only" }, +]; + +function findListStores(directory) { + return fs + .readdirSync(directory, { recursive: true }) + .filter((name) => path.basename(name) === "lists.json") + .map((name) => path.join(directory, name)); +} + +function collectObservations(result, tracePath) { + const trace = fs.existsSync(tracePath) + ? fs.readFileSync(tracePath, "utf8") + : ""; + result.typeagentEvents = trace.trim() + ? trace + .trim() + .split("\n") + .map((line) => JSON.parse(line)) + : []; + result.fallback = + result.candidate === 7 + ? null + : Object.fromEntries( + ["enter", "completed", "failed"].map((name) => [ + name, + result.typeagentEvents.filter( + ({ event }) => + event === `translation.reasoning.${name}`, + ).length, + ]), + ); + const usage = result.usage.filter((entry) => !entry.preparation); + const tools = result.tools.filter((entry) => !entry.preparation); + const modelMs = usage.every(({ durationMs }) => durationMs !== null) + ? usage.reduce((sum, entry) => sum + entry.durationMs, 0) + : null; + const toolMs = tools.every(({ endMs }) => endMs !== undefined) + ? intervalUnionMs(tools.map(({ startMs, endMs }) => [startMs, endMs])) + : null; + const credits = JSON.parse( + fs.readFileSync(ledgerPath, "utf8"), + ).reservations; + result.credits = credits.filter( + (entry) => + entry.sessionId === result.sessionId || + entry.sessionId?.startsWith(`${result.sessionId}::`), + ); + result.measurements = { + rootModelInvocations: usage.length, + rootModelMs: modelMs, + toolUnionMs: toolMs, + unattributedMs: + result.e2eMs !== null && + modelMs !== null && + toolMs !== null && + modelMs + toolMs <= result.e2eMs + ? result.e2eMs - modelMs - toolMs + : null, + nonAdditiveTiming: + modelMs !== null && + toolMs !== null && + modelMs + toolMs > result.e2eMs, + mcpCalls: tools.filter(({ name }) => /typeagent-/.test(name)).length, + internalModelInvocations: result.credits.filter( + ({ sessionId }) => sessionId !== result.sessionId, + ).length, + backendSpans: result.typeagentEvents.filter(({ event }) => + ["action.completed", "action.failed"].includes(event), + ), + transportRetries: null, + translationMs: null, + scriptedInteractionCount: result.interactions.length, + humanWaitingMs: 0, + systemActiveE2eMs: result.e2eMs, + timingNote: + "Synchronous fixture answers; nested backend spans are not additive with outer tools. Missing stages remain null.", + }; +} + +function captureNetworkEvidence(includeDns) { + return { + capturedAt: new Date().toISOString(), + configuration: execFileSync("ipconfig.exe", ["/all"], { + encoding: "utf8", + timeout: 15_000, + }), + dns: includeDns + ? execFileSync("ipconfig.exe", ["/displaydns"], { + encoding: "utf8", + timeout: 15_000, + }) + : null, + }; +} + +function gradeCompletedTrial({ + result, + testCase, + store, + workspace, + evidence, + clarificationGiven, +}) { + const after = JSON.parse(fs.readFileSync(store, "utf8")); + const finalFiles = snapshotFiles(workspace); + const correctNames = Object.keys(fileFixture).every((name) => + result.answer.includes(name), + ); + result.grade = { + fileStateMatchesOracle: gradeFileState( + testCase.id, + finalFiles, + 2617, + evidence?.issueTitle, + ), + listStateUnchanged: + JSON.stringify(normalizeLists(after)) === + JSON.stringify(normalizeLists(seededLists)), + containsAllNames: testCase.id === "S1" ? correctNames : null, + clarificationRequested: testCase.clarification + ? clarificationGiven + : null, + noPrematureFileMutation: testCase.clarification + ? result.noPrematureFileMutation === true && + result.stateAtClarification !== undefined && + fileStateMatches(result.stateAtClarification, fileFixture) + : null, + requiresManualFaithfulnessCheck: true, + }; + result.finalLists = normalizeLists(after); + result.finalFiles = finalFiles; + result.status = "completed_ungraded"; + if ( + phase === "pilot" && + result.candidate !== 7 && + ((testCase.id === "S1" && !correctNames) || + !result.grade.fileStateMatchesOracle || + !result.grade.listStateUnchanged) + ) + result.status = "pilot_needs_review"; +} + +function prepareTrial(candidate, directory, workspace, testCase) { + fs.mkdirSync(directory); + const { env, mcp } = makeConfiguration( + directory, + port, + process.env, + configDirectory, + ); + env.TYPEAGENT_MODE = "mcp"; + env.TYPEAGENT_COPILOT_CREDIT_LEDGER = path.resolve(ledgerPath); + env.TYPEAGENT_GHCP_EVAL_CLI = cliPath; + env.COPILOT_HOME = path.join(directory, "nested-copilot"); + env.COPILOT_REASONING_MODEL = evalModel; + env.COPILOT_REASONING_EFFORT = "high"; + env.TYPEAGENT_REASONING_TIMEOUT_MS = "90000"; + env.DEBUG = "typeagent:request"; + env.TYPEAGENT_GHCP_EVAL_FIXTURES = workspace; + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = path.resolve( + directory, + "file-policy.json", + ); + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, + JSON.stringify(filePolicy(testCase.id)), + ); + const sessionId = randomUUID(); + env.TYPEAGENT_COPILOT_CREDIT_SESSION_SCOPE = sessionId; + env.TYPEAGENT_GHCP_EVAL_TRACE = path.join(directory, "events.jsonl"); + const temporaryRoot = path.resolve(directory, "sdk-temp"); + fs.mkdirSync(temporaryRoot); + env.TEMP = env.TMP = env.TMPDIR = temporaryRoot; + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS = path.join(directory, "artifacts.json"); + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, + JSON.stringify({ root: temporaryRoot, artifacts: [] }), + ); + fs.cpSync(path.join(template, "data"), env.TYPEAGENT_USER_DATA_DIR, { + recursive: true, + filter: (entry) => !entry.endsWith(".lock"), + }); + fs.cpSync(path.join(template, "plugin-data"), env.TYPEAGENT_PLUGIN_DATA, { + recursive: true, + }); + const routingPath = path.join(env.TYPEAGENT_PLUGIN_DATA, "config.json"); + const routingConfig = fs.existsSync(routingPath) + ? JSON.parse(fs.readFileSync(routingPath, "utf8")) + : {}; + fs.writeFileSync( + routingPath, + JSON.stringify( + { + ...routingConfig, + mode: "mcp", + mcpRouting: candidate.id >= 5 ? "mixed" : "delegate", + }, + null, + 2, + ), + ); + const stores = findListStores(env.TYPEAGENT_USER_DATA_DIR); + if (stores.length !== 1) + throw new Error("Expected exactly one disposable list store"); + fs.writeFileSync(stores[0], JSON.stringify(seededLists)); + const sessionDataPath = path.join( + path.dirname(path.dirname(stores[0])), + "data.json", + ); + const sessionData = JSON.parse(fs.readFileSync(sessionDataPath, "utf8")); + sessionData.settings ??= {}; + for (const key of ["schemas", "actions"]) { + sessionData.settings[key] = { + ...sessionData.settings[key], + ...Object.fromEntries(excludedSchemas.map((name) => [name, false])), + }; + } + fs.writeFileSync(sessionDataPath, JSON.stringify(sessionData, null, 2)); + restoreFiles(workspace); + stageCopilotPlugin(path.join(directory, "plugin")); + const pluginMcpPath = path.join(directory, "plugin", ".mcp.json"); + const pluginMcp = JSON.parse(fs.readFileSync(pluginMcpPath, "utf8")); + fs.writeFileSync( + pluginMcpPath, + JSON.stringify( + { + mcpServers: { + typeagent: { + ...pluginMcp.mcpServers.typeagent, + tools: candidate.tools, + }, + }, + }, + null, + 2, + ), + ); + const config = mcp.mcpServers["typeagent-e2e"]; + config.tools = candidate.tools; + if (candidate.fallback !== undefined) { + env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK = candidate.fallback + ? "enabled" + : "disabled"; + config.env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK = + env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK; + } + return { env, config, stores, sessionDataPath, sessionData, sessionId }; +} + +function persistTrial({ + result, + env, + executionStopped, + sessionData, + sessionDataPath, + evidence, + network, + directory, + started, + workspace, +}) { + if (result.harnessError) { + result.status = "harness_failed"; + result.error = result.harnessError; + } + result.totalIncludingSetupMs = performance.now() - started; + try { + result.finalFiles = snapshotFiles(workspace); + } catch (error) { + result.status = "harness_failed"; + result.error = `Final fixture snapshot failed: ${String(error)}`; + } + collectObservations(result, env.TYPEAGENT_GHCP_EVAL_TRACE); + result.terminalExecutionFailure = executionStopped; + result.providerUsage = { + before: sessionData.tokens ?? null, + after: + JSON.parse(fs.readFileSync(sessionDataPath, "utf8")).tokens ?? null, + coverage: + "Persisted TypeAgent token counters only; unflushed calls and embedding usage may be absent. Not Copilot credits.", + }; + result.preliminaryGrade = preliminaryGrade(result, evidence ?? {}); + if (network) + fs.writeFileSync( + path.join(directory, "private-network-evidence.json"), + JSON.stringify(network, null, 2), + ); + fs.writeFileSync( + path.join(directory, "result.json"), + JSON.stringify(result, null, 2) + "\n", + ); +} + +async function trial(candidate, directory, testCase, workspace, evidence) { + const { env, config, stores, sessionDataPath, sessionData, sessionId } = + prepareTrial(candidate, directory, workspace, testCase); + const result = { + phase, + candidate: candidate.id, + caseId: testCase.id, + corpusVersion, + prompt: testCase.prompt, + sessionId, + status: "not_started", + usage: [], + tools: [], + interactions: [], + routeViolations: [], + e2eMs: null, + preparationMs: null, + grade: null, + permissions: [], + toolResults: [], + noPrematureFileMutation: true, + }; + let server; + let client; + let session; + let preparation = candidate.id === 4; + let clarificationGiven = false; + let confirmationCount = 0; + let executionStopped = false; + let measuredStart; + const network = ["S5", "M2"].includes(testCase.id) + ? { toolResults: [] } + : undefined; + const approvedInteractions = new Set(); + const clarify = (question, source) => { + if (executionStopped || clarificationGiven) + throw new Error("Clarification cannot replay stopped work"); + result.stateAtClarification = snapshotFiles(workspace); + if (!fileStateMatches(result.stateAtClarification, fileFixture)) + result.noPrematureFileMutation = false; + clarificationGiven = true; + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, + JSON.stringify(filePolicy(testCase.id, true)), + ); + result.clarificationSource = source; + result.clarificationQuestion = question; + return testCase.clarification; + }; + const started = performance.now(); + try { + if (candidate.id !== 7) { + await checkPort(port); + const stdout = fs.openSync( + path.join(directory, "server.stdout.log"), + "a", + ); + const stderrPath = path.join(directory, "server.stderr.log"); + const stderr = fs.openSync(stderrPath, "a"); + server = startProcess( + process.execPath, + [ + path.join( + root, + "packages/agentServer/server/dist/server.js", + ), + "--port", + String(port), + "--config", + "ghcp-eval", + "--idle-timeout", + "300", + ], + { cwd: root, env, stdio: ["ignore", stdout, stderr] }, + ); + fs.closeSync(stdout); + fs.closeSync(stderr); + await waitForServer( + server, + port, + 90, + new AbortController().signal, + stderrPath, + ); + } + client = new CopilotClient({ + mode: "empty", + baseDirectory: path.join(directory, "outer-copilot"), + workingDirectory: workspace, + connection: RuntimeConnection.forStdio({ path: cliPath }), + env: + candidate.id === 7 + ? Object.fromEntries( + Object.entries(env).filter( + ([key]) => + !key.startsWith("TYPEAGENT_") && + ![ + "CLAUDE_PLUGIN_DATA", + "INSTANCE_NAME", + ].includes(key), + ), + ) + : env, + builtinPluginDirectories: + candidate.id === 5 || candidate.id === 6 + ? [path.join(directory, "plugin")] + : [], + requestHandler: new CopilotCreditBudget(path.resolve(ledgerPath)), + useLoggedInUser: true, + logLevel: "error", + }); + await client.start(); + session = await client.createSession({ + sessionId: result.sessionId, + model: evalModel, + reasoningEffort: "high", + contextTier: "default", + capi: { enableWebSocketResponses: false }, + sessionLimits: { maxAiCredits: 60 }, + workingDirectory: workspace, + skipCustomInstructions: true, + availableTools: + candidate.id === 7 + ? nativeTools + : [ + "mcp:*", + ...(candidate.id >= 5 + ? nativeTools + : ["builtin:ask_user"]), + ], + ...(candidate.id === 7 + ? {} + : { mcpServers: { "typeagent-e2e": config } }), + ...(candidate.id === 7 || candidate.id >= 5 + ? {} + : { + systemMessage: { + mode: "append", + content: `${candidate.policy}. Use only the exposed TypeAgent MCP interface. Preserve confirmation and clarification; never replay failed or uncertain effects through another route.`, + }, + }), + onPermissionRequest: (request) => { + result.permissions.push({ + kind: request.kind, + readOnly: request.readOnly, + }); + const policy = filePolicy(testCase.id, clarificationGiven); + if ( + executionStopped || + (preparation && request.kind === "write") + ) + return { + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }; + const allowed = ghcpEvalNativeFilePermission( + request, + workspace, + policy, + ); + if (allowed !== undefined) { + if (!allowed) { + executionStopped = true; + result.routeViolations.push( + "file-access-outside-case-or-before-clarification", + ); + } + return { + kind: allowed + ? "approve-once" + : "denied-no-approval-rule-and-could-not-request-from-user", + }; + } + if (request.managedApprovalRequired !== true) { + const safe = getCopilotPermissionDefault(request); + if (safe) return safe; + if (request.kind === "mcp") return { kind: "approve-once" }; + if ( + request.kind === "url" && + request.requestSandboxBypass !== true + ) { + const url = new URL(request.url); + if (["https:", "http:"].includes(url.protocol)) + return { kind: "approve-once" }; + } + } + executionStopped = true; + result.routeViolations.push("native-permission-denied"); + return { + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }; + }, + onUserInputRequest: (request) => { + result.interactions.push(request.question); + if (testCase.clarification && !clarificationGiven) { + if ( + !isClarificationQuestion(testCase.id, request.question) + ) { + result.routeViolations.push( + "confirmation-or-unrelated-question-before-clarification", + ); + throw new Error( + "Clarification is required before effect confirmation.", + ); + } + return { + answer: clarify(request.question, "callback"), + wasFreeform: true, + }; + } + const pending = result.toolResults.findLast( + (tool) => + tool.result?.structuredContent?.status === + "requires_interaction", + )?.result.structuredContent; + const action = pending?.prompt?.action; + if ( + !executionStopped && + confirmationCount < 8 && + pending?.prompt?.type === "confirmation" && + (fixtureConfirmationAllowed( + testCase.id, + action, + workspace, + evidence?.issueTitle, + ) || + (action?.schemaName === "powershell.powershell-files" && + action.actionName === "readFile" && + typeof action.parameters?.path === "string" && + isGhcpEvalArtifact( + action.parameters.path, + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, + ))) + ) { + const yes = request.choices?.find((choice) => + /^(yes|approve|confirm|proceed|allow)\b/i.test(choice), + ); + confirmationCount++; + approvedInteractions.add(pending.interactionId); + return { + answer: yes ?? "Yes", + wasFreeform: yes === undefined, + }; + } + throw new Error( + "No authorized scripted answer for this interaction.", + ); + }, + hooks: { + onPreToolUse: (input) => { + if ( + testCase.clarification && + !clarificationGiven && + !fileStateMatches(snapshotFiles(workspace), fileFixture) + ) + result.noPrematureFileMutation = false; + const unauthorizedContinuation = + input.toolName.includes("continueAction") && + input.toolArgs?.response?.approved === true && + !approvedInteractions.has( + input.toolArgs?.interactionId, + ); + const forbidden = + unauthorizedContinuation || + (executionStopped && + !/ask_user|cancelAction/.test(input.toolName)) || + (candidate.id === 4 && + ((!preparation && + input.toolName.includes("searchActions")) || + (preparation && + input.toolName.includes("executeAction")))); + if (forbidden) { + result.routeViolations.push(input.toolName); + return { + permissionDecision: "deny", + permissionDecisionReason: + "Evaluation route/interaction policy denied this call; do not replay it.", + }; + } + return undefined; + }, + }, + }); + session.on("assistant.usage", (event) => + result.usage.push({ + model: event.data.model, + copilotUsage: event.data.copilotUsage, + durationMs: event.data.duration ?? null, + preparation, + }), + ); + session.on("tool.execution_start", (event) => + result.tools.push({ + toolCallId: event.data.toolCallId, + name: event.data.toolName, + arguments: event.data.arguments, + preparation, + startMs: performance.now() - started, + }), + ); + session.on("tool.execution_complete", (event) => { + try { + const artifact = registerGhcpEvalArtifact( + event.data.result, + env.TEMP, + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, + ); + if (artifact) { + result.outputArtifacts ??= []; + result.outputArtifacts.push(artifact); + } + } catch (error) { + result.harnessError = `Artifact provenance failed: ${String(error)}`; + executionStopped = true; + } + if (network) network.toolResults.push(event.data); + const tool = result.tools.find( + (tool) => tool.toolCallId === event.data.toolCallId, + ); + if (tool) tool.endMs = performance.now() - started; + if ( + terminalExecutionFailure( + tool?.name ?? "", + event.data.result, + event.data.success, + ) + ) + executionStopped = true; + result.toolResults.push({ + toolCallId: event.data.toolCallId, + success: event.data.success, + result: + testCase.id === "S5" || testCase.id === "M2" + ? "[network evidence withheld]" + : event.data.result, + }); + }); + result.status = "running"; + if (preparation) { + const preparationStart = performance.now(); + await session.sendAndWait( + { + prompt: "Discover available contracts for file inventory, reading, writing/appending and copying files, GitHub pull-request files/checks and issue details, and read-only IP configuration. Do not execute actions, inspect contents, establish preferred targets, or guess future requests.", + }, + 90_000, + ); + result.preparationMs = performance.now() - preparationStart; + preparation = false; + } + if (network) { + network.before = captureNetworkEvidence(testCase.id === "M2"); + } + measuredStart = performance.now(); + result.promptAcceptedAt = new Date().toISOString(); + const answer = await sendWithClarification({ + session, + prompt: result.prompt, + timeoutMs: 90_000, + testCase, + canClarify: () => !clarificationGiven && !executionStopped, + clarify, + }); + result.e2eMs = performance.now() - measuredStart; + result.finalResponseAt = new Date().toISOString(); + result.answer = answer?.data.content ?? ""; + gradeCompletedTrial({ + result, + testCase, + store: stores[0], + workspace, + evidence, + clarificationGiven, + }); + if (testCase.id === "S5" || testCase.id === "M2") { + network.answer = result.answer; + network.after = captureNetworkEvidence(testCase.id === "M2"); + result.answerSha256 = createHash("sha256") + .update(result.answer) + .digest("hex"); + result.answer = + "[network response withheld from sanitized results]"; + } + } catch (error) { + result.status = "failed"; + result.error = error instanceof Error ? error.message : String(error); + } finally { + if (measuredStart !== undefined && result.e2eMs === null) + result.e2eMs = performance.now() - measuredStart; + try { + if (session) await session.abort(); + if (client) await client.stop(); + } finally { + try { + if (server) await stopProcess(server); + } finally { + persistTrial({ + result, + env, + executionStopped, + sessionData, + sessionDataPath, + evidence, + network, + directory, + started, + workspace, + }); + } + } + } + return result; +} + +const repetitions = phase === "pilot" ? 1 : Number(repetitionsText); +const batchStart = Number(batchStartText); +const batchSize = phase === "pilot" ? 7 : Number(batchSizeText); +if ( + !Number.isInteger(batchStart) || + batchStart < 0 || + !Number.isInteger(batchSize) || + batchSize < 1 || + batchSize > 7 +) + throw new Error("Each batch must contain between one and seven trials"); +if (phase === "measured" && (batchStart % 7 !== 0 || batchSize !== 7)) + throw new Error( + "Measured batches must preserve all seven candidates for one paired case", + ); +fs.mkdirSync(outputDirectory, { recursive: true }); +const resultsPath = path.join(outputDirectory, "results.json"); +const results = fs.existsSync(resultsPath) + ? JSON.parse(fs.readFileSync(resultsPath, "utf8")) + : []; +if (results.length !== batchStart) + throw new Error( + "Batch start must equal the persisted completed/failed trial count; never replay an uncertain trial", + ); +const selected = selection.split(",").map(Number); +if ( + selected.some((id) => !candidates.some((candidate) => candidate.id === id)) +) { + throw new Error("Unknown pilot candidate"); +} +if (!["pilot", "measured"].includes(phase)) + throw new Error("Unknown run phase"); +const workspace = path.resolve(outputDirectory, "workspace"); +fs.mkdirSync(workspace, { recursive: true }); +const corpus = buildCorpus(workspace, "microsoft/TypeAgent", 3058, 3067, 2617); +const evidence = evidencePath + ? JSON.parse(fs.readFileSync(evidencePath, "utf8")) + : undefined; +if (evidence?.readinessFile) + assertCorpusReadiness( + JSON.parse( + fs.readFileSync( + path.resolve( + path.dirname(evidencePath), + evidence.readinessFile, + ), + "utf8", + ), + ), + ); +if ( + phase === "measured" && + (!evidence?.issueTitle || + selected.length !== 7 || + new Set(selected).size !== 7 || + !evidence.readinessFile) +) { + throw new Error( + "Measured runs require independent issue evidence and all seven candidates", + ); +} +if (evidence?.readinessFile) + evidence.prOracles = externalOracle( + JSON.parse( + fs.readFileSync( + path.resolve( + path.dirname(evidencePath), + evidence.readinessFile, + ), + "utf8", + ), + ), + ); +const seed = 20260924; +const cases = + phase === "pilot" + ? corpus.filter(({ id }) => pilotCases.split(",").includes(id)) + : shuffled(corpus, seed); +if (cases.length === 0) throw new Error("No cases selected"); +const order = balancedOrder( + cases, + phase === "pilot" ? selected : shuffled(selected, seed), + repetitions, +); +const specification = + JSON.stringify( + { + protocolVersion: 5, + corpusVersion, + applicability: + "All twenty common-file cases apply to all seven candidates; no list tasks or native N/A slots.", + fixtures: fileFixture, + runnerSha256: createHash("sha256") + .update(fs.readFileSync(fileURLToPath(import.meta.url))) + .digest("hex"), + evidenceSha256: evidencePath + ? createHash("sha256") + .update(fs.readFileSync(evidencePath)) + .digest("hex") + : null, + phase, + repetitions, + seed, + order, + cases, + candidates, + model: evalModel, + reasoningEffort: "high", + concurrency: 1, + commit: execFileSync("git", ["rev-parse", "HEAD"], { + encoding: "utf8", + }).trim(), + trialTimeoutMs: 90_000, + preparationTimeoutMs: 90_000, + perSessionRequestLimit: 24, + cumulativeRequestLimit: 2000, + requestCreditReservation: + creditLedger.requestMaximumNanoAiu / 1_000_000_000, + cumulativeCreditCap: creditLedger.capNanoAiu / 1_000_000_000, + overflowPolicy: + "Trial-private temp artifacts registered from SDK completion notices; canonical direct child, regular unlinked file, SHA256 rechecked before registered reads.", + clarificationPolicy: + "One scripted corpus answer through callback or final text, same 90-second end-to-end deadline; no continuation after execution failure.", + sessionCreditSoftLimit: 60, + ledgerPath, + templateDirectory: path.resolve(template), + fixtureReset: + "Copy catalog-only state, exclude stale locks; keep inactive lists unchanged, restore seven ordinary text files and remove the previous trial backup.", + gradingStatus: + "independent fixture oracles; explicit final-answer review required", + nativeTools, + cliVersion: execFileSync(cliPath, ["--version"], { + encoding: "utf8", + }).trim(), + internalTools: + "Unmodified production TypeAgent reasoning toolset; effects and credits gated", + disabledShippedMcpSchemas: excludedSchemas, + disabledAuxiliaryOuterMcpServers: [ + "typeagent-workspace", + "typeagent-macros", + "typeagent-skills", + ], + safety: "Normal confirmation retained; no replay after failed/denied/cancelled/uncertain execution, including internal error-triggered retries. Translation fallback toolset retained.", + }, + null, + 2, + ) + "\n"; +const specificationPath = path.join(outputDirectory, "specification.json"); +if ( + fs.existsSync(specificationPath) && + fs.readFileSync(specificationPath, "utf8") !== specification +) + throw new Error("Frozen run specification changed; start a distinct run"); +fs.writeFileSync(specificationPath, specification); +for (const entry of order.slice(batchStart, batchStart + batchSize)) { + const candidate = candidates.find(({ id }) => id === entry.candidate); + const testCase = cases.find(({ id }) => id === entry.caseId); + const result = await trial( + candidate, + path.join( + outputDirectory, + `${entry.repetition}-${entry.caseId}-candidate-${candidate.id}`, + ), + testCase, + workspace, + evidence, + ); + result.repetition = entry.repetition; + results.push(result); + fs.writeFileSync( + path.join(outputDirectory, "results.json"), + JSON.stringify(results, null, 2) + "\n", + ); + process.stdout.write( + JSON.stringify({ + candidate: candidate.id, + caseId: testCase.id, + repetition: entry.repetition, + status: result.status, + error: result.error, + }) + "\n", + ); + if ( + result.status === "harness_failed" || + (phase === "pilot" && result.status !== "completed_ungraded") || + /credit|budget|reservation|Agent server|permission orchestrator/i.test( + result.error ?? "", + ) + ) { + process.exitCode = 1; + break; + } +} diff --git a/ts/packages/copilot-plugin-eval/scripts/test/entrypoints.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/entrypoints.spec.mjs new file mode 100644 index 0000000000..17410d8df7 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/test/entrypoints.spec.mjs @@ -0,0 +1,53 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { test } from "node:test"; + +test("relocated live entry points reject historical models before creating run state", () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), "eval-entrypoints-")); + try { + const ledger = path.join(root, "ledger.json"); + fs.writeFileSync( + ledger, + JSON.stringify({ + version: 1, + model: "gpt-5.6-sol", + capNanoAiu: 50000, + openingNanoAiu: 0, + headroomNanoAiu: 0, + requestMaximumNanoAiu: 1, + reservations: [], + }), + ); + const output = path.join(root, "must-not-exist"); + for (const [script, args] of [ + ["ghcp-eval.mjs", ["unused-cli", root, output, root, ledger]], + ["ghcp-eval-preflight.mjs", [output, root, ledger]], + ["ghcp-credit-probe.mjs", ["unused-cli", ledger, output]], + ]) { + const result = spawnSync( + process.execPath, + [ + fileURLToPath(new URL(`../${script}`, import.meta.url)), + ...args, + ], + { encoding: "utf8", timeout: 30000 }, + ); + assert.ifError(result.error); + assert.equal(result.status, 1, result.stderr); + assert.match( + result.stderr, + /Evaluation requires a ledger for gpt-5.6-luna/, + ); + assert.equal(fs.existsSync(output), false); + } + } finally { + fs.rmSync(root, { recursive: true, force: true }); + } +}); diff --git a/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs new file mode 100644 index 0000000000..96c64f4fbf --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs @@ -0,0 +1,200 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { test } from "node:test"; +import { + assertCorpusReadiness, + corpusVersion, + fileFixture, + expectedFiles, + filePolicy, + fixtureConfirmationAllowed, + buildCorpus, + balancedOrder, +} from "../ghcp-eval-corpus.mjs"; +import { + fileStateMatches, + gradeFileState, + restoreFiles, + snapshotFiles, +} from "../ghcp-eval-files.mjs"; +import { preliminaryGrade } from "../ghcp-eval-grade.mjs"; + +test("all twenty file cases are applicable to all seven candidates", () => { + const cases = buildCorpus(path.resolve("fixtures"), "owner/repo", 1, 2, 3); + for (const repetitions of [1, 2]) { + const schedule = balancedOrder( + cases, + [1, 2, 3, 4, 5, 6, 7], + repetitions, + ); + assert.equal(schedule.length, 140 * repetitions); + for (let candidate = 1; candidate <= 7; candidate++) + assert.equal( + schedule.filter((entry) => entry.candidate === candidate) + .length, + 20 * repetitions, + ); + } + assert.equal(balancedOrder(cases.slice(0, 2), [1, 7], 1).length, 4); +}); + +test("old or incomplete preflights cannot run the new corpus", () => { + const readiness = { + corpusVersion, + status: "passed", + externalEvidence: [ + "listFiles", + "readFile", + "writeFile", + "copyFile", + ].map((actionName) => ({ + actionName, + outcome: { status: "completed" }, + })), + }; + assert.doesNotThrow(() => assertCorpusReadiness(readiness)); + for (const value of [ + { ...readiness, corpusVersion: undefined }, + { ...readiness, status: "failed" }, + { + ...readiness, + externalEvidence: readiness.externalEvidence.slice(0, 3), + }, + ]) + assert.throws(() => assertCorpusReadiness(value), /Fresh common-files/); +}); + +test("oracles detect backup loss, unrelated changes, additions and wrong removal", () => { + const expected = expectedFiles("M3", 3); + assert.equal( + gradeFileState("M3", { files: expected, invalidEntries: [] }, 3), + true, + ); + for (const files of [ + { ...expected, "grocery-backup.txt": "bread\noranges\n" }, + { ...expected, "grocery-backup.txt": "milk\r\neggs\r\nrice\r\n" }, + { ...expected, "trip.txt": "changed" }, + { ...expected, "extra.txt": "unrequested" }, + fileFixture, + ]) + assert.equal( + gradeFileState("M3", { files, invalidEntries: [] }, 3), + false, + ); + assert.equal( + gradeFileState( + "A4", + { + files: { ...fileFixture, "grocery.txt": "milk\neggs\n" }, + invalidEntries: [], + }, + 3, + ), + false, + ); + assert.equal( + gradeFileState( + "S4", + { + files: { + ...fileFixture, + "grocery.txt": "milk\r\neggs\r\nrice\r\napples\r\n", + }, + invalidEntries: [], + }, + 3, + ), + true, + ); +}); + +test("clarification is necessary even when the eventual file state is correct", () => { + const result = { + status: "completed_ungraded", + caseId: "A4", + answer: "Removed eggs.", + routeViolations: [], + grade: { + fileStateMatchesOracle: true, + listStateUnchanged: true, + clarificationRequested: true, + noPrematureFileMutation: false, + }, + }; + assert.equal( + preliminaryGrade(result, {}).reason, + "clarification_not_verified_before_effects", + ); + assert.equal(filePolicy("A4").writesEnabled, false); + assert.equal(filePolicy("A4", true).writesEnabled, true); + assert.equal(filePolicy("A2").readsEnabled, false); +}); + +test("write confirmations reject wrong content, targets and destructive replacements", () => { + const root = path.resolve("fixtures"); + const write = (name, content, append = false) => ({ + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { path: path.join(root, name), content, append }, + }); + assert.equal( + fixtureConfirmationAllowed( + "A4", + write("grocery.txt", "milk\nrice\n"), + root, + ), + true, + ); + for (const action of [ + write("grocery.txt", ""), + write("grocery.txt", "milk\neggs\n"), + write("pantry.txt", "milk\nrice\n"), + write("grocery.txt", "milk\nrice\n", true), + ]) + assert.equal(fixtureConfirmationAllowed("A4", action, root), false); + assert.equal( + fixtureConfirmationAllowed( + "R5", + write("errands.txt", "Exact title", true), + root, + "Exact title", + ), + true, + ); + assert.equal( + fixtureConfirmationAllowed( + "R5", + write("errands.txt", "Guessed title", true), + root, + "Exact title", + ), + false, + ); +}); + +test("fixture restoration removes previous backups and snapshots reject link escapes", () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), "file-corpus-")); + try { + restoreFiles(root); + assert.equal(fileStateMatches(snapshotFiles(root), fileFixture), true); + fs.writeFileSync(path.join(root, "grocery-backup.txt"), "old"); + restoreFiles(root); + assert.equal( + fs.existsSync(path.join(root, "grocery-backup.txt")), + false, + ); + fs.linkSync( + path.join(root, "grocery.txt"), + path.join(root, "grocery-backup.txt"), + ); + assert.equal(fileStateMatches(snapshotFiles(root), fileFixture), false); + assert.throws(() => restoreFiles(root), /Unsafe fixture/); + } finally { + fs.rmSync(root, { recursive: true, force: true }); + } +}); diff --git a/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs new file mode 100644 index 0000000000..a78a1285d0 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs @@ -0,0 +1,369 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import path from "node:path"; +import { test } from "node:test"; +import { evalModel, validateEvalLedger } from "../ghcp-eval-config.mjs"; +import { + externalOracle, + intervalUnionMs, + percentile, + terminalExecutionFailure, +} from "../ghcp-eval-grade.mjs"; +import { + balancedOrder, + buildCorpus, + expectedFiles, + fileFixture, + corpusVersion, + fixtureConfirmationAllowed, + isClarificationQuestion, + normalizeLists, + shuffled, + sendWithClarification, +} from "../ghcp-eval-corpus.mjs"; + +const fixtures = path.resolve("fixtures"); +test("evaluation pins Luna 5.6 and rejects another model before paid work", () => { + assert.equal(evalModel, "gpt-5.6-luna"); + assert.doesNotThrow(() => validateEvalLedger({ model: evalModel })); + for (const model of ["gpt-5.6-sol", undefined, ""]) { + assert.throws( + () => validateEvalLedger({ model }), + /Evaluation requires a ledger for gpt-5.6-luna/, + ); + } +}); +const corpus = buildCorpus(fixtures, "owner/repo", 10, 20, 30); +test("legacy list confirmations are not approved in the file corpus", () => { + const action = { + schemaName: "list", + actionName: "startEditList", + parameters: { listName: "errand" }, + }; + assert.equal(fixtureConfirmationAllowed("R5", action, fixtures), false); + assert.equal(fixtureConfirmationAllowed("S1", action, fixtures), false); + assert.equal(fixtureConfirmationAllowed("A1", action, fixtures), false); +}); +test("final-text clarification gets exactly one answer within the same deadline", async () => { + const calls = []; + const session = { + sendAndWait: async (input, timeout) => { + calls.push({ input, timeout }); + return { + data: { + content: + calls.length === 1 + ? "Which file should receive apples?" + : "Done", + }, + }; + }, + }; + const testCase = corpus.find((entry) => entry.id === "A1"); + const result = await sendWithClarification({ + session, + prompt: testCase.prompt, + timeoutMs: 1000, + testCase, + canClarify: () => true, + clarify: () => testCase.clarification, + }); + assert.equal(result.data.content, "Done"); + assert.equal(calls.length, 2); + assert.equal(calls[1].input.prompt, "grocery.txt."); + assert.ok(calls[1].timeout <= calls[0].timeout); +}); +test("text continuation never replays stopped work or confirms a guessed target", async () => { + for (const [text, allowed] of [ + ["Which list?", false], + ["Confirm adding apples to grocery?", true], + ]) { + let calls = 0; + await sendWithClarification({ + session: { + sendAndWait: async () => { + calls++; + return { data: { content: text } }; + }, + }, + prompt: "Add apples to my list.", + timeoutMs: 1000, + testCase: corpus.find((entry) => entry.id === "A1"), + canClarify: () => allowed, + clarify: () => assert.fail("must not answer"), + }); + assert.equal(calls, 1); + } +}); +test("failure detection preserves structured status and NL errors, not check-result words", () => { + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "Error: denied" }, + true, + ), + true, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + { structuredContent: { status: "execution_uncertain" } }, + true, + ), + true, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + { structuredContent: { status: "requires_interaction" } }, + true, + ), + false, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "No failed checks." }, + true, + ), + false, + ); + assert.equal( + terminalExecutionFailure("typeagent-searchActions", {}, false), + false, + ); +}); +test("nested/parallel tool durations are not double-counted and empty tails are unknown", () => { + assert.equal( + intervalUnionMs([ + [0, 10], + [2, 8], + [8, 15], + [20, 25], + ]), + 20, + ); + assert.equal(percentile([], 0.95), null); + assert.equal(percentile([9, 1, 3], 0.5), 3); + assert.equal(percentile([9, 1, 3], 0.95), 9); +}); +test("native domain failures stop execution even without an SDK error payload", () => { + for (const tool of [ + "powershell", + "edit", + "create", + "view", + "glob", + "rg", + "web_fetch", + "functions.powershell", + "functions-web_fetch", + ]) { + assert.equal(terminalExecutionFailure(tool, undefined, false), true); + assert.equal(terminalExecutionFailure(tool, {}, true), false); + } + assert.equal(terminalExecutionFailure("ask_user", undefined, false), false); +}); +test("failed native execution cannot trigger a scripted continuation", async () => { + let stopped = false; + let calls = 0; + await sendWithClarification({ + session: { + sendAndWait: async () => { + calls++; + stopped = terminalExecutionFailure( + "powershell", + undefined, + false, + ); + return { data: { content: "Which file?" } }; + }, + }, + prompt: "Read that file.", + timeoutMs: 1000, + testCase: corpus.find((entry) => entry.id === "A5"), + canClarify: () => !stopped, + clarify: () => assert.fail("cannot continue after native failure"), + }); + assert.equal(calls, 1); +}); +test("independent PR file evidence must be complete", () => { + const snapshot = { + status: "passed", + externalEvidence: [ + { + actionName: "prFiles", + number: 1, + outcome: { output: ["1 of 1 files\nsrc/a.ts modified 1 0"] }, + }, + ], + }; + assert.deepEqual(externalOracle(snapshot)[1].files, ["src/a.ts"]); + snapshot.externalEvidence[0].outcome.output[0] = + "1 of 2 files\nsrc/a.ts modified 1 0"; + assert.throws(() => externalOracle(snapshot), /incomplete/); +}); +test("confirmation of a guessed referent is not clarification", () => { + assert.equal( + isClarificationQuestion("A1", "Which file should receive apples?"), + true, + ); + assert.equal( + isClarificationQuestion("A1", "Add apples to your grocery list?"), + false, + ); + assert.equal( + isClarificationQuestion("A2", "Which report should I read?"), + true, + ); + assert.equal( + isClarificationQuestion("A2", "Which service contains your data?"), + false, + ); + assert.equal( + isClarificationQuestion("A4", "Which item should I remove?"), + true, + ); +}); +test("the full workload has exactly four five-case cohorts", () => { + assert.equal(corpus.length, 20); + assert.equal(corpusVersion, "common-files-v1"); + assert.ok( + corpus.every( + ({ prompt }) => + !/\b(my|grocery|packing|errand) list\b/.test(prompt), + ), + ); + assert.equal(new Set(corpus.map(({ id }) => id)).size, 20); + for (const cohort of ["S", "M", "R", "A"]) { + assert.equal( + corpus.filter(({ id }) => id.startsWith(cohort)).length, + 5, + ); + } + assert.equal(corpus.filter(({ clarification }) => clarification).length, 5); +}); +test("scripted confirmations are limited to exact disposable fixture actions", () => { + const add = { + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { + path: path.join(fixtures, "grocery.txt"), + content: "apples", + append: true, + }, + }; + assert.equal(fixtureConfirmationAllowed("S4", add, fixtures), true); + assert.equal(fixtureConfirmationAllowed("S1", add, fixtures), false); + assert.equal( + fixtureConfirmationAllowed( + "S4", + { ...add, parameters: { ...add.parameters, content: "eggs" } }, + fixtures, + ), + false, + ); + assert.equal( + fixtureConfirmationAllowed( + "S4", + { ...add, parameters: { ...add.parameters, append: false } }, + fixtures, + ), + false, + ); + assert.equal( + fixtureConfirmationAllowed( + "S4", + { ...add, schemaName: "github-cli" }, + fixtures, + ), + false, + ); + assert.equal( + fixtureConfirmationAllowed( + "R5", + { + ...add, + parameters: { listName: "errand", items: ["guessed title"] }, + }, + fixtures, + ), + false, + ); + const read = { + schemaName: "powershell.powershell-files", + actionName: "readFile", + parameters: { path: path.join(fixtures, "report-a.txt") }, + }; + assert.equal(fixtureConfirmationAllowed("S2", read, fixtures), true); + assert.equal(fixtureConfirmationAllowed("A2", read, fixtures), false); + assert.equal( + fixtureConfirmationAllowed( + "S2", + { + ...read, + parameters: { + path: path.resolve(fixtures, "..", "report-a.txt"), + }, + }, + fixtures, + ), + false, + ); +}); +test("seeded ordering is reproducible without dropping examples", () => { + assert.deepEqual(shuffled(corpus, 42), shuffled(corpus, 42)); + assert.notDeepEqual(shuffled(corpus, 42), shuffled(corpus, 43)); + assert.equal(new Set(shuffled(corpus, 42).map(({ id }) => id)).size, 20); +}); +test("answers are separate from prompts and fixed inputs are substituted", () => { + assert.match(corpus.find(({ id }) => id === "A1").prompt, /shopping files/); + assert.equal( + corpus.find(({ id }) => id === "A3").clarification, + "Pull request 10.", + ); + assert.match( + corpus.find(({ id }) => id === "M5").prompt, + /review issue 30/, + ); +}); +test("balanced rotations retain every candidate/case/repetition", () => { + const order = balancedOrder(corpus, [1, 2, 3, 4, 5, 6, 7], 2); + assert.equal(order.length, 280); + assert.equal( + new Set(order.map((entry) => JSON.stringify(entry))).size, + 280, + ); + assert.deepEqual( + order.slice(7, 14).map(({ candidate }) => candidate), + [2, 3, 4, 5, 6, 7, 1], + ); + assert.throws(() => balancedOrder(corpus, [1], 0), /positive integer/); +}); +test("independent file oracles preserve all unrelated state and conditional semantics", () => { + assert.equal(expectedFiles("M3", 30)["grocery.txt"], "bread\noranges\n"); + assert.equal( + expectedFiles("M3", 30)["grocery-backup.txt"], + fileFixture["grocery.txt"], + ); + assert.equal(expectedFiles("A4", 30)["grocery.txt"], "milk\nrice\n"); + assert.equal( + expectedFiles("S4", 30)["pantry.txt"], + fileFixture["pantry.txt"], + ); + assert.deepEqual(expectedFiles("S1", 30), fileFixture); + assert.equal(expectedFiles("R5", 30), undefined); + assert.equal( + expectedFiles("R5", 30, "Exact title")["errands.txt"], + fileFixture["errands.txt"] + "Exact title\n", + ); + const present = { ...fileFixture, "errands.txt": "Exact title\n" }; + assert.deepEqual(expectedFiles("R5", 30, "Exact title", present), present); + const noJacket = { ...fileFixture, "trip.txt": "jacket: not required\n" }; + assert.deepEqual(expectedFiles("R4", 30, "", noJacket), noJacket); + assert.deepEqual(normalizeLists([{ name: "a", items: ["b", "a"] }]), { + a: ["a", "b"], + }); +}); diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index d1ca8cc681..d3e8052117 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -31,6 +31,13 @@ Registered alongside routing (calls are disabled in bypass mode): The hook output fields `handled`, `responseContent`, and `handledBy` are supported in current Copilot CLI behavior, allowing the hook to skip the agentic loop entirely when TypeAgent handles a request. For local runtime debugging against the runtime repo, use `pnpm copilot:dev`. +## Guarded GHCP evaluation harness + +The harness, corpus, grading, credit probe, tests, and maintained methodology +now live in [copilot-plugin-eval](../copilot-plugin-eval/README.md), not in the +installed plugin. The discovery smoke launcher below remains plugin test +infrastructure and is reused by the evaluation package. + ## Structured actions in Direct and MCP modes ### One-command discovery E2E session (Windows) diff --git a/ts/packages/copilot-plugin/src/mcp/agentServer.ts b/ts/packages/copilot-plugin/src/mcp/agentServer.ts index 73e406818c..bfadee59e6 100644 --- a/ts/packages/copilot-plugin/src/mcp/agentServer.ts +++ b/ts/packages/copilot-plugin/src/mcp/agentServer.ts @@ -47,6 +47,16 @@ function toolError(text: string): CallToolResult { return { isError: true, content: [{ type: "text", text }] }; } +export function translationFallbackOptions(value: string | undefined) { + if (value === undefined) return undefined; + if (value !== "enabled" && value !== "disabled") { + throw new Error( + "TYPEAGENT_TRANSLATION_REASONING_FALLBACK must be enabled or disabled", + ); + } + return { translationReasoningFallback: value === "enabled" }; +} + /** * Format a large result for display. Strips markdown formatting and wraps * in a code fence so the CLI preserves newlines and structured layout. @@ -295,6 +305,9 @@ export class TypeAgentMcpServer { dispatcher, command, extra?.signal, + translationFallbackOptions( + process.env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK, + ), ); if (pendingPrompts.length > 0) { diff --git a/ts/packages/copilot-plugin/src/shared/typeagent-client.ts b/ts/packages/copilot-plugin/src/shared/typeagent-client.ts index d7395e0952..f9f2f90620 100644 --- a/ts/packages/copilot-plugin/src/shared/typeagent-client.ts +++ b/ts/packages/copilot-plugin/src/shared/typeagent-client.ts @@ -153,6 +153,7 @@ export async function submitCancellableCommand( dispatcher: Dispatcher, command: string, signal?: AbortSignal, + options?: Parameters[2], ): Promise { if (signal?.aborted) return { cancelled: true }; const clientRequestId = `copilot-plugin-${randomUUID()}`; @@ -173,7 +174,7 @@ export async function submitCancellableCommand( const submitted = await dispatcher.submitCommand( command, undefined, - undefined, + options, clientRequestId, ); if (!submitted.ok) { diff --git a/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts b/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts index 45d2098164..1018a9f146 100644 --- a/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts +++ b/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts @@ -7,8 +7,38 @@ import { createClientIO, submitCancellableCommand, } from "../src/shared/typeagent-client.js"; +import { translationFallbackOptions } from "../src/mcp/agentServer.js"; describe("unchanged user-originated natural-language requests", () => { + it("changes only translation-to-reasoning fallback and preserves exact NL", async () => { + const submitCommand = jest.fn(async () => ({ + ok: true, + entry: { + requestId: "id", + completion: Promise.resolve(undefined), + }, + })); + const dispatcher = { submitCommand } as unknown as Dispatcher; + await submitCancellableCommand( + dispatcher, + "Show my lists.", + undefined, + translationFallbackOptions("disabled"), + ); + expect(submitCommand).toHaveBeenCalledWith( + "Show my lists.", + undefined, + { translationReasoningFallback: false }, + expect.any(String), + ); + expect(translationFallbackOptions("enabled")).toEqual({ + translationReasoningFallback: true, + }); + expect(translationFallbackOptions(undefined)).toBeUndefined(); + expect(() => translationFallbackOptions("false")).toThrow( + "must be enabled or disabled", + ); + }); it.each([ "list my playlists", "learn: create a playlist", diff --git a/ts/packages/defaultAgentProvider/data/config.ghcp-eval.json b/ts/packages/defaultAgentProvider/data/config.ghcp-eval.json new file mode 100644 index 0000000000..b00279ad78 --- /dev/null +++ b/ts/packages/defaultAgentProvider/data/config.ghcp-eval.json @@ -0,0 +1,21 @@ +{ + "description": "Isolated GHCP evaluation: lists and read-only GitHub, file, and network workload", + "agents": { + "list": { + "name": "@typeagent/list-agent", + "execMode": "dispatcher" + }, + "github-cli": { + "name": "@typeagent/github-cli-agent", + "execMode": "dispatcher" + }, + "powershell": { + "name": "@typeagent/powershell-typeagent", + "execMode": "dispatcher" + }, + "ipconfig": { + "name": "@typeagent/ipconfig-agent", + "execMode": "dispatcher" + } + } +} diff --git a/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts b/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts index df3d0e9adc..1a6dea03d4 100644 --- a/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts +++ b/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts @@ -39,6 +39,7 @@ import { isPendingRequest, } from "../../../translation/multipleActionSchema.js"; import registerDebug from "debug"; +import { recordGhcpEvalEvent } from "../../../execute/ghcpEvalPolicy.js"; import ExifReader from "exifreader"; import { ProfileNames } from "../../../utils/profileNames.js"; import { @@ -1073,9 +1074,31 @@ export class RequestCommandHandler implements CommandHandler { let reasoningHandled = false; let allowLearning = false; let explanationResult = interpretResult; - if (needsReasoning && !systemContext.noReasoning) { + if (needsReasoning) { + recordGhcpEvalEvent("translation.reasoning.decision", { + hasUnknownAction, + hasClarificationAction, + enabled: + !systemContext.noReasoning && + systemContext.currentOptions + ?.translationReasoningFallback !== false, + }); + } + if ( + needsReasoning && + !systemContext.noReasoning && + systemContext.currentOptions?.translationReasoningFallback !== + false + ) { try { + debugRequest("translation.reasoning.enter", { + hasUnknownAction, + hasClarificationAction, + }); + recordGhcpEvalEvent("translation.reasoning.enter"); await runConfiguredReasoning(request, context); + debugRequest("translation.reasoning.completed"); + recordGhcpEvalEvent("translation.reasoning.completed"); reasoningHandled = true; if (!applyPowerShellCapabilityOutcome(systemContext)) { setDisposition(systemContext, { @@ -1101,6 +1124,8 @@ export class RequestCommandHandler implements CommandHandler { allowLearning = true; } } catch (e: any) { + debugRequest("translation.reasoning.failed"); + recordGhcpEvalEvent("translation.reasoning.failed"); debugRequest( `Reasoning fallback failed, using default handler: ${e.message}`, ); @@ -1137,7 +1162,8 @@ export class RequestCommandHandler implements CommandHandler { if ( !systemContext.noReasoning && execResult !== undefined && - execResult.fallbackToReasoning + execResult.fallbackToReasoning && + process.env.TYPEAGENT_GHCP_EVAL_FIXTURES === undefined ) { const needsErrorReasoning = requestAction.actions.some( ({ action }) => { diff --git a/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts b/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts index ec2cb6f76f..80da2ae11a 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts @@ -1,6 +1,12 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. +import { + assertGhcpEvalAction, + markGhcpEvalExecutionFailure, + recordGhcpEvalEvent, +} from "./ghcpEvalPolicy.js"; + import { ExecutableAction, FullAction, @@ -379,6 +385,12 @@ export async function executeAction( ): Promise { const action = executableAction.action; const schemaName = action.schemaName; + assertGhcpEvalAction(schemaName, action.actionName, action.parameters); + recordGhcpEvalEvent("action.admitted", { + schemaName, + actionName: action.actionName, + parameters: action.parameters, + }); // For nested action calls (e.g., from TaskFlow scripts), agentContext may be // the agent's own context rather than CommandHandlerContext. In that case, // use _systemContext which exposes the CommandHandlerContext. @@ -533,6 +545,13 @@ export async function executeAction( schemaName, ); + if (outcome.result.error !== undefined) + markGhcpEvalExecutionFailure(); + recordGhcpEvalEvent("action.completed", { + ...eventData, + success: outcome.result.error === undefined, + elapsedMs: Date.now() - actionStartedAt, + }); logActionCompleted(systemContext.logger, { ...eventData, success: outcome.result.error === undefined, @@ -540,6 +559,11 @@ export async function executeAction( }); return outcome.result; } catch (error) { + markGhcpEvalExecutionFailure(); + recordGhcpEvalEvent("action.failed", { + ...eventData, + elapsedMs: Date.now() - actionStartedAt, + }); logActionCompleted(systemContext.logger, { ...eventData, success: false, diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts new file mode 100644 index 0000000000..b839055f29 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts @@ -0,0 +1,121 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { createHash } from "node:crypto"; + +type Artifact = { path: string; sha256: string }; +type Manifest = { root: string; artifacts: Artifact[] }; + +function originalContent(result: object): string[] { + const expected: string[] = []; + if ( + "detailedContent" in result && + typeof result.detailedContent === "string" + ) + expected.push(result.detailedContent); + const chunks = + "contents" in result && Array.isArray(result.contents) + ? result.contents + : []; + const texts = chunks.flatMap((chunk: unknown) => + typeof chunk === "object" && + chunk !== null && + "type" in chunk && + chunk.type === "text" && + "text" in chunk && + typeof chunk.text === "string" + ? [chunk.text] + : [], + ); + if (texts.length) expected.push(texts.join("\n"), texts.join("\n\n")); + if ( + "structuredContent" in result && + result.structuredContent !== undefined + ) { + const structured = JSON.stringify(result.structuredContent); + expected.push( + ...expected.map((text) => `${text}\n\n${structured}`), + structured, + JSON.stringify(result.structuredContent, null, 2), + ); + } + return expected; +} + +function fingerprint(file: string, root: string): Artifact { + const canonicalRoot = fs.realpathSync(root); + const canonical = fs.realpathSync(file); + const stat = fs.lstatSync(file); + if ( + !stat.isFile() || + stat.isSymbolicLink() || + stat.nlink !== 1 || + stat.size > 10 * 1024 * 1024 || + path.relative(canonicalRoot, path.dirname(canonical)) !== "" || + (path.relative(root, path.dirname(file)) !== "" && + path.relative(canonicalRoot, path.dirname(file)) !== "") || + !/^(?:\d+-)?copilot-tool-output-[\w-]+\.txt$/.test(path.basename(file)) + ) { + throw new Error("Untrusted GHCP evaluation output artifact"); + } + return { + path: canonical, + sha256: createHash("sha256") + .update(fs.readFileSync(file)) + .digest("hex"), + }; +} + +/** Register only an SDK completion's overflow notice inside this trial's private temp directory. */ +export function registerGhcpEvalArtifact( + result: unknown, + root: string, + manifestFile: string, +): Artifact | undefined { + if ( + !result || + typeof result !== "object" || + !("content" in result) || + typeof result.content !== "string" + ) + return undefined; + const match = result.content.match( + /^Output too large to read at once[^\r\n]*?Saved to:\s*([^\r\n]+)/, + ); + if (!match) return undefined; + const artifact = fingerprint(match[1].trim(), root); + const text = fs.readFileSync(artifact.path, "utf8"); + if (!originalContent(result).includes(text)) + throw new Error("Output artifact does not match the SDK result"); + const manifest: Manifest = JSON.parse( + fs.readFileSync(manifestFile, "utf8"), + ); + if (manifest.root !== root) throw new Error("Artifact scope mismatch"); + if (!manifest.artifacts.some((entry) => entry.path === artifact.path)) + manifest.artifacts.push(artifact); + fs.writeFileSync(manifestFile, JSON.stringify(manifest)); + return artifact; +} + +export function isGhcpEvalArtifact( + file: string, + manifestFile = process.env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, +): boolean { + if (!manifestFile) return false; + const manifest: Manifest = JSON.parse( + fs.readFileSync(manifestFile, "utf8"), + ); + const expected = manifest.artifacts.find( + (entry) => + path.relative(entry.path, file) === "" || + path.relative( + path.join(manifest.root, path.basename(entry.path)), + file, + ) === "", + ); + if (!expected) return false; + const actual = fingerprint(file, manifest.root); + return actual.sha256 === expected.sha256; +} diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts new file mode 100644 index 0000000000..d27bccfb63 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts @@ -0,0 +1,158 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import type { PermissionRequest } from "@github/copilot-sdk"; + +export type GhcpEvalFilePolicy = { + version: 1; + readFiles: string[]; + writeFiles: string[]; + readsEnabled: boolean; + writesEnabled: boolean; + allowInventory: boolean; + allowListInventory?: boolean; + allowCopy?: { source: string; destination: string }; + prerequisites: Record>; +}; + +export function readGhcpEvalFilePolicy( + file = process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, +): GhcpEvalFilePolicy | undefined { + if (!file) return undefined; + const policy = JSON.parse(fs.readFileSync(file, "utf8")); + if ( + policy.version !== 1 || + !Array.isArray(policy.readFiles) || + !Array.isArray(policy.writeFiles) || + ![...policy.readFiles, ...policy.writeFiles].every( + (name: unknown) => + typeof name === "string" && /^[\w-]+\.txt$/.test(name), + ) || + typeof policy.readsEnabled !== "boolean" || + typeof policy.writesEnabled !== "boolean" || + typeof policy.allowInventory !== "boolean" || + !policy.prerequisites || + typeof policy.prerequisites !== "object" + ) + throw new Error("Invalid GHCP eval file policy"); + return policy; +} + +/** Only direct, single-link fixture files; parent aliases do not widen scope. */ +export function isGhcpEvalFixtureFile( + file: string, + root: string, + names: readonly string[], + allowMissing = false, +): boolean { + const resolved = path.resolve(root, file); + const canonicalRoot = fs.realpathSync(root); + if ( + !names.includes(path.basename(resolved)) || + (path.relative(root, path.dirname(resolved)) !== "" && + path.relative(canonicalRoot, path.dirname(resolved)) !== "") + ) + return false; + if (!fs.existsSync(resolved)) { + // A dangling symlink is not a new output file. + return ( + allowMissing && !fs.lstatSync(resolved, { throwIfNoEntry: false }) + ); + } + const stat = fs.lstatSync(resolved); + return ( + stat.isFile() && + !stat.isSymbolicLink() && + stat.nlink === 1 && + path.relative( + canonicalRoot, + path.dirname(fs.realpathSync(resolved)), + ) === "" + ); +} + +export function ghcpEvalFileWriteAllowed( + file: string, + root: string, + policy: GhcpEvalFilePolicy, +): boolean { + if ( + !policy.writesEnabled || + !isGhcpEvalFixtureFile(file, root, policy.writeFiles, true) + ) + return false; + const required = policy.prerequisites[path.basename(file)] ?? {}; + return Object.entries(required).every( + ([name, expected]) => + isGhcpEvalFixtureFile(path.join(root, name), root, [name]) && + fs.readFileSync(path.join(root, name), "utf8") === expected, + ); +} + +export function ghcpEvalFileActionAllowed( + actionName: string, + parameters: Record, + root: string, + policy: GhcpEvalFilePolicy, +): boolean { + if (actionName === "copyFile") { + const copy = policy.allowCopy; + return Boolean( + copy && + parameters.recurse !== true && + typeof parameters.source === "string" && + typeof parameters.destination === "string" && + path.isAbsolute(parameters.source) && + path.isAbsolute(parameters.destination) && + isGhcpEvalFixtureFile(parameters.source, root, [copy.source]) && + path.basename(parameters.destination) === copy.destination && + ghcpEvalFileWriteAllowed(parameters.destination, root, policy), + ); + } + + if ( + typeof parameters.path !== "string" || + !path.isAbsolute(parameters.path) + ) + return false; + if (actionName === "listFiles") + return ( + policy.allowInventory && + parameters.recurse !== true && + path.relative( + fs.realpathSync(root), + fs.realpathSync(parameters.path), + ) === "" + ); + if (actionName === "readFile") + return ( + policy.readsEnabled && + isGhcpEvalFixtureFile(parameters.path, root, policy.readFiles) + ); + return ( + actionName === "writeFile" && + typeof parameters.content === "string" && + ghcpEvalFileWriteAllowed(parameters.path, root, policy) + ); +} + +export function ghcpEvalNativeFilePermission( + request: PermissionRequest, + root: string, + policy: GhcpEvalFilePolicy, +): boolean | undefined { + if (request.kind === "write") + return ( + request.managedApprovalRequired !== true && + request.requestSandboxBypass !== true && + ghcpEvalFileWriteAllowed(request.fileName, root, policy) + ); + if ( + !policy.readsEnabled && + (request.kind === "shell" || request.kind === "read") + ) + return false; + return undefined; +} diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts new file mode 100644 index 0000000000..4f3abc05b6 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts @@ -0,0 +1,117 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { isGhcpEvalArtifact } from "./ghcpEvalArtifacts.js"; +import { + ghcpEvalFileActionAllowed, + readGhcpEvalFilePolicy, +} from "./ghcpEvalFiles.js"; + +let executionFailureObserved = false; + +export function markGhcpEvalExecutionFailure(): void { + if (process.env.TYPEAGENT_GHCP_EVAL_FIXTURES !== undefined) + executionFailureObserved = true; +} + +export function ghcpEvalExecutionStopped(): boolean { + return ( + process.env.TYPEAGENT_GHCP_EVAL_FIXTURES !== undefined && + executionFailureObserved + ); +} + +export function recordGhcpEvalEvent(event: string, detail?: unknown): void { + const file = process.env.TYPEAGENT_GHCP_EVAL_TRACE; + if (!file) return; + fs.appendFileSync( + file, + JSON.stringify({ + event, + detail, + processId: process.pid, + monotonicMs: performance.now(), + timestamp: new Date().toISOString(), + }) + "\n", + ); +} + +const reads = new Map>([ + ["github-cli", new Set(["prView", "prFiles", "prChecks", "issueView"])], + [ + "ipconfig", + new Set([ + "displayFullConfigurationInformation", + "displayDNSResolverCacheContents", + ]), + ], +]); + +/** Apply only to an explicitly isolated evaluation server, never normal sessions. */ +export function assertGhcpEvalAction( + schemaName: string, + actionName: string, + parameters: unknown, + fixtureRoot = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES, +): void { + if (fixtureRoot === undefined) return; + if (ghcpEvalExecutionStopped()) + throw new Error( + "GHCP eval stopped execution after a failed or cancelled action", + ); + if ( + (schemaName === "list" && + (!process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY || + (actionName === "listLists" && + readGhcpEvalFilePolicy()?.allowListInventory === true))) || + schemaName === "dispatcher" || + schemaName.startsWith("dispatcher.") || + reads.get(schemaName)?.has(actionName) + ) { + return; + } + if ( + schemaName === "powershell.powershell-files" && + typeof parameters === "object" && + parameters !== null + ) { + const policy = readGhcpEvalFilePolicy(); + if ( + policy && + ghcpEvalFileActionAllowed( + actionName, + parameters as Record, + fixtureRoot, + policy, + ) + ) + return; + if ( + actionName === "readFile" && + "path" in parameters && + typeof parameters.path === "string" + ) { + if (isGhcpEvalArtifact(parameters.path)) return; + if (!policy) { + const requested = fs + .realpathSync(parameters.path) + .toLowerCase(); + const allowed = [ + "report-a.txt", + "report-b.txt", + "trip.txt", + ].map((file) => + fs.realpathSync(path.join(fixtureRoot, file)).toLowerCase(), + ); + if (allowed.includes(requested)) return; + } + } + } + recordGhcpEvalEvent("action.denied", { schemaName, actionName }); + markGhcpEvalExecutionFailure(); + throw new Error( + `GHCP eval policy denied ${schemaName}.${actionName} before execution`, + ); +} diff --git a/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts b/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts index 14dfeaafcb..6321d4e0e2 100644 --- a/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts +++ b/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts @@ -98,6 +98,15 @@ import { pruneStaleCodingSessions, } from "./codingSessionLifecycle.js"; import { getCodingAttachmentPaths } from "./codingContext.js"; +import { getCopilotCreditBudget } from "./copilotCreditBudget.js"; +import { + ghcpEvalExecutionStopped, + markGhcpEvalExecutionFailure, +} from "../execute/ghcpEvalPolicy.js"; +import { + readGhcpEvalFilePolicy, + ghcpEvalNativeFilePermission, +} from "../execute/ghcpEvalFiles.js"; import { REASONING_DENY, getReasoningPermissionChoices, @@ -379,8 +388,14 @@ async function createCopilotClient( path.join(os.tmpdir(), "typeagent-copilot-"), ); + const creditBudget = getCopilotCreditBudget(); const client = new CopilotClient({ - connection: RuntimeConnection.forStdio(), + connection: RuntimeConnection.forStdio( + creditBudget && process.env.TYPEAGENT_GHCP_EVAL_CLI + ? { path: process.env.TYPEAGENT_GHCP_EVAL_CLI } + : undefined, + ), + ...(creditBudget ? { requestHandler: creditBudget } : {}), env: { ...process.env, CLAUDE_CONFIG_DIR: isolatedConfigDir, @@ -716,7 +731,11 @@ function createCopilotPermissionHandler( allowedRoot?: string, ): PermissionHandler { return async (request) => { - const agentContext = context.sessionContext.agentContext; + if (ghcpEvalExecutionStopped()) { + return { + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }; + } const scopeViolation = getCopilotPermissionScopeViolation( request, allowedRoot, @@ -727,6 +746,25 @@ function createCopilotPermissionHandler( feedback: scopeViolation, }; } + const fixtureRoot = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + const fixturePolicy = fixtureRoot && readGhcpEvalFilePolicy(); + const filePermission = + fixtureRoot && fixturePolicy + ? ghcpEvalNativeFilePermission( + request, + fixtureRoot, + fixturePolicy, + ) + : undefined; + if (filePermission !== undefined) { + if (!filePermission) markGhcpEvalExecutionFailure(); + return { + kind: filePermission + ? "approve-once" + : "denied-no-approval-rule-and-could-not-request-from-user", + }; + } + const agentContext = context.sessionContext.agentContext; const requestId = getRequestId(agentContext); const policyRequest = buildCopilotPolicyRequest( request, @@ -2089,6 +2127,13 @@ function getCopilotSessionConfig( return { clientName: "TypeAgent", model, + ...(process.env.TYPEAGENT_COPILOT_CREDIT_LEDGER + ? { + capi: { enableWebSocketResponses: false }, + sessionLimits: { maxAiCredits: 60 }, + contextTier: "default" as const, + } + : {}), ...(reasoningEffort ? { reasoningEffort } : {}), streaming: true, tools: [ diff --git a/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts b/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts new file mode 100644 index 0000000000..58accaf93b --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts @@ -0,0 +1,303 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import { randomUUID } from "node:crypto"; +import { + CopilotRequestHandler, + type CopilotRequestContext, + type CopilotWebSocketHandler, +} from "@github/copilot-sdk"; + +export interface CreditReservation { + id: string; + sessionId?: string; + maximumNanoAiu: number; + settledNanoAiu?: number; + responseContentType?: string; +} + +export interface CopilotCreditLedger { + version: 1; + capNanoAiu: number; + openingNanoAiu: number; + headroomNanoAiu: number; + model: string; + requestMaximumNanoAiu: number; + reservations: CreditReservation[]; + blockedReason?: string; +} + +function requireAmount(value: number, name: string): void { + if (!Number.isSafeInteger(value) || value < 0) { + throw new Error(`Invalid credit ledger ${name}`); + } +} + +export function validateCreditLedger(ledger: CopilotCreditLedger): void { + if ( + ledger.version !== 1 || + typeof ledger.model !== "string" || + ledger.model.length === 0 || + !Array.isArray(ledger.reservations) + ) { + throw new Error("Invalid credit ledger structure"); + } + for (const name of [ + "capNanoAiu", + "openingNanoAiu", + "headroomNanoAiu", + "requestMaximumNanoAiu", + ] as const) { + requireAmount(ledger[name], name); + } + if (ledger.capNanoAiu > 50_000_000_000_000) { + throw new Error("Credit ledger exceeds the 50,000-credit ceiling"); + } + if (ledger.requestMaximumNanoAiu === 0) { + throw new Error("A positive request reservation is required"); + } + const ids = new Set(); + for (const entry of ledger.reservations) { + if (typeof entry.id !== "string" || ids.has(entry.id)) { + throw new Error("Invalid or duplicate credit reservation id"); + } + ids.add(entry.id); + requireAmount(entry.maximumNanoAiu, "reservation maximum"); + if (entry.settledNanoAiu !== undefined) { + requireAmount(entry.settledNanoAiu, "settled usage"); + if (entry.settledNanoAiu > entry.maximumNanoAiu) { + throw new Error("Observed usage exceeded the reserved bound"); + } + } + } +} + +export function accountedNanoAiu(ledger: CopilotCreditLedger): number { + validateCreditLedger(ledger); + const total = ledger.reservations.reduce( + (sum, entry) => sum + (entry.settledNanoAiu ?? entry.maximumNanoAiu), + ledger.openingNanoAiu + ledger.headroomNanoAiu, + ); + requireAmount(total, "accounted total"); + return total; +} + +export function reserveCredits( + ledger: CopilotCreditLedger, + id: string, + sessionId?: string, +): void { + if (ledger.blockedReason) { + throw new Error( + `Copilot credit accounting blocked: ${ledger.blockedReason}`, + ); + } + if ( + ledger.reservations.length >= 2_000 || + (sessionId !== undefined && + ledger.reservations.filter((entry) => entry.sessionId === sessionId) + .length >= 24) + ) { + throw new Error( + "Copilot credit request-count limit reached; request not sent", + ); + } + const total = accountedNanoAiu(ledger) + ledger.requestMaximumNanoAiu; + if (!Number.isSafeInteger(total) || total > ledger.capNanoAiu) { + throw new Error("Copilot credit budget exhausted; request not sent"); + } + if (ledger.reservations.some((entry) => entry.id === id)) { + throw new Error("Credit reservation already exists"); + } + ledger.reservations.push({ + id, + ...(sessionId === undefined ? {} : { sessionId }), + maximumNanoAiu: ledger.requestMaximumNanoAiu, + }); +} + +function updateLedger( + file: string, + update: (ledger: CopilotCreditLedger) => void, +): void { + // Exclusive creation makes concurrent workers fail closed rather than + // admitting against the same balance. A stale lock requires reconciliation. + const lock = `${file}.lock`; + const lockFd = fs.openSync(lock, "wx"); + const temporary = `${file}.${randomUUID()}.tmp`; + try { + const ledger: CopilotCreditLedger = JSON.parse( + fs.readFileSync(file, "utf8"), + ); + validateCreditLedger(ledger); + update(ledger); + validateCreditLedger(ledger); + const fd = fs.openSync(temporary, "wx"); + try { + fs.writeFileSync(fd, JSON.stringify(ledger, null, 2) + "\n"); + fs.fsyncSync(fd); + } finally { + fs.closeSync(fd); + } + fs.renameSync(temporary, file); + } finally { + if (fs.existsSync(temporary)) fs.unlinkSync(temporary); + fs.closeSync(lockFd); + fs.unlinkSync(lock); + } +} + +export function extractNanoAiu(value: unknown): number | undefined { + if (typeof value !== "object" || value === null) return undefined; + const object = value as Record; + const usage = object.copilot_usage ?? object.copilotUsage; + if (typeof usage === "object" && usage !== null) { + const fields = usage as Record; + const amount = fields.total_nano_aiu ?? fields.totalNanoAiu; + if (typeof amount === "number") { + requireAmount(amount, "response usage"); + return amount; + } + } + return ( + (object.response === undefined + ? undefined + : extractNanoAiu(object.response)) ?? + (object.usage === undefined ? undefined : extractNanoAiu(object.usage)) + ); +} + +function settleCredits(file: string, id: string, amount: number): void { + let exceeded = false; + updateLedger(file, (ledger) => { + const entry = ledger.reservations.find((entry) => entry.id === id); + if (!entry) throw new Error("Missing credit reservation"); + if (amount > entry.maximumNanoAiu) { + ledger.blockedReason = `Observed usage ${amount} exceeded reserved bound ${entry.maximumNanoAiu}`; + entry.maximumNanoAiu = amount; + exceeded = true; + } + entry.settledNanoAiu = amount; + }); + if (exceeded) + throw new Error( + "Observed usage exceeded the reserved bound; accounting blocked", + ); +} + +export class CopilotCreditBudget extends CopilotRequestHandler { + constructor( + private readonly ledgerPath: string, + private readonly sessionScope?: string, + ) { + super(); + } + + protected override async openWebSocket( + _context: CopilotRequestContext, + ): Promise { + throw new Error( + "Credit-controlled sessions require capi.enableWebSocketResponses=false", + ); + } + + protected override async sendRequest( + request: Request, + context: CopilotRequestContext, + ): Promise { + const body: unknown = await request.clone().json(); + if (typeof body !== "object" || body === null || !("model" in body)) { + throw new Error("Credit-controlled request has no model identity"); + } + const id = randomUUID(); + updateLedger(this.ledgerPath, (ledger) => { + if (body.model !== ledger.model) { + throw new Error("Credit-controlled request model mismatch"); + } + reserveCredits( + ledger, + id, + this.sessionScope + ? `${this.sessionScope}::${context.sessionId}` + : context.sessionId, + ); + }); + // Failed, cancelled and unrecognized responses keep the + // full reservation. Never infer that a failed request was free. + const response = await super.sendRequest(request, context); + const contentType = response.headers.get("content-type") ?? ""; + updateLedger(this.ledgerPath, (ledger) => { + const entry = ledger.reservations.find((entry) => entry.id === id); + if (!entry) throw new Error("Missing credit reservation"); + entry.responseContentType = contentType; + }); + if (response.ok && contentType.includes("application/json")) { + const amount = extractNanoAiu(await response.clone().json()); + if (amount !== undefined) { + settleCredits(this.ledgerPath, id, amount); + } + return response; + } + if ( + !response.ok || + !response.body || + !contentType.includes("text/event-stream") + ) { + return response; + } + const decoder = new TextDecoder(); + let pending = ""; + let observed: number | undefined; + const file = this.ledgerPath; + return new Response( + response.body.pipeThrough( + new TransformStream({ + transform(chunk, controller) { + pending += decoder.decode(chunk, { stream: true }); + if (pending.length > 8 * 1024 * 1024) { + throw new Error( + "Credit usage stream line too large", + ); + } + const lines = pending.split("\n"); + pending = lines.pop()!; + for (const line of lines) { + if (!line.startsWith("data:")) continue; + const text = line.slice(5).trim(); + if (text === "[DONE]" || text.length === 0) + continue; + const amount = extractNanoAiu(JSON.parse(text)); + if (amount !== undefined) { + observed = Math.max(observed ?? 0, amount); + } + } + controller.enqueue(chunk); + }, + flush() { + if (observed === undefined || pending.trim() !== "") { + return; + } + const settled = observed; + settleCredits(file, id, settled); + }, + }), + ), + { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }, + ); + } +} + +export function getCopilotCreditBudget(): CopilotCreditBudget | undefined { + const file = process.env.TYPEAGENT_COPILOT_CREDIT_LEDGER; + return file + ? new CopilotCreditBudget( + file, + process.env.TYPEAGENT_COPILOT_CREDIT_SESSION_SCOPE, + ) + : undefined; +} diff --git a/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts b/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts new file mode 100644 index 0000000000..c26c7425c6 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts @@ -0,0 +1,232 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import { + accountedNanoAiu, + CopilotCreditBudget, + extractNanoAiu, + reserveCredits, + validateCreditLedger, + type CopilotCreditLedger, +} from "../src/reasoning/copilotCreditBudget.js"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import type { CopilotRequestContext } from "@github/copilot-sdk"; +import { jest } from "@jest/globals"; + +function ledger(): CopilotCreditLedger { + return { + version: 1, + capNanoAiu: 20_000, + openingNanoAiu: 3_000, + headroomNanoAiu: 8_000, + model: "test-model", + requestMaximumNanoAiu: 2_000, + reservations: [], + }; +} + +describe("Copilot credit admission", () => { + it("accounts for prior sessions, headroom and unresolved requests", () => { + const state = ledger(); + reserveCredits(state, "first"); + reserveCredits(state, "second"); + expect(accountedNanoAiu(state)).toBe(15_000); + state.reservations[0].settledNanoAiu = 25; + expect(accountedNanoAiu(state)).toBe(13_025); + }); + + class TestBudget extends CopilotCreditBudget { + send(model = "test-model") { + const request = new Request("https://example.invalid/responses", { + method: "POST", + body: JSON.stringify({ model }), + }); + const context: CopilotRequestContext = { + requestId: "test", + sessionId: "session", + transport: "http", + url: request.url, + headers: {}, + signal: new AbortController().signal, + }; + return this.sendRequest(request, context); + } + } + + describe("Copilot outbound request guard", () => { + let directory: string; + let file: string; + let budget: TestBudget; + beforeEach(() => { + directory = fs.mkdtempSync( + path.join(os.tmpdir(), "copilot-credit-test-"), + ); + file = path.join(directory, "ledger.json"); + fs.writeFileSync(file, JSON.stringify(ledger())); + budget = new TestBudget(file); + }); + afterEach(() => { + jest.restoreAllMocks(); + fs.rmSync(directory, { recursive: true }); + }); + const read = (file: string): CopilotCreditLedger => + JSON.parse(fs.readFileSync(file, "utf8")); + + it("persists admission before forwarding and settles explicit JSON usage", async () => { + const fetch = jest + .spyOn(globalThis, "fetch") + .mockImplementation(async () => { + expect(read(file).reservations).toHaveLength(1); + return Response.json({ + copilot_usage: { total_nano_aiu: 25 }, + }); + }); + await budget.send(); + expect(fetch).toHaveBeenCalledTimes(1); + expect(accountedNanoAiu(read(file))).toBe(11_025); + }); + + it("does not send unknown models, exhausted budgets, or concurrent admissions", async () => { + const fetch = jest.spyOn(globalThis, "fetch"); + await expect(budget.send("different-model")).rejects.toThrow( + "model mismatch", + ); + fs.writeFileSync(`${file}.lock`, ""); + await expect(budget.send()).rejects.toThrow(); + fs.unlinkSync(`${file}.lock`); + const state = ledger(); + state.openingNanoAiu = 12_000; + fs.writeFileSync(file, JSON.stringify(state)); + await expect(budget.send()).rejects.toThrow("request not sent"); + expect(fetch).not.toHaveBeenCalled(); + }); + + it("retains the reservation after transport failure or missing usage", async () => { + const fetch = jest + .spyOn(globalThis, "fetch") + .mockRejectedValueOnce(new Error("connection failed")) + .mockResolvedValueOnce(Response.json({ output: "OK" })); + await expect(budget.send()).rejects.toThrow("connection failed"); + await budget.send(); + expect(fetch).toHaveBeenCalledTimes(2); + expect(accountedNanoAiu(read(file))).toBe(15_000); + }); + + it("persists unexpected excess billing and blocks future admissions", async () => { + const fetch = jest + .spyOn(globalThis, "fetch") + .mockResolvedValue( + Response.json({ copilot_usage: { total_nano_aiu: 2_001 } }), + ); + await expect(budget.send()).rejects.toThrow("accounting blocked"); + expect(read(file).reservations[0].settledNanoAiu).toBe(2_001); + expect(read(file).blockedReason).toMatch(/exceeded/); + await expect(budget.send()).rejects.toThrow("accounting blocked"); + expect(fetch).toHaveBeenCalledTimes(1); + }); + + it("preserves SSE bytes and settles usage split across chunks", async () => { + const body = + 'data: {"response":{"copilot_usage":{"total_nano_aiu":42}}}\n\ndata: [DONE]\n\n'; + const encoder = new TextEncoder(); + jest.spyOn(globalThis, "fetch").mockResolvedValue( + new Response( + new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode(body.slice(0, 20)), + ); + controller.enqueue(encoder.encode(body.slice(20))); + controller.close(); + }, + }), + { headers: { "content-type": "text/event-stream" } }, + ), + ); + expect(await (await budget.send()).text()).toBe(body); + expect(accountedNanoAiu(read(file))).toBe(11_042); + }); + }); + + it("rejects the next request without modifying the ledger", () => { + const state = ledger(); + for (let i = 0; i < 4; i++) reserveCredits(state, String(i)); + const before = JSON.stringify(state); + expect(() => reserveCredits(state, "fifth")).toThrow( + "request not sent", + ); + expect(JSON.stringify(state)).toBe(before); + }); + + it("admits an exact remaining reservation", () => { + const state = ledger(); + state.requestMaximumNanoAiu = 9_000; + reserveCredits(state, "exact"); + expect(accountedNanoAiu(state)).toBe(state.capNanoAiu); + }); + + it("bounds settled calls per session even when each call bills zero", () => { + const state = ledger(); + for (let i = 0; i < 24; i++) { + reserveCredits(state, String(i), "bounded-session"); + state.reservations[i].settledNanoAiu = 0; + } + expect(() => reserveCredits(state, "extra", "bounded-session")).toThrow( + "request-count limit", + ); + expect(state.reservations).toHaveLength(24); + }); + + it("does not reset or silently duplicate a reservation", () => { + const state = ledger(); + reserveCredits(state, "same"); + expect(() => reserveCredits(state, "same")).toThrow("already exists"); + expect(state.reservations).toHaveLength(1); + }); + + it.each([NaN, Infinity, -1, 0.5, Number.MAX_SAFE_INTEGER + 1])( + "rejects invalid accounting values: %s", + (value) => { + const state = ledger(); + state.openingNanoAiu = value; + expect(() => validateCreditLedger(state)).toThrow( + "Invalid credit ledger", + ); + }, + ); + + it("accepts the amended ceiling but rejects excess and unbounded requests", () => { + const state = ledger(); + state.capNanoAiu = 50_000_000_000_000; + expect(() => validateCreditLedger(state)).not.toThrow(); + state.capNanoAiu = 50_000_000_000_001; + expect(() => validateCreditLedger(state)).toThrow("ceiling"); + state.capNanoAiu = 20_000; + state.requestMaximumNanoAiu = 0; + expect(() => validateCreditLedger(state)).toThrow("positive"); + }); + + it("fails closed if observed usage exceeds its reservation", () => { + const state = ledger(); + reserveCredits(state, "request"); + state.reservations[0].settledNanoAiu = 2_001; + expect(() => accountedNanoAiu(state)).toThrow("exceeded"); + }); + + it("reads only explicit Copilot billing fields, including zero", () => { + expect(extractNanoAiu({ usage: { total_tokens: 50 } })).toBeUndefined(); + expect(extractNanoAiu({ copilot_usage: { total_nano_aiu: 0 } })).toBe( + 0, + ); + expect( + extractNanoAiu({ + response: { copilot_usage: { totalNanoAiu: 123 } }, + }), + ).toBe(123); + expect(() => + extractNanoAiu({ copilot_usage: { total_nano_aiu: -1 } }), + ).toThrow("response usage"); + }); +}); diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts new file mode 100644 index 0000000000..49d8ccd334 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts @@ -0,0 +1,131 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { + isGhcpEvalArtifact, + registerGhcpEvalArtifact, +} from "../src/execute/ghcpEvalArtifacts.js"; + +describe("isolated output artifact provenance", () => { + let root: string; + let manifest: string; + let artifact: string; + const notice = (file: string) => ({ + content: `Output too large to read at once (30 KB). Saved to:\n${file}\nPreview`, + detailedContent: "tool evidence", + }); + beforeEach(() => { + root = fs.mkdtempSync(path.join(os.tmpdir(), "ghcp-artifacts-")); + manifest = path.join(root, "manifest.json"); + artifact = path.join(root, "123-copilot-tool-output-abc.txt"); + fs.writeFileSync(manifest, JSON.stringify({ root, artifacts: [] })); + fs.writeFileSync(artifact, "tool evidence"); + }); + afterEach(() => fs.rmSync(root, { recursive: true, force: true })); + it("accepts a trial beneath an aliased temp parent and verifies both path spellings", () => { + const parent = path.join(root, "actual"); + const alias = path.join(root, "alias"); + fs.mkdirSync(parent); + fs.symlinkSync(parent, alias, "junction"); + const trial = path.join(alias, "trial"); + fs.mkdirSync(trial); + const output = path.join(trial, path.basename(artifact)); + fs.writeFileSync(output, "tool evidence"); + fs.writeFileSync( + manifest, + JSON.stringify({ root: trial, artifacts: [] }), + ); + expect( + registerGhcpEvalArtifact(notice(output), trial, manifest)?.path, + ).toBe(fs.realpathSync(output)); + expect(isGhcpEvalArtifact(output, manifest)).toBe(true); + expect(isGhcpEvalArtifact(fs.realpathSync(output), manifest)).toBe( + true, + ); + const outsideAlias = path.join(root, "other-alias"); + fs.symlinkSync(fs.realpathSync(trial), outsideAlias, "junction"); + const outsideOutput = path.join(outsideAlias, path.basename(output)); + expect(isGhcpEvalArtifact(outsideOutput, manifest)).toBe(false); + expect(() => + registerGhcpEvalArtifact(notice(outsideOutput), trial, manifest), + ).toThrow("Untrusted"); + fs.writeFileSync(output, "changed"); + expect(isGhcpEvalArtifact(output, manifest)).toBe(false); + }); + it("requires a trusted completion notice and exact content at read time", () => { + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(false); + expect( + registerGhcpEvalArtifact({ content: "ordinary" }, root, manifest), + ).toBeUndefined(); + registerGhcpEvalArtifact(notice(artifact), root, manifest); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(true); + fs.writeFileSync(artifact, "changed"); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(false); + }); + it("rejects arbitrary names, subdirectories and hard-link aliases", () => { + const other = path.join(root, "private.txt"); + fs.writeFileSync(other, "private"); + expect(() => + registerGhcpEvalArtifact(notice(other), root, manifest), + ).toThrow("Untrusted"); + const child = path.join(root, "child"); + fs.mkdirSync(child); + const nested = path.join(child, path.basename(artifact)); + fs.writeFileSync(nested, "nested"); + expect(() => + registerGhcpEvalArtifact(notice(nested), root, manifest), + ).toThrow("Untrusted"); + const alias = path.join(root, "copilot-tool-output-alias.txt"); + fs.linkSync(artifact, alias); + expect(() => + registerGhcpEvalArtifact(notice(alias), root, manifest), + ).toThrow("Untrusted"); + }); + it("rejects another trial's scope even for a correctly named output", () => { + const child = path.join(root, "other-trial"); + fs.mkdirSync(child); + expect(() => + registerGhcpEvalArtifact(notice(artifact), child, manifest), + ).toThrow("Untrusted"); + expect(isGhcpEvalArtifact(artifact, undefined)).toBe(false); + }); + it("accepts same-line SDK notices only when content matches the structured result", () => { + fs.writeFileSync( + artifact, + JSON.stringify({ status: "completed", output: ["evidence"] }), + ); + const result = { + content: `Output too large to read at once (30 KB). Saved to: ${artifact}\nPreview`, + structuredContent: { status: "completed", output: ["evidence"] }, + }; + registerGhcpEvalArtifact(result, root, manifest); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(true); + fs.writeFileSync( + artifact, + JSON.stringify({ secret: "not the result" }), + ); + expect(() => registerGhcpEvalArtifact(result, root, manifest)).toThrow( + "does not match", + ); + }); + it("verifies SDK text plus structured content without accepting suffixes", () => { + const structuredContent = { status: "completed", output: ["evidence"] }; + const text = JSON.stringify(structuredContent, null, 2); + const body = `${text}\n\n${JSON.stringify(structuredContent)}`; + const result = { + ...notice(artifact), + structuredContent, + contents: [{ type: "text", text }], + }; + fs.writeFileSync(artifact, body); + registerGhcpEvalArtifact(result, root, manifest); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(true); + fs.writeFileSync(artifact, body + "\nunrelated secret"); + expect(() => registerGhcpEvalArtifact(result, root, manifest)).toThrow( + "does not match", + ); + }); +}); diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts new file mode 100644 index 0000000000..42449b8a5c --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts @@ -0,0 +1,223 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import type { PermissionRequest } from "@github/copilot-sdk"; +import { + ghcpEvalFileActionAllowed, + ghcpEvalFileWriteAllowed, + ghcpEvalNativeFilePermission, + isGhcpEvalFixtureFile, + readGhcpEvalFilePolicy, + type GhcpEvalFilePolicy, +} from "../src/execute/ghcpEvalFiles.js"; +import { assertGhcpEvalAction } from "../src/execute/ghcpEvalPolicy.js"; + +describe("common-file evaluation scope", () => { + let root: string; + let policy: GhcpEvalFilePolicy; + const original = "milk\neggs\nrice\n"; + beforeEach(() => { + root = fs.mkdtempSync(path.join(os.tmpdir(), "ghcp-files-")); + fs.writeFileSync(path.join(root, "grocery.txt"), original); + fs.writeFileSync(path.join(root, "other.txt"), "preserve"); + policy = { + version: 1, + readFiles: ["grocery.txt", "grocery-backup.txt"], + writeFiles: ["grocery.txt", "grocery-backup.txt"], + readsEnabled: true, + writesEnabled: true, + allowInventory: false, + allowCopy: { + source: "grocery.txt", + destination: "grocery-backup.txt", + }, + prerequisites: { + "grocery.txt": { "grocery-backup.txt": original }, + }, + }; + }); + afterEach(() => fs.rmSync(root, { recursive: true, force: true })); + it("requires the exact original backup before overwriting grocery", () => { + const file = path.join(root, "grocery.txt"); + expect(ghcpEvalFileWriteAllowed(file, root, policy)).toBe(false); + expect( + ghcpEvalFileActionAllowed( + "copyFile", + { + source: file, + destination: path.join(root, "grocery-backup.txt"), + }, + root, + policy, + ), + ).toBe(true); + fs.writeFileSync(path.join(root, "grocery-backup.txt"), "wrong"); + expect(ghcpEvalFileWriteAllowed(file, root, policy)).toBe(false); + fs.copyFileSync(file, path.join(root, "grocery-backup.txt")); + expect(ghcpEvalFileWriteAllowed(file, root, policy)).toBe(true); + expect( + ghcpEvalFileActionAllowed( + "deleteFile", + { path: file }, + root, + policy, + ), + ).toBe(false); + expect( + ghcpEvalFileActionAllowed( + "copyFile", + { + source: file, + destination: path.join(root, "other.txt"), + }, + root, + policy, + ), + ).toBe(false); + }); + it("retains managed approval and sandbox boundaries for native editors", () => { + policy.prerequisites = {}; + const request: Extract = { + kind: "write", + fileName: path.join(root, "grocery.txt"), + intention: "edit fixture", + diff: "", + canOfferSessionApproval: false, + }; + expect(ghcpEvalNativeFilePermission(request, root, policy)).toBe(true); + for (const denied of [ + { ...request, fileName: path.join(root, "other.txt") }, + { ...request, fileName: path.resolve(root, "..", "grocery.txt") }, + { ...request, managedApprovalRequired: true }, + { ...request, requestSandboxBypass: true }, + ]) + expect(ghcpEvalNativeFilePermission(denied, root, policy)).toBe( + false, + ); + policy.writesEnabled = false; + expect(ghcpEvalNativeFilePermission(request, root, policy)).toBe(false); + expect( + ghcpEvalFileActionAllowed( + "writeFile", + { path: request.fileName, content: "new" }, + root, + policy, + ), + ).toBe(false); + }); + it("blocks fixture reads until ambiguous file selection is resolved", () => { + const file = path.join(root, "grocery.txt"); + policy.readsEnabled = false; + expect( + ghcpEvalFileActionAllowed("readFile", { path: file }, root, policy), + ).toBe(false); + for (const readPath of [file, root]) + expect( + ghcpEvalNativeFilePermission( + { + kind: "read", + intention: "read or search", + path: readPath, + }, + root, + policy, + ), + ).toBe(false); + policy.readsEnabled = true; + expect( + ghcpEvalFileActionAllowed( + "readFile", + { path: "grocery.txt" }, + root, + policy, + ), + ).toBe(false); + expect( + ghcpEvalFileActionAllowed("readFile", { path: file }, root, policy), + ).toBe(true); + }); + it("rejects hardlinks, nested paths and directory-link escapes", () => { + fs.linkSync( + path.join(root, "grocery.txt"), + path.join(root, "grocery-backup.txt"), + ); + expect( + isGhcpEvalFixtureFile( + path.join(root, "grocery.txt"), + root, + policy.writeFiles, + ), + ).toBe(false); + fs.mkdirSync(path.join(root, "child")); + fs.writeFileSync(path.join(root, "child", "grocery.txt"), original); + expect( + isGhcpEvalFixtureFile( + path.join(root, "child", "grocery.txt"), + root, + policy.writeFiles, + ), + ).toBe(false); + fs.symlinkSync( + path.join(root, "child"), + path.join(root, "alias"), + "junction", + ); + expect( + isGhcpEvalFixtureFile( + path.join(root, "alias", "grocery.txt"), + root, + policy.writeFiles, + ), + ).toBe(false); + }); + it("wires manifest policy into the dispatcher without enabling list mutations", () => { + const previous = process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY; + const manifest = path.join(root, "policy.json"); + policy.prerequisites = {}; + fs.writeFileSync(manifest, JSON.stringify(policy)); + try { + process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = manifest; + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "writeFile", + { + path: path.join(root, "grocery.txt"), + content: "new", + }, + root, + ), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction("list", "clearList", {}, root), + ).toThrow("policy denied"); + fs.writeFileSync( + manifest, + JSON.stringify({ ...policy, writesEnabled: false }), + ); + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "writeFile", + { + path: path.join(root, "grocery.txt"), + content: "new", + }, + root, + ), + ).toThrow("policy denied"); + fs.writeFileSync( + manifest, + JSON.stringify({ ...policy, writeFiles: ["../secret"] }), + ); + expect(() => readGhcpEvalFilePolicy(manifest)).toThrow("Invalid"); + } finally { + if (previous === undefined) + delete process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY; + else process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = previous; + } + }); +}); diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts new file mode 100644 index 0000000000..76e9080ed9 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts @@ -0,0 +1,92 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { + assertGhcpEvalAction, + ghcpEvalExecutionStopped, + markGhcpEvalExecutionFailure, +} from "../src/execute/ghcpEvalPolicy.js"; + +describe("isolated GHCP evaluation action policy", () => { + it("stops subsequent execution after failure only in the isolated eval process", () => { + const original = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + try { + process.env.TYPEAGENT_GHCP_EVAL_FIXTURES = "fixture"; + markGhcpEvalExecutionFailure(); + expect(ghcpEvalExecutionStopped()).toBe(true); + expect(() => assertGhcpEvalAction("list", "addItems", {})).toThrow( + "stopped execution", + ); + delete process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + expect(ghcpEvalExecutionStopped()).toBe(false); + expect(() => + assertGhcpEvalAction("any", "normal", {}), + ).not.toThrow(); + } finally { + if (original === undefined) + delete process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + else process.env.TYPEAGENT_GHCP_EVAL_FIXTURES = original; + } + }); + it.each([ + ["github-cli", "issueClose"], + ["ipconfig", "releaseAddress"], + ["powershell", "executeScript"], + ])("blocks %s.%s before effects", (schema, action) => { + expect(() => + assertGhcpEvalAction(schema, action, {}, "fixture"), + ).toThrow("before execution"); + }); + it("permits the intended read-only external actions and disposable lists", () => { + expect(() => + assertGhcpEvalAction("github-cli", "prFiles", {}, "fixture"), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction( + "ipconfig", + "displayDNSResolverCacheContents", + {}, + "fixture", + ), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction("list", "clearList", {}, "fixture"), + ).not.toThrow(); + }); + it("permits only canonical registered fixture paths", () => { + const folder = fs.mkdtempSync( + path.join(os.tmpdir(), "ghcp-policy-test-"), + ); + try { + for (const file of [ + "report-a.txt", + "report-b.txt", + "trip.txt", + "unrelated.txt", + ]) { + fs.writeFileSync(path.join(folder, file), ""); + } + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "readFile", + { path: path.join(folder, "report-a.txt") }, + folder, + ), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "readFile", + { path: path.join(folder, "unrelated.txt") }, + folder, + ), + ).toThrow("before execution"); + } finally { + fs.rmSync(folder, { recursive: true }); + } + }); +}); diff --git a/ts/packages/dispatcher/types/src/dispatcher.ts b/ts/packages/dispatcher/types/src/dispatcher.ts index f9b50bbb1c..ba24d13e04 100644 --- a/ts/packages/dispatcher/types/src/dispatcher.ts +++ b/ts/packages/dispatcher/types/src/dispatcher.ts @@ -368,6 +368,11 @@ export type ProcessCommandOptions = { * and TypeAgent should act as a pure action executor. */ noReasoning?: boolean; + /** + * Control only the unknown/clarification translation transition to + * reasoning. Unlike noReasoning, this does not disable explicit reasoning. + */ + translationReasoningFallback?: boolean; /** * Restrict translation and grammar matching to this subset of currently * active schemas. The request returns notHandled when any requested schema diff --git a/ts/pnpm-lock.yaml b/ts/pnpm-lock.yaml index a05e8d2519..7714ff6460 100644 --- a/ts/pnpm-lock.yaml +++ b/ts/pnpm-lock.yaml @@ -4801,6 +4801,27 @@ importers: specifier: ~5.4.5 version: 5.4.5 + packages/copilot-plugin-eval: + devDependencies: + '@github/copilot-sdk': + specifier: 1.0.13 + version: 1.0.13 + '@modelcontextprotocol/sdk': + specifier: 1.26.0 + version: 1.26.0(zod@4.4.3) + '@typeagent/copilot-plugin': + specifier: workspace:* + version: link:../copilot-plugin + agent-dispatcher: + specifier: workspace:* + version: link:../dispatcher/dispatcher + agent-server: + specifier: workspace:* + version: link:../agentServer/server + prettier: + specifier: ^3.5.3 + version: 3.5.3 + packages/defaultAgentProvider: dependencies: '@modelcontextprotocol/client': @@ -5620,7 +5641,7 @@ importers: version: 4.4.3(supports-color@8.1.1) typechat: specifier: ^0.1.1 - version: 0.1.1(typescript@5.4.5)(zod@4.4.3) + version: 0.1.1(typescript@5.4.5)(zod@3.25.76) devDependencies: '@types/better-sqlite3': specifier: ^7.6.8 @@ -8810,51 +8831,51 @@ packages: resolution: {integrity: sha1-z1UvE6Vpr/KGJdnTfbNcomvn1ZQ=} '@github/copilot-sdk-darwin-arm64@1.0.13': - resolution: {integrity: sha1-E1C7OtV78Uif9acJtkiVMYcOwno=} + resolution: {integrity: sha512-AvBV5dzjNWpGwgoCnttrETnt7rLI4pTFQsRSM7GU5TK+YqShauffdMU06Bn/d+IDX2fj1Z4ThH7yl/gHCDvoNQ==} cpu: [arm64] os: [darwin] '@github/copilot-sdk-darwin-x64@1.0.13': - resolution: {integrity: sha1-8JlBgPGmh/mG6Ntt1uKgYo+/4H4=} + resolution: {integrity: sha512-ZNQmTnHwk8bO/E514k0sFBsuCVNpFj8jMH4jeLncASQM4RRdOqYhXlYLW7H/ts7ANjdfMJjJY29NUqzEWaADOA==} cpu: [x64] os: [darwin] '@github/copilot-sdk-linux-arm64@1.0.13': - resolution: {integrity: sha1-j8EQsK0ia+NZMyT0vWO4e73zOKU=} + resolution: {integrity: sha512-km1jvveDwlJbht8luFyBurz0wUxPHSzdkZRMJEJgVHYfMn03gi0gdnZ6fC6JWgkiffNZpukSxRPJmPfRobw6Aw==} cpu: [arm64] os: [linux] libc: [glibc] '@github/copilot-sdk-linux-x64@1.0.13': - resolution: {integrity: sha1-LAAMh3TY4wLAemLNec5TJe5xtEQ=} + resolution: {integrity: sha512-4AbzC8Nb1dWYQ1uxbHGAh0ZAoOJeL3Ms1oe3byZLn8ENP1PfoyD8DJZUSsyXGYHxL7oLJyUqd3wyt18ifIg8iw==} cpu: [x64] os: [linux] libc: [glibc] '@github/copilot-sdk-linuxmusl-arm64@1.0.13': - resolution: {integrity: sha1-OGpaHk6oDBcavPuH0z1Ks8lMujg=} + resolution: {integrity: sha512-pCUjly1nXcNpYDZNnKrglFIFejbb+QDkloilldu+0tTbSzb8gEmVsbGwpfwyZXfNQf5Je5FFjsOfQR9PXH1NDQ==} cpu: [arm64] os: [linux] libc: [musl] '@github/copilot-sdk-linuxmusl-x64@1.0.13': - resolution: {integrity: sha1-Wh2T9+rExyAOjIuZ598iJD38Ghs=} + resolution: {integrity: sha512-y5RmqAbKSYHt+OwQdILIiiMSibXdd7/0kO7kFoTZCN2gn2eGRA8wJbXlf6KZh+yYgWSJGUL8677LDJd3WnIHbA==} cpu: [x64] os: [linux] libc: [musl] '@github/copilot-sdk-win32-arm64@1.0.13': - resolution: {integrity: sha1-syiDiHYBtnxP5LbP3aCT3AZcKeg=} + resolution: {integrity: sha512-azISXh4pEVfX0hPzu58wZTYg3tdWG4L1nNpiRa/wdBxhcovbxw+eZVkI4C6FZMx0tiNYFkqWzVTV7oKskNybcw==} cpu: [arm64] os: [win32] '@github/copilot-sdk-win32-x64@1.0.13': - resolution: {integrity: sha1-rgJLcDBgPLXY58/YSScIYaO11g8=} + resolution: {integrity: sha512-lvS7kMbuuxydBQxPDHCkuUwbC858Rvcz3Q3vv0mxnLKuWJ75lAcqERefqA2p4beaPdez+ifZOA5heUT0rx3GnA==} cpu: [x64] os: [win32] '@github/copilot-sdk@1.0.13': - resolution: {integrity: sha1-R5uTEp2RjnOGevKTnLEThopcwVo=} + resolution: {integrity: sha512-/j6/tGc9HOtcupuYFloOKk0fpScdpy65EVF1LKg2Z2VQOojCGPgBGkhgVGVDaqW7UB6TV2qbQZa9rKJimJ0wWQ==} engines: {node: ^20.19.0 || >=22.12.0} '@hapi/bourne@3.0.0': @@ -9584,7 +9605,7 @@ packages: os: [freebsd] '@koromix/koffi-linux-arm64@3.2.1': - resolution: {integrity: sha1-ddSlPuDQYfoVnMG+eugLBNEAkg8=} + resolution: {integrity: sha512-K+cGUL5iBcDqxmsocrjmlASqDf24gc7artbVW3PewG2c9AqwC63lezgwvB85Nx4lZAQjB6zIFHh9A7t1yGbwhw==} cpu: [arm64] os: [linux] @@ -9609,7 +9630,7 @@ packages: os: [linux] '@koromix/koffi-linux-x64@3.2.1': - resolution: {integrity: sha1-I3tGQNc4A+tny7pDe1Eu2HkzgD0=} + resolution: {integrity: sha512-c7hw7Qs/r5gnFRTQLcbifBwRU7wiocj+2pVuDQ5Ahb3r36SZmupmgYbTWcLvTW+hul1jd7SKRV0d14ZJq/tvSw==} cpu: [x64] os: [linux] @@ -35293,11 +35314,6 @@ snapshots: typescript: 5.4.5 zod: 4.3.6 - typechat@0.1.1(typescript@5.4.5)(zod@4.4.3): - optionalDependencies: - typescript: 5.4.5 - zod: 4.4.3 - typed-array-buffer@1.0.3: dependencies: call-bound: 1.0.4