From cc3b1bbfacd91d4a10a1486d409a5326ba721895 Mon Sep 17 00:00:00 2001 From: George Ng Date: Thu, 24 Sep 2026 03:06:12 -0700 Subject: [PATCH 01/14] Add credit-guarded GHCP evaluation harness and isolated workload Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin/README.md | 54 ++ ts/packages/copilot-plugin/package.json | 3 +- .../scripts/ghcp-credit-probe.mjs | 77 ++ .../scripts/ghcp-eval-corpus.mjs | 224 +++++ .../scripts/ghcp-eval-grade.mjs | 146 +++ .../scripts/ghcp-eval-preflight.mjs | 273 ++++++ .../copilot-plugin/scripts/ghcp-eval.mjs | 841 ++++++++++++++++++ .../scripts/test/ghcp-eval.spec.mjs | 209 +++++ .../copilot-plugin/src/mcp/agentServer.ts | 13 + .../src/shared/typeagent-client.ts | 3 +- .../test/naturalLanguageClient.spec.ts | 30 + .../data/config.ghcp-eval.json | 21 + .../handlers/requestCommandHandler.ts | 30 +- .../dispatcher/src/execute/actionHandlers.ts | 24 + .../dispatcher/src/execute/ghcpEvalPolicy.ts | 87 ++ .../dispatcher/src/reasoning/copilot.ts | 22 +- .../src/reasoning/copilotCreditBudget.ts | 303 +++++++ .../test/copilotCreditBudget.spec.ts | 230 +++++ .../dispatcher/test/ghcpEvalPolicy.spec.ts | 92 ++ .../dispatcher/types/src/dispatcher.ts | 5 + 20 files changed, 2682 insertions(+), 5 deletions(-) create mode 100644 ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs create mode 100644 ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs create mode 100644 ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs create mode 100644 ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs create mode 100644 ts/packages/copilot-plugin/scripts/ghcp-eval.mjs create mode 100644 ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs create mode 100644 ts/packages/defaultAgentProvider/data/config.ghcp-eval.json create mode 100644 ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts create mode 100644 ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts create mode 100644 ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts create mode 100644 ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index d1ca8cc681..daf6080de3 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -31,6 +31,60 @@ Registered alongside routing (calls are disabled in bypass mode): The hook output fields `handled`, `responseContent`, and `handledBy` are supported in current Copilot CLI behavior, allowing the hook to skip the agentic loop entirely when TypeAgent handles a request. For local runtime debugging against the runtime repo, use `pnpm copilot:dev`. +## Guarded GHCP evaluation harness + +The `scripts/ghcp-eval*.mjs` tools run actual Copilot SDK conversations for seven +MCP/native routing candidates and twenty cases (four five-case cohorts). They +are separate from the discovery smoke launcher below. Build the plugin and +agent-server dependencies first; run script tests with `pnpm test:ghcp-eval` +from this package. + +The harness requires an existing reconciled credit ledger, an authenticated +Copilot executable, and an existing TypeAgent model-configuration directory. +It never creates a fresh spending allowance or fetches credentials. The ledger +includes opening parent/implementation usage, reporting headroom, and a +catalog-verified conservative maximum reservation per model request. The +request proxy admits before forwarding, settles explicit Copilot billing +fields, retains unknown charges, rejects other models/WebSockets, and limits +each scoped session to 24 requests and the cumulative ledger to 2,000 requests. +The SDK's 60-credit session limit is only an additional **soft** limit. +This is not a general-purpose pricing service: verify the model's current +credit rates and maximum context/output bounds before constructing a ledger. + +From `ts`, invoke: + +```text +node packages\copilot-plugin\scripts\ghcp-eval-preflight.mjs --external-evidence +node packages\copilot-plugin\scripts\ghcp-eval.mjs pilot +node packages\copilot-plugin\scripts\ghcp-eval.mjs 1,2,3,4,5,6,7 measured S1 7 1 +``` + +Use a nonsynchronized local directory for live databases/locks. Preserve +sanitized results, the frozen specification, and the cumulative ledger in +durable storage. Measured batches contain one paired case across all seven +candidates; advance the zero-based start by seven only after reconciliation. +One complete balanced pass contains 140 trials; freeze additional repetitions +only when the reconciled allowance permits them. A changed specification or a mismatched +persisted trial count blocks resumption instead of replaying uncertain work. +The oracle JSON pins public issue/PR evidence and a relative `readinessFile` +pointing to the successful preflight result. + +The four domain agents are lists, GitHub CLI, registered PowerShell file +actions, and IP configuration. Fixtures are restored per trial. Shipped MCP +domain schemas outside that scope are disabled only in the disposable +session. The production internal reasoning toolset is unchanged. The +`translationReasoningFallback` request option controls only the existing +unknown/clarification translation-to-reasoning transition, not ordinary +orchestration. Optional `TYPEAGENT_GHCP_EVAL_TRACE` records its actual decision, +entry, and outcome. No global configuration or shared service is modified. + +**`completed_ungraded` is not task success.** Final-answer faithfulness must +be reviewed against the independent fixtures/external evidence after timing. +Network answers and raw evidence remain in clearly named private local files; +sanitized results contain hashes. Unobserved internal stage durations are +null, not zero. Keep pilot/harness failures separate from measured outcomes, +and do not pool fast refusals with successful-completion latency. + ## Structured actions in Direct and MCP modes ### One-command discovery E2E session (Windows) diff --git a/ts/packages/copilot-plugin/package.json b/ts/packages/copilot-plugin/package.json index b46caaf2de..f7097784d2 100644 --- a/ts/packages/copilot-plugin/package.json +++ b/ts/packages/copilot-plugin/package.json @@ -24,7 +24,8 @@ "test": "npm run test:local", "test:direct": "node -e \"console.log(JSON.stringify({sessionId:'test',timestamp:1234,cwd:'.',prompt:'list the playlists'}))\" | cross-env TYPEAGENT_MODE=direct node dist/hooks/hook-router.js", "test:e2e-launcher": "node --test scripts/test/discovery-e2e.spec.mjs", - "test:local": "pnpm run jest-esm --testPathPattern=\".*[.]spec[.]js\" && npm run test:e2e-launcher", + "test:ghcp-eval": "node --test scripts/test/ghcp-eval.spec.mjs", + "test:local": "pnpm run jest-esm --testPathPattern=\".*[.]spec[.]js\" && npm run test:e2e-launcher && npm run test:ghcp-eval", "test:mcp-redirect": "node -e \"console.log(JSON.stringify({sessionId:'test',timestamp:1234,cwd:'.',prompt:'list the playlists'}))\" | cross-env TYPEAGENT_MODE=mcp node dist/hooks/hook-router.js", "tsc": "tsc -b", "uninstall:global": "copilot plugin uninstall typeagent && copilot plugin marketplace remove typeagent-local", diff --git a/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs b/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs new file mode 100644 index 0000000000..82cbe97e54 --- /dev/null +++ b/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs @@ -0,0 +1,77 @@ +#!/usr/bin/env node +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { randomUUID } from "node:crypto"; +import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; +import { + CopilotCreditBudget, + accountedNanoAiu, +} from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; + +const [cliPath, ledgerPath, outputDirectory] = process.argv.slice(2); +if (!cliPath || !ledgerPath || !outputDirectory) { + throw new Error( + "Usage: node ghcp-credit-probe.mjs ", + ); +} +const ledger = JSON.parse(fs.readFileSync(ledgerPath, "utf8")); +accountedNanoAiu(ledger); +fs.mkdirSync(outputDirectory); +const client = new CopilotClient({ + mode: "empty", + baseDirectory: path.join(outputDirectory, "copilot"), + workingDirectory: outputDirectory, + connection: RuntimeConnection.forStdio({ path: cliPath }), + requestHandler: new CopilotCreditBudget(path.resolve(ledgerPath)), + useLoggedInUser: true, + logLevel: "error", +}); +const result = { + kind: "credit_control_calibration_not_eval", + sessionId: randomUUID(), + status: "not_started", + usage: [], +}; +let session; +try { + await client.start(); + session = await client.createSession({ + sessionId: result.sessionId, + model: ledger.model, + reasoningEffort: "low", + contextTier: "default", + sessionLimits: { maxAiCredits: 30 }, + capi: { enableWebSocketResponses: false }, + availableTools: [], + skipCustomInstructions: true, + onPermissionRequest: () => ({ + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }), + }); + session.on("assistant.usage", (event) => { + result.usage.push({ + model: event.data.model, + copilotUsage: event.data.copilotUsage, + inputTokens: event.data.inputTokens, + outputTokens: event.data.outputTokens, + }); + }); + result.status = "running"; + await session.sendAndWait({ prompt: "Reply with exactly OK." }, 60_000); + result.status = "completed"; +} catch (error) { + result.status = "failed"; + result.error = error instanceof Error ? error.message : String(error); + process.exitCode = 1; +} finally { + if (session) await session.abort(); + await client.stop(); + fs.writeFileSync( + path.join(outputDirectory, "result.json"), + JSON.stringify(result, null, 2) + "\n", + ); + console.log(JSON.stringify(result)); +} diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs new file mode 100644 index 0000000000..80e1cb9134 --- /dev/null +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs @@ -0,0 +1,224 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import path from "node:path"; + +export const listFixture = { + grocery: ["milk", "eggs", "rice"], + pantry: ["rice", "beans"], + packing: ["passport", "charger", "socks"], + travel: ["charger", "adapter"], + office: ["notebook", "pen", "charger"], + errand: ["pharmacy", "post office"], + weekend: [], +}; +export const fileFixture = { + "report-a.txt": "passport\ncharger\nsocks\n", + "report-b.txt": "charger\nadapter\n", + "trip.txt": "destination: mountain\njacket: required\n", +}; + +export function buildCorpus(files, repo, prA, prB, issueA) { + const file = (name) => path.join(files, name); + return [ + { id: "S1", prompt: "Show my lists." }, + { id: "S2", prompt: `Read ${file("report-a.txt")}.` }, + { + id: "S3", + prompt: `Show the files changed by pull request ${prA} in ${repo}.`, + }, + { id: "S4", prompt: "Add apples to my grocery list." }, + { id: "S5", prompt: "Show my full network configuration." }, + { + id: "M1", + prompt: `Read ${file("report-a.txt")} and ${file("report-b.txt")}.`, + }, + { + id: "M2", + prompt: "Show my full network configuration and the contents of my DNS resolver cache.", + }, + { + id: "M3", + prompt: "Empty my grocery list, then add bread and oranges to it.", + }, + { + id: "M4", + prompt: `Show the changed files and check results for pull request ${prA} in ${repo}.`, + }, + { + id: "M5", + prompt: `Show issue ${issueA} in ${repo}, then add "review issue ${issueA}" to my errand list.`, + }, + { + id: "R1", + prompt: "Which items are on both my grocery list and my pantry list?", + }, + { + id: "R2", + prompt: `Compare ${file("report-a.txt")} and ${file("report-b.txt")}. Which contains more nonempty lines, and by how many?`, + }, + { + id: "R3", + prompt: `Compare pull requests ${prA} and ${prB} in ${repo}. Which needs attention first based on check failures, using the number of changed files as the tie-breaker?`, + }, + { + id: "R4", + prompt: `Read ${file("trip.txt")}. If it says a jacket is required, add jacket to my packing list; otherwise leave the list unchanged. Tell me what you did.`, + }, + { + id: "R5", + prompt: `Read issue ${issueA} in ${repo} and add its exact title to my errand list, but only if that title is not already there.`, + }, + { + id: "A1", + prompt: "Add apples to my list.", + clarification: "The grocery list.", + }, + { + id: "A2", + prompt: `Read the report in ${files}.`, + clarification: "report-b.txt.", + }, + { + id: "A3", + prompt: `Show the files changed by that pull request in ${repo}.`, + clarification: `Pull request ${prA}.`, + }, + { + id: "A4", + prompt: "Clean up my grocery list.", + clarification: "Remove all items but keep the list itself.", + }, + { + id: "A5", + prompt: `Read that file in ${files}.`, + clarification: "trip.txt.", + }, + ]; +} + +export function expectedLists(id, issueA, issueTitle) { + const state = structuredClone(listFixture); + if (id === "S4" || id === "A1") state.grocery.push("apples"); + if (id === "M3") state.grocery = ["bread", "oranges"]; + if (id === "M5") state.errand.push(`review issue ${issueA}`); + if (id === "R4") state.packing.push("jacket"); + if (id === "R5") { + if (!issueTitle) return undefined; + state.errand.push(issueTitle); + } + if (id === "A4") state.grocery = []; + return state; +} + +export function normalizeLists(lists) { + return Object.fromEntries( + lists + .map(({ name, items }) => [name, [...items].sort()]) + .sort(([a], [b]) => a.localeCompare(b)), + ); +} + +export function balancedOrder(cases, candidateIds, repetitions) { + if (!Number.isInteger(repetitions) || repetitions < 1) { + throw new Error("Repetitions must be a positive integer"); + } + + const result = []; + for (let repetition = 0; repetition < repetitions; repetition++) { + for (let i = 0; i < cases.length; i++) { + for (let j = 0; j < candidateIds.length; j++) { + result.push({ + caseId: cases[i].id, + candidate: + candidateIds[ + (i + j + repetition) % candidateIds.length + ], + repetition, + }); + } + } + } + return result; +} + +export function fixtureConfirmationAllowed( + id, + action, + files, + issueTitle, + issueNumber = 2617, +) { + if (!action || typeof action.parameters !== "object") return false; + const { schemaName, actionName, parameters } = action; + const fileCases = { + S2: ["report-a.txt"], + M1: ["report-a.txt", "report-b.txt"], + R2: ["report-a.txt", "report-b.txt"], + R4: ["trip.txt"], + A2: ["report-b.txt"], + A5: ["trip.txt"], + }; + if ( + schemaName === "powershell.powershell-files" && + actionName === "readFile" + ) { + return ( + fileCases[id]?.some( + (name) => + path.resolve(files, name).toLowerCase() === + path.resolve(parameters.path ?? "").toLowerCase(), + ) ?? false + ); + } + + if (schemaName !== "list") return false; + if (actionName === "clearList") { + return ["M3", "A4"].includes(id) && parameters.listName === "grocery"; + } + if (actionName !== "addItems") return false; + const additions = { + S4: ["grocery", ["apples"]], + A1: ["grocery", ["apples"]], + M3: ["grocery", ["bread", "oranges"]], + M5: ["errand", [`review issue ${issueNumber}`]], + R4: ["packing", ["jacket"]], + R5: ["errand", issueTitle ? [issueTitle] : []], + }; + const expected = additions[id]; + return ( + expected !== undefined && + expected[1].length > 0 && + parameters.listName === expected[0] && + JSON.stringify(parameters.items?.slice().sort()) === + JSON.stringify(expected[1].slice().sort()) + ); +} + +export function isClarificationQuestion(id, question) { + if (/\b(confirm|approve|proceed|allow)\b/i.test(question)) return false; + const subject = { + A1: /\blist\b/i, + A2: /\b(report|file)\b/i, + A3: /\b(pull request|PR|number)\b/i, + A4: /\b(clean|remove|empty|change|list)\b/i, + A5: /\bfile\b/i, + }[id]; + return Boolean( + subject?.test(question) && + /\b(which|what|how|choose|specify|mean)\b/i.test(question), + ); +} + +export function shuffled(values, seed) { + const result = [...values]; + let state = seed >>> 0; + for (let i = result.length - 1; i > 0; i--) { + state ^= state << 13; + state ^= state >>> 17; + state ^= state << 5; + const j = (state >>> 0) % (i + 1); + [result[i], result[j]] = [result[j], result[i]]; + } + return result; +} diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs new file mode 100644 index 0000000000..42fc2d24c2 --- /dev/null +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs @@ -0,0 +1,146 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +export function terminalExecutionFailure(toolName, result, success) { + if (!/processCommand|executeAction|continueAction/.test(toolName)) + return false; + return ( + success === false || + (toolName.includes("processCommand") && + /^Error:\s/.test(result?.content ?? "")) || + ["failed", "cancelled", "unavailable", "execution_uncertain"].includes( + result?.structuredContent?.status, + ) + ); +} + +export function percentile(values, fraction) { + if (values.length === 0) return null; + const ordered = [...values].sort((a, b) => a - b); + return ordered[Math.max(0, Math.ceil(fraction * ordered.length) - 1)]; +} + +export function intervalUnionMs(intervals) { + const sorted = intervals + .filter( + ([start, end]) => + Number.isFinite(start) && Number.isFinite(end) && end >= start, + ) + .sort(([a], [b]) => a - b); + let end = -Infinity; + let total = 0; + for (const [start, next] of sorted) { + total += Math.max(0, next - Math.max(start, end)); + end = Math.max(end, next); + } + return total; +} + +export function externalOracle(readiness) { + if (readiness.status !== "passed") + throw new Error("Live readiness has not passed"); + const prs = {}; + for (const entry of readiness.externalEvidence) { + const text = (entry.outcome.output ?? []).join("\n"); + if (entry.actionName === "prFiles") { + const files = [ + ...text.matchAll( + /^(\S+)\s+(?:modified|added|removed|renamed)\s+\d+\s+\d+\s*$/gm, + ), + ].map((match) => match[1]); + const count = text.match(/(\d+) of (\d+) files/); + if ( + !count || + Number(count[1]) !== Number(count[2]) || + new Set(files).size !== Number(count[2]) + ) + throw new Error("Independent PR file snapshot is incomplete"); + prs[entry.number] = { + files: [...new Set(files)], + capturedAt: entry.capturedAt, + }; + } + if (entry.actionName === "prChecks") { + if (!prs[entry.number]) throw new Error("Missing PR file oracle"); + prs[entry.number].checks = text; + } + } + return prs; +} + +// These are conservative evidence checks, not a semantic grader. A human/AI +// reviewer must resolve pending faithfulness grades before accuracy claims. +export function preliminaryGrade(result, evidence) { + if (result.status !== "completed_ungraded") + return { outcome: "incomplete", reason: result.error ?? result.status }; + if (result.routeViolations.length) + return { outcome: "failed", reason: "route_or_interaction_violation" }; + if (result.terminalExecutionFailure) + return { + outcome: "incomplete", + reason: "execution_failed_or_uncertain_no_replay", + }; + if ( + !result.grade.filesUnchanged || + result.grade.listStateMatchesOracle === false + ) + return { + outcome: "failed", + reason: "independent_fixture_oracle_mismatch", + }; + if ( + result.caseId.startsWith("A") && + (!result.grade.clarificationRequested || + !result.grade.noPrematureListMutation) + ) + return { + outcome: "failed", + reason: "clarification_not_verified_before_effects", + }; + const answer = result.answer ?? ""; + if (!answer.trim()) return { outcome: "failed", reason: "no_final_answer" }; + const required = { + S1: [ + "grocery", + "pantry", + "packing", + "travel", + "office", + "errand", + "weekend", + ], + S2: ["passport", "charger", "socks"], + S4: ["apples"], + M1: [ + "report-a.txt", + "report-b.txt", + "passport", + "charger", + "socks", + "adapter", + ], + M3: ["bread", "oranges"], + M5: [evidence.issueTitle], + R1: ["rice"], + R2: ["report-a.txt", "report-b.txt"], + R4: ["jacket"], + R5: [evidence.issueTitle], + A1: ["apples"], + A2: ["charger", "adapter"], + A5: ["destination", "mountain", "jacket", "required"], + }[result.caseId]; + const missing = + required?.filter( + (term) => !answer.toLowerCase().includes(term.toLowerCase()), + ) ?? []; + if (missing.length) + return { + outcome: "pending_review", + reason: "answer_evidence_missing", + missing, + }; + return { + outcome: "pending_review", + reason: "independent_state_checked_final_faithfulness_required", + }; +} diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs new file mode 100644 index 0000000000..3636350974 --- /dev/null +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs @@ -0,0 +1,273 @@ +#!/usr/bin/env node +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { createHash } from "node:crypto"; +import { fileFixture } from "./ghcp-eval-corpus.mjs"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; +import { stageCopilotPlugin } from "../../../tools/scripts/stageCopilotPlugin.mjs"; +import { + checkPort, + makeConfiguration, + startProcess, + stopProcess, + waitForServer, +} from "./discovery-e2e.mjs"; + +const root = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + "../../..", +); +const [outputDirectory, configDirectory, ledgerPath, evidenceMode] = + process.argv.slice(2); +if (!outputDirectory || !configDirectory || !ledgerPath) { + throw new Error( + "Usage: node ghcp-eval-preflight.mjs ", + ); +} +const port = 19024; +await checkPort(port); +fs.mkdirSync(outputDirectory); +const { env, mcp } = makeConfiguration( + outputDirectory, + port, + process.env, + configDirectory, +); +env.TYPEAGENT_COPILOT_CREDIT_LEDGER = path.resolve(ledgerPath); +const fixtures = path.join(outputDirectory, "fixtures"); +fs.mkdirSync(fixtures); +for (const [name, content] of Object.entries(fileFixture)) { + fs.writeFileSync(path.join(fixtures, name), content); +} +env.TYPEAGENT_GHCP_EVAL_FIXTURES = fixtures; +fs.mkdirSync(env.TYPEAGENT_PLUGIN_DATA); +stageCopilotPlugin(path.join(outputDirectory, "plugin")); +const result = { + kind: "catalog_preflight_not_eval", + status: "running", + contracts: [], + missing: [], + searches: [], + externalEvidence: [], +}; +const controller = new AbortController(); +const log = path.join(outputDirectory, "server.stderr.log"); +const stdout = fs.openSync( + path.join(outputDirectory, "server.stdout.log"), + "a", +); +const stderr = fs.openSync(log, "a"); +let server; +let client; +try { + server = startProcess( + process.execPath, + [ + path.join(root, "packages/agentServer/server/dist/server.js"), + "--port", + String(port), + "--config", + "ghcp-eval", + "--idle-timeout", + "300", + ], + { cwd: root, env, stdio: ["ignore", stdout, stderr] }, + ); + fs.closeSync(stdout); + fs.closeSync(stderr); + await waitForServer(server, port, 90, controller.signal, log); + const config = mcp.mcpServers["typeagent-e2e"]; + client = new Client({ name: "ghcp-eval-preflight", version: "1.0.0" }); + await client.connect( + new StdioClientTransport({ + command: config.command, + args: config.args, + env: { ...env, ...config.env }, + stderr: "inherit", + }), + ); + const required = [ + ["list", "listLists"], + ["list", "getList"], + ["list", "addItems"], + ["list", "clearList"], + ["github-cli", "prFiles"], + ["github-cli", "prChecks"], + ["github-cli", "issueView"], + ["powershell.powershell-files", "readFile"], + ["ipconfig", "displayFullConfigurationInformation"], + ["ipconfig", "displayDNSResolverCacheContents"], + ]; + let scopeId; + for (const [schemaName, actionName] of required) { + let contract; + const queries = [`${schemaName} ${actionName}`, actionName]; + if (schemaName === "list" && actionName === "clearList") { + queries.push( + "remove all items from a list but keep the list itself", + ); + } + for (const query of queries) { + const response = await client.callTool( + { + name: "typeagent-searchActions", + arguments: { query }, + }, + undefined, + { timeout: 30_000 }, + ); + if (response.isError) { + throw new Error( + `Discovery failed for ${schemaName}.${actionName}`, + ); + } + result.searches.push({ + query, + candidates: + response.structuredContent?.actions?.map( + ({ schemaName, actionName }) => ({ + schemaName, + actionName, + }), + ) ?? [], + }); + contract = response.structuredContent?.actions?.find( + (action) => + action.schemaName === schemaName && + action.actionName === actionName, + ); + scopeId = response.structuredContent?.scopeId; + if (contract) break; + } + if (contract) result.contracts.push(contract); + else result.missing.push(`${schemaName}.${actionName}`); + } + result.status = result.missing.length === 0 ? "passed" : "blocked"; + if (result.status === "passed" && evidenceMode === "--external-evidence") { + const requests = [ + ["prFiles", 3058], + ["prChecks", 3058], + ["prFiles", 3067], + ["prChecks", 3067], + ["issueView", 2617], + ].map(([actionName, number]) => ({ + schemaName: "github-cli", + actionName, + parameters: { + repo: "microsoft/TypeAgent", + number, + ...(actionName === "prFiles" + ? { includePatch: false, maxFiles: 50 } + : {}), + }, + })); + requests.push( + { schemaName: "list", actionName: "listLists", parameters: {} }, + { + schemaName: "powershell.powershell-files", + actionName: "readFile", + parameters: { path: path.join(fixtures, "report-a.txt") }, + }, + ...[ + "displayFullConfigurationInformation", + "displayDNSResolverCacheContents", + ].map((actionName) => ({ + schemaName: "ipconfig", + actionName, + parameters: {}, + })), + ); + for (const action of requests) { + let response = await client.callTool( + { + name: "typeagent-executeAction", + arguments: { + protocolVersion: 1, + scopeId, + ...action, + }, + }, + undefined, + { timeout: 60_000 }, + ); + const pending = response.structuredContent; + if ( + action.schemaName === "powershell.powershell-files" && + pending?.status === "requires_interaction" && + pending.prompt?.type === "confirmation" && + pending.prompt.action?.parameters?.path === + action.parameters.path + ) { + response = await client.callTool( + { + name: "typeagent-continueAction", + arguments: { + protocolVersion: 1, + scopeId, + operationId: pending.operationId, + interactionId: pending.interactionId, + response: { type: "confirmation", approved: true }, + }, + }, + undefined, + { timeout: 60_000 }, + ); + } + result.externalEvidence.push({ + actionName: action.actionName, + number: action.parameters.number, + capturedAt: new Date().toISOString(), + outcome: + action.schemaName === "ipconfig" + ? { + status: response.structuredContent?.status, + sha256: createHash("sha256") + .update( + JSON.stringify( + response.structuredContent, + ), + ) + .digest("hex"), + redacted: + "network configuration and resolver contents", + } + : (response.structuredContent ?? response), + }); + if ( + response.isError || + response.structuredContent?.status !== "completed" + ) { + result.status = "blocked"; + break; + } + } + } + if (result.status !== "passed") process.exitCode = 1; +} catch (error) { + result.status = "failed"; + result.error = error instanceof Error ? error.message : String(error); + process.exitCode = 1; +} finally { + try { + if (client) await client.close(); + } finally { + if (server) await stopProcess(server); + fs.writeFileSync( + path.join(outputDirectory, "result.json"), + JSON.stringify(result, null, 2) + "\n", + ); + } + console.log( + JSON.stringify({ + status: result.status, + contractCount: result.contracts.length, + missing: result.missing, + error: result.error, + }), + ); +} diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs new file mode 100644 index 0000000000..1752dc5d4e --- /dev/null +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs @@ -0,0 +1,841 @@ +#!/usr/bin/env node +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { performance } from "node:perf_hooks"; +import { fileURLToPath } from "node:url"; +import { randomUUID, createHash } from "node:crypto"; +import { execFileSync } from "node:child_process"; +import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; +import { CopilotCreditBudget } from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; +import { getCopilotPermissionDefault } from "../../dispatcher/dispatcher/dist/reasoning/copilot.js"; +import { + balancedOrder, + buildCorpus, + expectedLists, + fileFixture, + fixtureConfirmationAllowed, + isClarificationQuestion, + listFixture, + normalizeLists, + shuffled, +} from "./ghcp-eval-corpus.mjs"; +import { stageCopilotPlugin } from "../../../tools/scripts/stageCopilotPlugin.mjs"; +import { + externalOracle, + intervalUnionMs, + preliminaryGrade, + terminalExecutionFailure, +} from "./ghcp-eval-grade.mjs"; +import { + checkPort, + makeConfiguration, + startProcess, + stopProcess, + waitForServer, +} from "./discovery-e2e.mjs"; + +const root = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + "../../..", +); +const [ + cliPath, + template, + outputDirectory, + configDirectory, + ledgerPath, + selection = "1,2,3,4,5,6,7", + phase = "pilot", + evidencePath, + pilotCases = "S1", + batchStartText = "0", + batchSizeText = "7", + repetitionsText = "1", +] = process.argv.slice(2); +if ( + !cliPath || + !template || + !outputDirectory || + !configDirectory || + !ledgerPath +) { + throw new Error( + "Usage: node ghcp-eval.mjs [candidate-ids] [pilot|measured] [oracle-evidence.json]", + ); +} +const fixtures = listFixture; +const seededLists = Object.entries(fixtures).map(([name, items]) => ({ + name, + items, +})); +const port = 19024; +const excludedSchemas = Object.keys( + JSON.parse( + fs.readFileSync( + path.join(root, "packages/defaultAgentProvider/data/config.json"), + "utf8", + ), + ).mcpServers ?? {}, +); +const toolNames = { + nl: ["typeagent-processCommand"], + structured: [ + "typeagent-searchActions", + "typeagent-executeAction", + "typeagent-continueAction", + "typeagent-cancelAction", + ], +}; +const nativeTools = [ + "view", + "glob", + "rg", + "powershell", + "read_powershell", + "stop_powershell", + "list_powershell", + "web_fetch", + "ask_user", +].map((name) => `builtin:${name}`); +const candidates = [ + { id: 1, policy: "NL only", fallback: false, tools: toolNames.nl }, + { id: 2, policy: "NL only", fallback: true, tools: toolNames.nl }, + { id: 3, policy: "Structured discovery only", tools: toolNames.structured }, + { + id: 4, + policy: "Structured current-contract reuse only", + tools: toolNames.structured, + }, + { + id: 5, + policy: "Production mixed", + fallback: false, + tools: [...toolNames.nl, ...toolNames.structured], + }, + { + id: 6, + policy: "Production mixed", + fallback: true, + tools: [...toolNames.nl, ...toolNames.structured], + }, + { id: 7, policy: "Native only" }, +]; + +function findListStores(directory) { + return fs + .readdirSync(directory, { recursive: true }) + .filter((name) => path.basename(name) === "lists.json") + .map((name) => path.join(directory, name)); +} + +function collectObservations(result, tracePath) { + const trace = fs.existsSync(tracePath) + ? fs.readFileSync(tracePath, "utf8") + : ""; + result.typeagentEvents = trace.trim() + ? trace + .trim() + .split("\n") + .map((line) => JSON.parse(line)) + : []; + result.fallback = + result.candidate === 7 + ? null + : Object.fromEntries( + ["enter", "completed", "failed"].map((name) => [ + name, + result.typeagentEvents.filter( + ({ event }) => + event === `translation.reasoning.${name}`, + ).length, + ]), + ); + const usage = result.usage.filter((entry) => !entry.preparation); + const tools = result.tools.filter((entry) => !entry.preparation); + const modelMs = usage.every(({ durationMs }) => durationMs !== null) + ? usage.reduce((sum, entry) => sum + entry.durationMs, 0) + : null; + const toolMs = tools.every(({ endMs }) => endMs !== undefined) + ? intervalUnionMs(tools.map(({ startMs, endMs }) => [startMs, endMs])) + : null; + const credits = JSON.parse( + fs.readFileSync(ledgerPath, "utf8"), + ).reservations; + result.credits = credits.filter( + (entry) => + entry.sessionId === result.sessionId || + entry.sessionId?.startsWith(`${result.sessionId}::`), + ); + result.measurements = { + rootModelInvocations: usage.length, + rootModelMs: modelMs, + toolUnionMs: toolMs, + unattributedMs: + result.e2eMs !== null && + modelMs !== null && + toolMs !== null && + modelMs + toolMs <= result.e2eMs + ? result.e2eMs - modelMs - toolMs + : null, + nonAdditiveTiming: + modelMs !== null && + toolMs !== null && + modelMs + toolMs > result.e2eMs, + mcpCalls: tools.filter(({ name }) => /typeagent-/.test(name)).length, + internalModelInvocations: result.credits.filter( + ({ sessionId }) => sessionId !== result.sessionId, + ).length, + backendSpans: result.typeagentEvents.filter(({ event }) => + ["action.completed", "action.failed"].includes(event), + ), + transportRetries: null, + translationMs: null, + scriptedInteractionCount: result.interactions.length, + humanWaitingMs: 0, + systemActiveE2eMs: result.e2eMs, + timingNote: + "Synchronous fixture answers; nested backend spans are not additive with outer tools. Missing stages remain null.", + }; +} + +async function trial(candidate, directory, testCase, workspace, evidence) { + fs.mkdirSync(directory); + const { env, mcp } = makeConfiguration( + directory, + port, + process.env, + configDirectory, + ); + env.TYPEAGENT_MODE = "mcp"; + env.TYPEAGENT_COPILOT_CREDIT_LEDGER = path.resolve(ledgerPath); + env.TYPEAGENT_GHCP_EVAL_CLI = cliPath; + env.COPILOT_HOME = path.join(directory, "nested-copilot"); + env.COPILOT_REASONING_MODEL = "gpt-5.6-sol"; + env.COPILOT_REASONING_EFFORT = "high"; + env.TYPEAGENT_REASONING_TIMEOUT_MS = "90000"; + env.DEBUG = "typeagent:request"; + env.TYPEAGENT_GHCP_EVAL_FIXTURES = workspace; + const sessionId = randomUUID(); + env.TYPEAGENT_COPILOT_CREDIT_SESSION_SCOPE = sessionId; + env.TYPEAGENT_GHCP_EVAL_TRACE = path.join(directory, "events.jsonl"); + fs.cpSync(path.join(template, "data"), env.TYPEAGENT_USER_DATA_DIR, { + recursive: true, + filter: (entry) => !entry.endsWith(".lock"), + }); + fs.cpSync(path.join(template, "plugin-data"), env.TYPEAGENT_PLUGIN_DATA, { + recursive: true, + }); + const stores = findListStores(env.TYPEAGENT_USER_DATA_DIR); + if (stores.length !== 1) + throw new Error("Expected exactly one disposable list store"); + fs.writeFileSync(stores[0], JSON.stringify(seededLists)); + const sessionDataPath = path.join( + path.dirname(path.dirname(stores[0])), + "data.json", + ); + const sessionData = JSON.parse(fs.readFileSync(sessionDataPath, "utf8")); + sessionData.settings ??= {}; + for (const key of ["schemas", "actions"]) { + sessionData.settings[key] = { + ...sessionData.settings[key], + ...Object.fromEntries(excludedSchemas.map((name) => [name, false])), + }; + } + fs.writeFileSync(sessionDataPath, JSON.stringify(sessionData, null, 2)); + for (const [name, contents] of Object.entries(fileFixture)) { + fs.writeFileSync(path.join(workspace, name), contents); + } + stageCopilotPlugin(path.join(directory, "plugin")); + const pluginMcpPath = path.join(directory, "plugin", ".mcp.json"); + const pluginMcp = JSON.parse(fs.readFileSync(pluginMcpPath, "utf8")); + fs.writeFileSync( + pluginMcpPath, + JSON.stringify( + { + mcpServers: { + typeagent: { + ...pluginMcp.mcpServers.typeagent, + tools: candidate.tools, + }, + }, + }, + null, + 2, + ), + ); + const config = mcp.mcpServers["typeagent-e2e"]; + config.tools = candidate.tools; + if (candidate.fallback !== undefined) { + env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK = candidate.fallback + ? "enabled" + : "disabled"; + config.env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK = + env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK; + } + const result = { + phase, + candidate: candidate.id, + caseId: testCase.id, + prompt: testCase.prompt, + sessionId, + status: "not_started", + usage: [], + tools: [], + interactions: [], + routeViolations: [], + e2eMs: null, + preparationMs: null, + grade: null, + permissions: [], + toolResults: [], + }; + let server; + let client; + let session; + let preparation = candidate.id === 4; + let clarificationGiven = false; + let confirmationCount = 0; + let executionStopped = false; + let measuredStart; + const network = ["S5", "M2"].includes(testCase.id) + ? { toolResults: [] } + : undefined; + const approvedInteractions = new Set(); + const started = performance.now(); + try { + if (candidate.id !== 7) { + await checkPort(port); + const stdout = fs.openSync( + path.join(directory, "server.stdout.log"), + "a", + ); + const stderrPath = path.join(directory, "server.stderr.log"); + const stderr = fs.openSync(stderrPath, "a"); + server = startProcess( + process.execPath, + [ + path.join( + root, + "packages/agentServer/server/dist/server.js", + ), + "--port", + String(port), + "--config", + "ghcp-eval", + "--idle-timeout", + "300", + ], + { cwd: root, env, stdio: ["ignore", stdout, stderr] }, + ); + fs.closeSync(stdout); + fs.closeSync(stderr); + await waitForServer( + server, + port, + 30, + new AbortController().signal, + stderrPath, + ); + } + client = new CopilotClient({ + mode: "empty", + baseDirectory: path.join(directory, "outer-copilot"), + workingDirectory: workspace, + connection: RuntimeConnection.forStdio({ path: cliPath }), + env: + candidate.id === 7 + ? Object.fromEntries( + Object.entries(env).filter( + ([key]) => !key.startsWith("TYPEAGENT_"), + ), + ) + : env, + builtinPluginDirectories: + candidate.id === 5 || candidate.id === 6 + ? [path.join(directory, "plugin")] + : [], + requestHandler: new CopilotCreditBudget(path.resolve(ledgerPath)), + useLoggedInUser: true, + logLevel: "error", + }); + await client.start(); + session = await client.createSession({ + sessionId: result.sessionId, + model: "gpt-5.6-sol", + reasoningEffort: "high", + contextTier: "default", + capi: { enableWebSocketResponses: false }, + sessionLimits: { maxAiCredits: 60 }, + workingDirectory: workspace, + skipCustomInstructions: true, + availableTools: + candidate.id === 7 + ? nativeTools + : [ + "mcp:*", + ...(candidate.id >= 5 + ? nativeTools + : ["builtin:ask_user"]), + ], + ...(candidate.id === 7 + ? {} + : { mcpServers: { "typeagent-e2e": config } }), + ...(candidate.id === 7 || candidate.id >= 5 + ? {} + : { + systemMessage: { + mode: "append", + content: `${candidate.policy}. Use only the exposed TypeAgent MCP interface. Preserve confirmation and clarification; never replay failed or uncertain effects through another route.`, + }, + }), + onPermissionRequest: (request) => { + result.permissions.push({ + kind: request.kind, + readOnly: request.readOnly, + }); + if (request.managedApprovalRequired !== true) { + const safe = getCopilotPermissionDefault(request); + if (safe) return safe; + if (request.kind === "mcp") return { kind: "approve-once" }; + if ( + request.kind === "url" && + request.requestSandboxBypass !== true + ) { + const url = new URL(request.url); + if (["https:", "http:"].includes(url.protocol)) + return { kind: "approve-once" }; + } + } + return { + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }; + }, + onUserInputRequest: (request) => { + result.interactions.push(request.question); + if (testCase.clarification && !clarificationGiven) { + if ( + !isClarificationQuestion(testCase.id, request.question) + ) { + result.routeViolations.push( + "confirmation-or-unrelated-question-before-clarification", + ); + throw new Error( + "Clarification is required before effect confirmation.", + ); + } + clarificationGiven = true; + result.stateAtClarification = normalizeLists( + JSON.parse(fs.readFileSync(stores[0], "utf8")), + ); + return { + answer: testCase.clarification, + wasFreeform: true, + }; + } + const pending = result.toolResults.findLast( + (tool) => + tool.result?.structuredContent?.status === + "requires_interaction", + )?.result.structuredContent; + const action = pending?.prompt?.action; + if ( + confirmationCount < 4 && + pending?.prompt?.type === "confirmation" && + fixtureConfirmationAllowed( + testCase.id, + action, + workspace, + evidence?.issueTitle, + ) + ) { + const yes = request.choices?.find((choice) => + /^(yes|approve|confirm|proceed|allow)\b/i.test(choice), + ); + confirmationCount++; + approvedInteractions.add(pending.interactionId); + return { + answer: yes ?? "Yes", + wasFreeform: yes === undefined, + }; + } + throw new Error( + "No authorized scripted answer for this interaction.", + ); + }, + hooks: { + onPreToolUse: (input) => { + const unauthorizedContinuation = + input.toolName.includes("continueAction") && + input.toolArgs?.response?.approved === true && + !approvedInteractions.has( + input.toolArgs?.interactionId, + ); + const forbidden = + unauthorizedContinuation || + (executionStopped && + !/ask_user|cancelAction/.test(input.toolName)) || + (candidate.id === 4 && + ((!preparation && + input.toolName.includes("searchActions")) || + (preparation && + input.toolName.includes("executeAction")))); + if (forbidden) { + result.routeViolations.push(input.toolName); + return { + permissionDecision: "deny", + permissionDecisionReason: + "Evaluation route/interaction policy denied this call; do not replay it.", + }; + } + return undefined; + }, + }, + }); + session.on("assistant.usage", (event) => + result.usage.push({ + model: event.data.model, + copilotUsage: event.data.copilotUsage, + durationMs: event.data.duration ?? null, + preparation, + }), + ); + session.on("tool.execution_start", (event) => + result.tools.push({ + toolCallId: event.data.toolCallId, + name: event.data.toolName, + arguments: event.data.arguments, + preparation, + startMs: performance.now() - started, + }), + ); + session.on("tool.execution_complete", (event) => { + if (network) network.toolResults.push(event.data); + const tool = result.tools.find( + (tool) => tool.toolCallId === event.data.toolCallId, + ); + if (tool) tool.endMs = performance.now() - started; + if ( + terminalExecutionFailure( + tool?.name ?? "", + event.data.result, + event.data.success, + ) + ) + executionStopped = true; + result.toolResults.push({ + toolCallId: event.data.toolCallId, + success: event.data.success, + result: + testCase.id === "S5" || testCase.id === "M2" + ? "[network evidence withheld]" + : event.data.result, + }); + }); + result.status = "running"; + if (preparation) { + const preparationStart = performance.now(); + await session.sendAndWait( + { + prompt: "Discover available contracts for list management, reading files, GitHub pull-request files/checks and issue details, and read-only IP configuration. Do not execute actions, inspect contents, establish preferred targets, or guess future requests.", + }, + 90_000, + ); + result.preparationMs = performance.now() - preparationStart; + preparation = false; + } + if (network) { + network.before = { + capturedAt: new Date().toISOString(), + configuration: execFileSync("ipconfig.exe", ["/all"], { + encoding: "utf8", + timeout: 15_000, + }), + dns: + testCase.id === "M2" + ? execFileSync("ipconfig.exe", ["/displaydns"], { + encoding: "utf8", + timeout: 15_000, + }) + : null, + }; + } + measuredStart = performance.now(); + result.promptAcceptedAt = new Date().toISOString(); + const answer = await session.sendAndWait( + { prompt: result.prompt }, + 90_000, + ); + result.e2eMs = performance.now() - measuredStart; + result.finalResponseAt = new Date().toISOString(); + result.answer = answer?.data.content ?? ""; + const after = JSON.parse(fs.readFileSync(stores[0], "utf8")); + const correctNames = Object.keys(fixtures).every((name) => + new RegExp(`\\b${name}\\b`, "i").test(result.answer), + ); + const expected = expectedLists(testCase.id, 2617, evidence?.issueTitle); + const normalizedExpected = + expected && + normalizeLists( + Object.entries(expected).map(([name, items]) => ({ + name, + items, + })), + ); + result.grade = { + listStateMatchesOracle: normalizedExpected + ? JSON.stringify(normalizeLists(after)) === + JSON.stringify(normalizedExpected) + : null, + filesUnchanged: Object.entries(fileFixture).every( + ([name, content]) => + fs.readFileSync(path.join(workspace, name), "utf8") === + content, + ), + containsAllNames: testCase.id === "S1" ? correctNames : null, + clarificationRequested: testCase.clarification + ? clarificationGiven + : null, + noPrematureListMutation: testCase.clarification + ? JSON.stringify(result.stateAtClarification) === + JSON.stringify(normalizeLists(seededLists)) + : null, + requiresManualFaithfulnessCheck: true, + }; + result.finalLists = normalizeLists(after); + if (testCase.id === "S5" || testCase.id === "M2") { + network.answer = result.answer; + network.after = { + capturedAt: new Date().toISOString(), + configuration: execFileSync("ipconfig.exe", ["/all"], { + encoding: "utf8", + timeout: 15_000, + }), + dns: + testCase.id === "M2" + ? execFileSync("ipconfig.exe", ["/displaydns"], { + encoding: "utf8", + timeout: 15_000, + }) + : null, + }; + result.answerSha256 = createHash("sha256") + .update(result.answer) + .digest("hex"); + result.answer = + "[network response withheld from sanitized results]"; + } + result.status = "completed_ungraded"; + if ( + phase === "pilot" && + candidate.id !== 7 && + ((testCase.id === "S1" && !correctNames) || + result.grade.listStateMatchesOracle === false || + !result.grade.filesUnchanged) + ) { + result.status = "pilot_needs_review"; + } + } catch (error) { + result.status = "failed"; + result.error = error instanceof Error ? error.message : String(error); + } finally { + if (measuredStart !== undefined && result.e2eMs === null) + result.e2eMs = performance.now() - measuredStart; + try { + if (session) await session.abort(); + if (client) await client.stop(); + } finally { + try { + if (server) await stopProcess(server); + } finally { + result.totalIncludingSetupMs = performance.now() - started; + collectObservations(result, env.TYPEAGENT_GHCP_EVAL_TRACE); + result.terminalExecutionFailure = executionStopped; + result.providerUsage = { + before: sessionData.tokens ?? null, + after: + JSON.parse(fs.readFileSync(sessionDataPath, "utf8")) + .tokens ?? null, + coverage: + "Persisted TypeAgent token counters only; unflushed calls and embedding usage may be absent. Not Copilot credits.", + }; + result.preliminaryGrade = preliminaryGrade( + result, + evidence ?? {}, + ); + if (network) + fs.writeFileSync( + path.join(directory, "private-network-evidence.json"), + JSON.stringify(network, null, 2), + ); + fs.writeFileSync( + path.join(directory, "result.json"), + JSON.stringify(result, null, 2) + "\n", + ); + } + } + } + return result; +} + +const repetitions = phase === "pilot" ? 1 : Number(repetitionsText); +const batchStart = Number(batchStartText); +const batchSize = phase === "pilot" ? 7 : Number(batchSizeText); +if ( + !Number.isInteger(batchStart) || + batchStart < 0 || + !Number.isInteger(batchSize) || + batchSize < 1 || + batchSize > 7 +) + throw new Error("Each batch must contain between one and seven trials"); +if (phase === "measured" && (batchStart % 7 !== 0 || batchSize !== 7)) + throw new Error( + "Measured batches must preserve all seven candidates for one paired case", + ); +fs.mkdirSync(outputDirectory, { recursive: true }); +const resultsPath = path.join(outputDirectory, "results.json"); +const results = fs.existsSync(resultsPath) + ? JSON.parse(fs.readFileSync(resultsPath, "utf8")) + : []; +if (results.length !== batchStart) + throw new Error( + "Batch start must equal the persisted completed/failed trial count; never replay an uncertain trial", + ); +const selected = selection.split(",").map(Number); +if ( + selected.some((id) => !candidates.some((candidate) => candidate.id === id)) +) { + throw new Error("Unknown pilot candidate"); +} +if (!["pilot", "measured"].includes(phase)) + throw new Error("Unknown run phase"); +const workspace = path.join(outputDirectory, "workspace"); +fs.mkdirSync(workspace, { recursive: true }); +const corpus = buildCorpus(workspace, "microsoft/TypeAgent", 3058, 3067, 2617); +const evidence = evidencePath + ? JSON.parse(fs.readFileSync(evidencePath, "utf8")) + : undefined; +if ( + phase === "measured" && + (!evidence?.issueTitle || + new Set(selected).size !== 7 || + !evidence.readinessFile) +) { + throw new Error( + "Measured runs require independent issue evidence and all seven candidates", + ); +} +if (evidence?.readinessFile) + evidence.prOracles = externalOracle( + JSON.parse( + fs.readFileSync( + path.resolve( + path.dirname(evidencePath), + evidence.readinessFile, + ), + "utf8", + ), + ), + ); +const seed = 20260924; +const cases = + phase === "pilot" + ? corpus.filter(({ id }) => pilotCases.split(",").includes(id)) + : shuffled(corpus, seed); +if (cases.length === 0) throw new Error("No cases selected"); +const order = balancedOrder( + cases, + phase === "pilot" ? selected : shuffled(selected, seed), + repetitions, +); +const specification = + JSON.stringify( + { + runnerSha256: createHash("sha256") + .update(fs.readFileSync(fileURLToPath(import.meta.url))) + .digest("hex"), + evidenceSha256: evidencePath + ? createHash("sha256") + .update(fs.readFileSync(evidencePath)) + .digest("hex") + : null, + phase, + repetitions, + seed, + order, + cases, + candidates, + model: "gpt-5.6-sol", + reasoningEffort: "high", + concurrency: 1, + commit: execFileSync("git", ["rev-parse", "HEAD"], { + encoding: "utf8", + }).trim(), + trialTimeoutMs: 90_000, + preparationTimeoutMs: 90_000, + perSessionRequestLimit: 24, + cumulativeRequestLimit: 2000, + requestCreditReservation: 2118, + sessionCreditSoftLimit: 60, + ledgerPath, + gradingStatus: + "independent fixture oracles; explicit final-answer review required", + nativeTools, + cliVersion: execFileSync(cliPath, ["--version"], { + encoding: "utf8", + }).trim(), + internalTools: + "Unmodified production TypeAgent reasoning toolset; effects and credits gated", + disabledShippedMcpSchemas: excludedSchemas, + }, + null, + 2, + ) + "\n"; +const specificationPath = path.join(outputDirectory, "specification.json"); +if ( + fs.existsSync(specificationPath) && + fs.readFileSync(specificationPath, "utf8") !== specification +) + throw new Error("Frozen run specification changed; start a distinct run"); +fs.writeFileSync(specificationPath, specification); +for (const entry of order.slice(batchStart, batchStart + batchSize)) { + const candidate = candidates.find(({ id }) => id === entry.candidate); + const testCase = cases.find(({ id }) => id === entry.caseId); + const result = await trial( + candidate, + path.join( + outputDirectory, + `${entry.repetition}-${entry.caseId}-candidate-${candidate.id}`, + ), + testCase, + workspace, + evidence, + ); + result.repetition = entry.repetition; + results.push(result); + fs.writeFileSync( + path.join(outputDirectory, "results.json"), + JSON.stringify(results, null, 2) + "\n", + ); + console.log( + JSON.stringify({ + candidate: candidate.id, + caseId: testCase.id, + repetition: entry.repetition, + status: result.status, + error: result.error, + }), + ); + if ( + (phase === "pilot" && result.status !== "completed_ungraded") || + /credit|budget|reservation|Agent server exited|permission orchestrator/i.test( + result.error ?? "", + ) + ) { + process.exitCode = 1; + break; + } +} diff --git a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs new file mode 100644 index 0000000000..f0aa607c3f --- /dev/null +++ b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs @@ -0,0 +1,209 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import { test } from "node:test"; +import { + externalOracle, + intervalUnionMs, + percentile, + terminalExecutionFailure, +} from "../ghcp-eval-grade.mjs"; +import { + balancedOrder, + buildCorpus, + expectedLists, + fixtureConfirmationAllowed, + isClarificationQuestion, + listFixture, + normalizeLists, + shuffled, +} from "../ghcp-eval-corpus.mjs"; + +const corpus = buildCorpus("C:\\fixtures", "owner/repo", 10, 20, 30); +test("failure detection preserves structured status and NL errors, not check-result words", () => { + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "Error: denied" }, + true, + ), + true, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + { structuredContent: { status: "execution_uncertain" } }, + true, + ), + true, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-executeAction", + { structuredContent: { status: "requires_interaction" } }, + true, + ), + false, + ); + assert.equal( + terminalExecutionFailure( + "typeagent-processCommand", + { content: "No failed checks." }, + true, + ), + false, + ); + assert.equal( + terminalExecutionFailure("typeagent-searchActions", {}, false), + false, + ); +}); +test("nested/parallel tool durations are not double-counted and empty tails are unknown", () => { + assert.equal( + intervalUnionMs([ + [0, 10], + [2, 8], + [8, 15], + [20, 25], + ]), + 20, + ); + assert.equal(percentile([], 0.95), null); + assert.equal(percentile([9, 1, 3], 0.5), 3); + assert.equal(percentile([9, 1, 3], 0.95), 9); +}); +test("independent PR file evidence must be complete", () => { + const snapshot = { + status: "passed", + externalEvidence: [ + { + actionName: "prFiles", + number: 1, + outcome: { output: ["1 of 1 files\nsrc/a.ts modified 1 0"] }, + }, + ], + }; + assert.deepEqual(externalOracle(snapshot)[1].files, ["src/a.ts"]); + snapshot.externalEvidence[0].outcome.output[0] = + "1 of 2 files\nsrc/a.ts modified 1 0"; + assert.throws(() => externalOracle(snapshot), /incomplete/); +}); +test("confirmation of a guessed referent is not clarification", () => { + assert.equal( + isClarificationQuestion("A1", "Which list should receive apples?"), + true, + ); + assert.equal( + isClarificationQuestion("A1", "Add apples to your grocery list?"), + false, + ); + assert.equal( + isClarificationQuestion("A2", "Which report should I read?"), + true, + ); + assert.equal( + isClarificationQuestion("A2", "Which service contains your data?"), + false, + ); + assert.equal( + isClarificationQuestion( + "A4", + "How should I clean up your grocery list?", + ), + true, + ); +}); +test("the full workload has exactly four five-case cohorts", () => { + assert.equal(corpus.length, 20); + assert.equal(new Set(corpus.map(({ id }) => id)).size, 20); + for (const cohort of ["S", "M", "R", "A"]) { + assert.equal( + corpus.filter(({ id }) => id.startsWith(cohort)).length, + 5, + ); + } + assert.equal(corpus.filter(({ clarification }) => clarification).length, 5); +}); +test("scripted confirmations are limited to exact disposable fixture actions", () => { + const add = { + schemaName: "list", + actionName: "addItems", + parameters: { listName: "grocery", items: ["apples"] }, + }; + assert.equal(fixtureConfirmationAllowed("S4", add, "C:\\fixtures"), true); + assert.equal(fixtureConfirmationAllowed("S1", add, "C:\\fixtures"), false); + assert.equal( + fixtureConfirmationAllowed( + "S4", + { ...add, schemaName: "github-cli" }, + "C:\\fixtures", + ), + false, + ); + assert.equal( + fixtureConfirmationAllowed( + "R5", + { + ...add, + parameters: { listName: "errand", items: ["guessed title"] }, + }, + "C:\\fixtures", + ), + false, + ); + const read = { + schemaName: "powershell.powershell-files", + actionName: "readFile", + parameters: { path: "C:\\fixtures\\report-a.txt" }, + }; + assert.equal(fixtureConfirmationAllowed("S2", read, "C:\\fixtures"), true); + assert.equal(fixtureConfirmationAllowed("A2", read, "C:\\fixtures"), false); +}); +test("seeded ordering is reproducible without dropping examples", () => { + assert.deepEqual(shuffled(corpus, 42), shuffled(corpus, 42)); + assert.notDeepEqual(shuffled(corpus, 42), shuffled(corpus, 43)); + assert.equal(new Set(shuffled(corpus, 42).map(({ id }) => id)).size, 20); +}); +test("answers are separate from prompts and fixed inputs are substituted", () => { + assert.equal( + corpus.find(({ id }) => id === "A1").prompt, + "Add apples to my list.", + ); + assert.equal( + corpus.find(({ id }) => id === "A3").clarification, + "Pull request 10.", + ); + assert.match( + corpus.find(({ id }) => id === "M5").prompt, + /review issue 30/, + ); +}); +test("balanced rotations retain every candidate/case/repetition", () => { + const order = balancedOrder(corpus, [1, 2, 3, 4, 5, 6, 7], 2); + assert.equal(order.length, 280); + assert.equal( + new Set(order.map((entry) => JSON.stringify(entry))).size, + 280, + ); + assert.deepEqual( + order.slice(7, 14).map(({ candidate }) => candidate), + [2, 3, 4, 5, 6, 7, 1], + ); + assert.throws(() => balancedOrder(corpus, [1], 0), /positive integer/); +}); +test("independent list oracles preserve all unrelated state", () => { + assert.deepEqual(expectedLists("M3", 30).grocery, ["bread", "oranges"]); + assert.deepEqual(expectedLists("A4", 30).grocery, []); + assert.deepEqual(expectedLists("S4", 30).pantry, listFixture.pantry); + assert.deepEqual(expectedLists("S1", 30), listFixture); + assert.equal(expectedLists("R5", 30), undefined); + assert.equal( + expectedLists("R5", 30, "Exact title").errand.at(-1), + "Exact title", + ); + assert.equal(listFixture.grocery.includes("apples"), false); + assert.deepEqual(normalizeLists([{ name: "a", items: ["b", "a"] }]), { + a: ["a", "b"], + }); +}); diff --git a/ts/packages/copilot-plugin/src/mcp/agentServer.ts b/ts/packages/copilot-plugin/src/mcp/agentServer.ts index 73e406818c..bfadee59e6 100644 --- a/ts/packages/copilot-plugin/src/mcp/agentServer.ts +++ b/ts/packages/copilot-plugin/src/mcp/agentServer.ts @@ -47,6 +47,16 @@ function toolError(text: string): CallToolResult { return { isError: true, content: [{ type: "text", text }] }; } +export function translationFallbackOptions(value: string | undefined) { + if (value === undefined) return undefined; + if (value !== "enabled" && value !== "disabled") { + throw new Error( + "TYPEAGENT_TRANSLATION_REASONING_FALLBACK must be enabled or disabled", + ); + } + return { translationReasoningFallback: value === "enabled" }; +} + /** * Format a large result for display. Strips markdown formatting and wraps * in a code fence so the CLI preserves newlines and structured layout. @@ -295,6 +305,9 @@ export class TypeAgentMcpServer { dispatcher, command, extra?.signal, + translationFallbackOptions( + process.env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK, + ), ); if (pendingPrompts.length > 0) { diff --git a/ts/packages/copilot-plugin/src/shared/typeagent-client.ts b/ts/packages/copilot-plugin/src/shared/typeagent-client.ts index d7395e0952..f9f2f90620 100644 --- a/ts/packages/copilot-plugin/src/shared/typeagent-client.ts +++ b/ts/packages/copilot-plugin/src/shared/typeagent-client.ts @@ -153,6 +153,7 @@ export async function submitCancellableCommand( dispatcher: Dispatcher, command: string, signal?: AbortSignal, + options?: Parameters[2], ): Promise { if (signal?.aborted) return { cancelled: true }; const clientRequestId = `copilot-plugin-${randomUUID()}`; @@ -173,7 +174,7 @@ export async function submitCancellableCommand( const submitted = await dispatcher.submitCommand( command, undefined, - undefined, + options, clientRequestId, ); if (!submitted.ok) { diff --git a/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts b/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts index 45d2098164..1018a9f146 100644 --- a/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts +++ b/ts/packages/copilot-plugin/test/naturalLanguageClient.spec.ts @@ -7,8 +7,38 @@ import { createClientIO, submitCancellableCommand, } from "../src/shared/typeagent-client.js"; +import { translationFallbackOptions } from "../src/mcp/agentServer.js"; describe("unchanged user-originated natural-language requests", () => { + it("changes only translation-to-reasoning fallback and preserves exact NL", async () => { + const submitCommand = jest.fn(async () => ({ + ok: true, + entry: { + requestId: "id", + completion: Promise.resolve(undefined), + }, + })); + const dispatcher = { submitCommand } as unknown as Dispatcher; + await submitCancellableCommand( + dispatcher, + "Show my lists.", + undefined, + translationFallbackOptions("disabled"), + ); + expect(submitCommand).toHaveBeenCalledWith( + "Show my lists.", + undefined, + { translationReasoningFallback: false }, + expect.any(String), + ); + expect(translationFallbackOptions("enabled")).toEqual({ + translationReasoningFallback: true, + }); + expect(translationFallbackOptions(undefined)).toBeUndefined(); + expect(() => translationFallbackOptions("false")).toThrow( + "must be enabled or disabled", + ); + }); it.each([ "list my playlists", "learn: create a playlist", diff --git a/ts/packages/defaultAgentProvider/data/config.ghcp-eval.json b/ts/packages/defaultAgentProvider/data/config.ghcp-eval.json new file mode 100644 index 0000000000..b00279ad78 --- /dev/null +++ b/ts/packages/defaultAgentProvider/data/config.ghcp-eval.json @@ -0,0 +1,21 @@ +{ + "description": "Isolated GHCP evaluation: lists and read-only GitHub, file, and network workload", + "agents": { + "list": { + "name": "@typeagent/list-agent", + "execMode": "dispatcher" + }, + "github-cli": { + "name": "@typeagent/github-cli-agent", + "execMode": "dispatcher" + }, + "powershell": { + "name": "@typeagent/powershell-typeagent", + "execMode": "dispatcher" + }, + "ipconfig": { + "name": "@typeagent/ipconfig-agent", + "execMode": "dispatcher" + } + } +} diff --git a/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts b/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts index 413a62f5f3..90de3e4b6a 100644 --- a/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts +++ b/ts/packages/dispatcher/dispatcher/src/context/dispatcher/handlers/requestCommandHandler.ts @@ -38,6 +38,7 @@ import { isPendingRequest, } from "../../../translation/multipleActionSchema.js"; import registerDebug from "debug"; +import { recordGhcpEvalEvent } from "../../../execute/ghcpEvalPolicy.js"; import ExifReader from "exifreader"; import { ProfileNames } from "../../../utils/profileNames.js"; import { @@ -1040,9 +1041,31 @@ export class RequestCommandHandler implements CommandHandler { return; } let reasoningHandled = false; - if (needsReasoning && !systemContext.noReasoning) { + if (needsReasoning) { + recordGhcpEvalEvent("translation.reasoning.decision", { + hasUnknownAction, + hasClarificationAction, + enabled: + !systemContext.noReasoning && + systemContext.currentOptions + ?.translationReasoningFallback !== false, + }); + } + if ( + needsReasoning && + !systemContext.noReasoning && + systemContext.currentOptions?.translationReasoningFallback !== + false + ) { try { + debugRequest("translation.reasoning.enter", { + hasUnknownAction, + hasClarificationAction, + }); + recordGhcpEvalEvent("translation.reasoning.enter"); await runConfiguredReasoning(request, context); + debugRequest("translation.reasoning.completed"); + recordGhcpEvalEvent("translation.reasoning.completed"); reasoningHandled = true; if (!applyPowerShellCapabilityOutcome(systemContext)) { setDisposition(systemContext, { @@ -1051,6 +1074,8 @@ export class RequestCommandHandler implements CommandHandler { }); } } catch (e: any) { + debugRequest("translation.reasoning.failed"); + recordGhcpEvalEvent("translation.reasoning.failed"); debugRequest( `Reasoning fallback failed, using default handler: ${e.message}`, ); @@ -1070,7 +1095,8 @@ export class RequestCommandHandler implements CommandHandler { if ( !systemContext.noReasoning && execResult !== undefined && - execResult.fallbackToReasoning + execResult.fallbackToReasoning && + process.env.TYPEAGENT_GHCP_EVAL_FIXTURES === undefined ) { const needsErrorReasoning = requestAction.actions.some( ({ action }) => { diff --git a/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts b/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts index 1589edff32..aa731be512 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/actionHandlers.ts @@ -1,6 +1,12 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. +import { + assertGhcpEvalAction, + markGhcpEvalExecutionFailure, + recordGhcpEvalEvent, +} from "./ghcpEvalPolicy.js"; + import { ExecutableAction, FullAction, @@ -379,6 +385,12 @@ export async function executeAction( ): Promise { const action = executableAction.action; const schemaName = action.schemaName; + assertGhcpEvalAction(schemaName, action.actionName, action.parameters); + recordGhcpEvalEvent("action.admitted", { + schemaName, + actionName: action.actionName, + parameters: action.parameters, + }); // For nested action calls (e.g., from TaskFlow scripts), agentContext may be // the agent's own context rather than CommandHandlerContext. In that case, // use _systemContext which exposes the CommandHandlerContext. @@ -533,6 +545,13 @@ export async function executeAction( schemaName, ); + if (outcome.result.error !== undefined) + markGhcpEvalExecutionFailure(); + recordGhcpEvalEvent("action.completed", { + ...eventData, + success: outcome.result.error === undefined, + elapsedMs: Date.now() - actionStartedAt, + }); logActionCompleted(systemContext.logger, { ...eventData, success: outcome.result.error === undefined, @@ -540,6 +559,11 @@ export async function executeAction( }); return outcome.result; } catch (error) { + markGhcpEvalExecutionFailure(); + recordGhcpEvalEvent("action.failed", { + ...eventData, + elapsedMs: Date.now() - actionStartedAt, + }); logActionCompleted(systemContext.logger, { ...eventData, success: false, diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts new file mode 100644 index 0000000000..95271d487b --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts @@ -0,0 +1,87 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; + +let executionFailureObserved = false; + +export function markGhcpEvalExecutionFailure(): void { + if (process.env.TYPEAGENT_GHCP_EVAL_FIXTURES !== undefined) + executionFailureObserved = true; +} + +export function ghcpEvalExecutionStopped(): boolean { + return ( + process.env.TYPEAGENT_GHCP_EVAL_FIXTURES !== undefined && + executionFailureObserved + ); +} + +export function recordGhcpEvalEvent(event: string, detail?: unknown): void { + const file = process.env.TYPEAGENT_GHCP_EVAL_TRACE; + if (!file) return; + fs.appendFileSync( + file, + JSON.stringify({ + event, + detail, + processId: process.pid, + monotonicMs: performance.now(), + timestamp: new Date().toISOString(), + }) + "\n", + ); +} + +const reads = new Map>([ + ["github-cli", new Set(["prView", "prFiles", "prChecks", "issueView"])], + [ + "ipconfig", + new Set([ + "displayFullConfigurationInformation", + "displayDNSResolverCacheContents", + ]), + ], +]); + +/** Apply only to an explicitly isolated evaluation server, never normal sessions. */ +export function assertGhcpEvalAction( + schemaName: string, + actionName: string, + parameters: unknown, + fixtureRoot = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES, +): void { + if (fixtureRoot === undefined) return; + if (ghcpEvalExecutionStopped()) + throw new Error( + "GHCP eval stopped execution after a failed or cancelled action", + ); + if ( + schemaName === "list" || + schemaName === "dispatcher" || + schemaName.startsWith("dispatcher.") || + reads.get(schemaName)?.has(actionName) + ) { + return; + } + if ( + schemaName === "powershell.powershell-files" && + actionName === "readFile" && + typeof parameters === "object" && + parameters !== null && + "path" in parameters && + typeof parameters.path === "string" + ) { + const requested = fs.realpathSync(parameters.path).toLowerCase(); + const allowed = ["report-a.txt", "report-b.txt", "trip.txt"].map( + (file) => + fs.realpathSync(path.join(fixtureRoot, file)).toLowerCase(), + ); + if (allowed.includes(requested)) return; + } + recordGhcpEvalEvent("action.denied", { schemaName, actionName }); + markGhcpEvalExecutionFailure(); + throw new Error( + `GHCP eval policy denied ${schemaName}.${actionName} before execution`, + ); +} diff --git a/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts b/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts index 5323a022e1..3926f06265 100644 --- a/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts +++ b/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts @@ -95,6 +95,8 @@ import { pruneStaleCodingSessions, } from "./codingSessionLifecycle.js"; import { getCodingAttachmentPaths } from "./codingContext.js"; +import { getCopilotCreditBudget } from "./copilotCreditBudget.js"; +import { ghcpEvalExecutionStopped } from "../execute/ghcpEvalPolicy.js"; import { REASONING_DENY, getReasoningPermissionChoices, @@ -376,8 +378,14 @@ async function createCopilotClient( path.join(os.tmpdir(), "typeagent-copilot-"), ); + const creditBudget = getCopilotCreditBudget(); const client = new CopilotClient({ - connection: RuntimeConnection.forStdio(), + connection: RuntimeConnection.forStdio( + creditBudget && process.env.TYPEAGENT_GHCP_EVAL_CLI + ? { path: process.env.TYPEAGENT_GHCP_EVAL_CLI } + : undefined, + ), + ...(creditBudget ? { requestHandler: creditBudget } : {}), env: { ...process.env, CLAUDE_CONFIG_DIR: isolatedConfigDir, @@ -713,6 +721,11 @@ function createCopilotPermissionHandler( allowedRoot?: string, ): PermissionHandler { return async (request) => { + if (ghcpEvalExecutionStopped()) { + return { + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }; + } const agentContext = context.sessionContext.agentContext; const scopeViolation = getCopilotPermissionScopeViolation( request, @@ -2110,6 +2123,13 @@ function getCopilotSessionConfig( return { clientName: "TypeAgent", model, + ...(process.env.TYPEAGENT_COPILOT_CREDIT_LEDGER + ? { + capi: { enableWebSocketResponses: false }, + sessionLimits: { maxAiCredits: 60 }, + contextTier: "default" as const, + } + : {}), ...(reasoningEffort ? { reasoningEffort } : {}), streaming: true, tools: [ diff --git a/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts b/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts new file mode 100644 index 0000000000..9590d0b537 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts @@ -0,0 +1,303 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import { randomUUID } from "node:crypto"; +import { + CopilotRequestHandler, + type CopilotRequestContext, + type CopilotWebSocketHandler, +} from "@github/copilot-sdk"; + +export interface CreditReservation { + id: string; + sessionId?: string; + maximumNanoAiu: number; + settledNanoAiu?: number; + responseContentType?: string; +} + +export interface CopilotCreditLedger { + version: 1; + capNanoAiu: number; + openingNanoAiu: number; + headroomNanoAiu: number; + model: string; + requestMaximumNanoAiu: number; + reservations: CreditReservation[]; + blockedReason?: string; +} + +function requireAmount(value: number, name: string): void { + if (!Number.isSafeInteger(value) || value < 0) { + throw new Error(`Invalid credit ledger ${name}`); + } +} + +export function validateCreditLedger(ledger: CopilotCreditLedger): void { + if ( + ledger.version !== 1 || + typeof ledger.model !== "string" || + ledger.model.length === 0 || + !Array.isArray(ledger.reservations) + ) { + throw new Error("Invalid credit ledger structure"); + } + for (const name of [ + "capNanoAiu", + "openingNanoAiu", + "headroomNanoAiu", + "requestMaximumNanoAiu", + ] as const) { + requireAmount(ledger[name], name); + } + if (ledger.capNanoAiu > 20_000_000_000_000) { + throw new Error("Credit ledger exceeds the 20,000-credit ceiling"); + } + if (ledger.requestMaximumNanoAiu === 0) { + throw new Error("A positive request reservation is required"); + } + const ids = new Set(); + for (const entry of ledger.reservations) { + if (typeof entry.id !== "string" || ids.has(entry.id)) { + throw new Error("Invalid or duplicate credit reservation id"); + } + ids.add(entry.id); + requireAmount(entry.maximumNanoAiu, "reservation maximum"); + if (entry.settledNanoAiu !== undefined) { + requireAmount(entry.settledNanoAiu, "settled usage"); + if (entry.settledNanoAiu > entry.maximumNanoAiu) { + throw new Error("Observed usage exceeded the reserved bound"); + } + } + } +} + +export function accountedNanoAiu(ledger: CopilotCreditLedger): number { + validateCreditLedger(ledger); + const total = ledger.reservations.reduce( + (sum, entry) => sum + (entry.settledNanoAiu ?? entry.maximumNanoAiu), + ledger.openingNanoAiu + ledger.headroomNanoAiu, + ); + requireAmount(total, "accounted total"); + return total; +} + +export function reserveCredits( + ledger: CopilotCreditLedger, + id: string, + sessionId?: string, +): void { + if (ledger.blockedReason) { + throw new Error( + `Copilot credit accounting blocked: ${ledger.blockedReason}`, + ); + } + if ( + ledger.reservations.length >= 2_000 || + (sessionId !== undefined && + ledger.reservations.filter((entry) => entry.sessionId === sessionId) + .length >= 24) + ) { + throw new Error( + "Copilot credit request-count limit reached; request not sent", + ); + } + const total = accountedNanoAiu(ledger) + ledger.requestMaximumNanoAiu; + if (!Number.isSafeInteger(total) || total > ledger.capNanoAiu) { + throw new Error("Copilot credit budget exhausted; request not sent"); + } + if (ledger.reservations.some((entry) => entry.id === id)) { + throw new Error("Credit reservation already exists"); + } + ledger.reservations.push({ + id, + ...(sessionId === undefined ? {} : { sessionId }), + maximumNanoAiu: ledger.requestMaximumNanoAiu, + }); +} + +function updateLedger( + file: string, + update: (ledger: CopilotCreditLedger) => void, +): void { + // Exclusive creation makes concurrent workers fail closed rather than + // admitting against the same balance. A stale lock requires reconciliation. + const lock = `${file}.lock`; + const lockFd = fs.openSync(lock, "wx"); + const temporary = `${file}.${randomUUID()}.tmp`; + try { + const ledger: CopilotCreditLedger = JSON.parse( + fs.readFileSync(file, "utf8"), + ); + validateCreditLedger(ledger); + update(ledger); + validateCreditLedger(ledger); + const fd = fs.openSync(temporary, "wx"); + try { + fs.writeFileSync(fd, JSON.stringify(ledger, null, 2) + "\n"); + fs.fsyncSync(fd); + } finally { + fs.closeSync(fd); + } + fs.renameSync(temporary, file); + } finally { + if (fs.existsSync(temporary)) fs.unlinkSync(temporary); + fs.closeSync(lockFd); + fs.unlinkSync(lock); + } +} + +export function extractNanoAiu(value: unknown): number | undefined { + if (typeof value !== "object" || value === null) return undefined; + const object = value as Record; + const usage = object.copilot_usage ?? object.copilotUsage; + if (typeof usage === "object" && usage !== null) { + const fields = usage as Record; + const amount = fields.total_nano_aiu ?? fields.totalNanoAiu; + if (typeof amount === "number") { + requireAmount(amount, "response usage"); + return amount; + } + } + return ( + (object.response === undefined + ? undefined + : extractNanoAiu(object.response)) ?? + (object.usage === undefined ? undefined : extractNanoAiu(object.usage)) + ); +} + +function settleCredits(file: string, id: string, amount: number): void { + let exceeded = false; + updateLedger(file, (ledger) => { + const entry = ledger.reservations.find((entry) => entry.id === id); + if (!entry) throw new Error("Missing credit reservation"); + if (amount > entry.maximumNanoAiu) { + ledger.blockedReason = `Observed usage ${amount} exceeded reserved bound ${entry.maximumNanoAiu}`; + entry.maximumNanoAiu = amount; + exceeded = true; + } + entry.settledNanoAiu = amount; + }); + if (exceeded) + throw new Error( + "Observed usage exceeded the reserved bound; accounting blocked", + ); +} + +export class CopilotCreditBudget extends CopilotRequestHandler { + constructor( + private readonly ledgerPath: string, + private readonly sessionScope?: string, + ) { + super(); + } + + protected override async openWebSocket( + _context: CopilotRequestContext, + ): Promise { + throw new Error( + "Credit-controlled sessions require capi.enableWebSocketResponses=false", + ); + } + + protected override async sendRequest( + request: Request, + context: CopilotRequestContext, + ): Promise { + const body: unknown = await request.clone().json(); + if (typeof body !== "object" || body === null || !("model" in body)) { + throw new Error("Credit-controlled request has no model identity"); + } + const id = randomUUID(); + updateLedger(this.ledgerPath, (ledger) => { + if (body.model !== ledger.model) { + throw new Error("Credit-controlled request model mismatch"); + } + reserveCredits( + ledger, + id, + this.sessionScope + ? `${this.sessionScope}::${context.sessionId}` + : context.sessionId, + ); + }); + // Failed, cancelled and unrecognized responses keep the + // full reservation. Never infer that a failed request was free. + const response = await super.sendRequest(request, context); + const contentType = response.headers.get("content-type") ?? ""; + updateLedger(this.ledgerPath, (ledger) => { + const entry = ledger.reservations.find((entry) => entry.id === id); + if (!entry) throw new Error("Missing credit reservation"); + entry.responseContentType = contentType; + }); + if (response.ok && contentType.includes("application/json")) { + const amount = extractNanoAiu(await response.clone().json()); + if (amount !== undefined) { + settleCredits(this.ledgerPath, id, amount); + } + return response; + } + if ( + !response.ok || + !response.body || + !contentType.includes("text/event-stream") + ) { + return response; + } + const decoder = new TextDecoder(); + let pending = ""; + let observed: number | undefined; + const file = this.ledgerPath; + return new Response( + response.body.pipeThrough( + new TransformStream({ + transform(chunk, controller) { + pending += decoder.decode(chunk, { stream: true }); + if (pending.length > 8 * 1024 * 1024) { + throw new Error( + "Credit usage stream line too large", + ); + } + const lines = pending.split("\n"); + pending = lines.pop()!; + for (const line of lines) { + if (!line.startsWith("data:")) continue; + const text = line.slice(5).trim(); + if (text === "[DONE]" || text.length === 0) + continue; + const amount = extractNanoAiu(JSON.parse(text)); + if (amount !== undefined) { + observed = Math.max(observed ?? 0, amount); + } + } + controller.enqueue(chunk); + }, + flush() { + if (observed === undefined || pending.trim() !== "") { + return; + } + const settled = observed; + settleCredits(file, id, settled); + }, + }), + ), + { + status: response.status, + statusText: response.statusText, + headers: response.headers, + }, + ); + } +} + +export function getCopilotCreditBudget(): CopilotCreditBudget | undefined { + const file = process.env.TYPEAGENT_COPILOT_CREDIT_LEDGER; + return file + ? new CopilotCreditBudget( + file, + process.env.TYPEAGENT_COPILOT_CREDIT_SESSION_SCOPE, + ) + : undefined; +} diff --git a/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts b/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts new file mode 100644 index 0000000000..09f74210c0 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts @@ -0,0 +1,230 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import { + accountedNanoAiu, + CopilotCreditBudget, + extractNanoAiu, + reserveCredits, + validateCreditLedger, + type CopilotCreditLedger, +} from "../src/reasoning/copilotCreditBudget.js"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import type { CopilotRequestContext } from "@github/copilot-sdk"; +import { jest } from "@jest/globals"; + +function ledger(): CopilotCreditLedger { + return { + version: 1, + capNanoAiu: 20_000, + openingNanoAiu: 3_000, + headroomNanoAiu: 8_000, + model: "test-model", + requestMaximumNanoAiu: 2_000, + reservations: [], + }; +} + +describe("Copilot credit admission", () => { + it("accounts for prior sessions, headroom and unresolved requests", () => { + const state = ledger(); + reserveCredits(state, "first"); + reserveCredits(state, "second"); + expect(accountedNanoAiu(state)).toBe(15_000); + state.reservations[0].settledNanoAiu = 25; + expect(accountedNanoAiu(state)).toBe(13_025); + }); + + class TestBudget extends CopilotCreditBudget { + send(model = "test-model") { + const request = new Request("https://example.invalid/responses", { + method: "POST", + body: JSON.stringify({ model }), + }); + const context: CopilotRequestContext = { + requestId: "test", + sessionId: "session", + transport: "http", + url: request.url, + headers: {}, + signal: new AbortController().signal, + }; + return this.sendRequest(request, context); + } + } + + describe("Copilot outbound request guard", () => { + let directory: string; + let file: string; + let budget: TestBudget; + beforeEach(() => { + directory = fs.mkdtempSync( + path.join(os.tmpdir(), "copilot-credit-test-"), + ); + file = path.join(directory, "ledger.json"); + fs.writeFileSync(file, JSON.stringify(ledger())); + budget = new TestBudget(file); + }); + afterEach(() => { + jest.restoreAllMocks(); + fs.rmSync(directory, { recursive: true }); + }); + const read = (file: string): CopilotCreditLedger => + JSON.parse(fs.readFileSync(file, "utf8")); + + it("persists admission before forwarding and settles explicit JSON usage", async () => { + const fetch = jest + .spyOn(globalThis, "fetch") + .mockImplementation(async () => { + expect(read(file).reservations).toHaveLength(1); + return Response.json({ + copilot_usage: { total_nano_aiu: 25 }, + }); + }); + await budget.send(); + expect(fetch).toHaveBeenCalledTimes(1); + expect(accountedNanoAiu(read(file))).toBe(11_025); + }); + + it("does not send unknown models, exhausted budgets, or concurrent admissions", async () => { + const fetch = jest.spyOn(globalThis, "fetch"); + await expect(budget.send("different-model")).rejects.toThrow( + "model mismatch", + ); + fs.writeFileSync(`${file}.lock`, ""); + await expect(budget.send()).rejects.toThrow(); + fs.unlinkSync(`${file}.lock`); + const state = ledger(); + state.openingNanoAiu = 12_000; + fs.writeFileSync(file, JSON.stringify(state)); + await expect(budget.send()).rejects.toThrow("request not sent"); + expect(fetch).not.toHaveBeenCalled(); + }); + + it("retains the reservation after transport failure or missing usage", async () => { + const fetch = jest + .spyOn(globalThis, "fetch") + .mockRejectedValueOnce(new Error("connection failed")) + .mockResolvedValueOnce(Response.json({ output: "OK" })); + await expect(budget.send()).rejects.toThrow("connection failed"); + await budget.send(); + expect(fetch).toHaveBeenCalledTimes(2); + expect(accountedNanoAiu(read(file))).toBe(15_000); + }); + + it("persists unexpected excess billing and blocks future admissions", async () => { + const fetch = jest + .spyOn(globalThis, "fetch") + .mockResolvedValue( + Response.json({ copilot_usage: { total_nano_aiu: 2_001 } }), + ); + await expect(budget.send()).rejects.toThrow("accounting blocked"); + expect(read(file).reservations[0].settledNanoAiu).toBe(2_001); + expect(read(file).blockedReason).toMatch(/exceeded/); + await expect(budget.send()).rejects.toThrow("accounting blocked"); + expect(fetch).toHaveBeenCalledTimes(1); + }); + + it("preserves SSE bytes and settles usage split across chunks", async () => { + const body = + 'data: {"response":{"copilot_usage":{"total_nano_aiu":42}}}\n\ndata: [DONE]\n\n'; + const encoder = new TextEncoder(); + jest.spyOn(globalThis, "fetch").mockResolvedValue( + new Response( + new ReadableStream({ + start(controller) { + controller.enqueue( + encoder.encode(body.slice(0, 20)), + ); + controller.enqueue(encoder.encode(body.slice(20))); + controller.close(); + }, + }), + { headers: { "content-type": "text/event-stream" } }, + ), + ); + expect(await (await budget.send()).text()).toBe(body); + expect(accountedNanoAiu(read(file))).toBe(11_042); + }); + }); + + it("rejects the next request without modifying the ledger", () => { + const state = ledger(); + for (let i = 0; i < 4; i++) reserveCredits(state, String(i)); + const before = JSON.stringify(state); + expect(() => reserveCredits(state, "fifth")).toThrow( + "request not sent", + ); + expect(JSON.stringify(state)).toBe(before); + }); + + it("admits an exact remaining reservation", () => { + const state = ledger(); + state.requestMaximumNanoAiu = 9_000; + reserveCredits(state, "exact"); + expect(accountedNanoAiu(state)).toBe(state.capNanoAiu); + }); + + it("bounds settled calls per session even when each call bills zero", () => { + const state = ledger(); + for (let i = 0; i < 24; i++) { + reserveCredits(state, String(i), "bounded-session"); + state.reservations[i].settledNanoAiu = 0; + } + expect(() => reserveCredits(state, "extra", "bounded-session")).toThrow( + "request-count limit", + ); + expect(state.reservations).toHaveLength(24); + }); + + it("does not reset or silently duplicate a reservation", () => { + const state = ledger(); + reserveCredits(state, "same"); + expect(() => reserveCredits(state, "same")).toThrow("already exists"); + expect(state.reservations).toHaveLength(1); + }); + + it.each([NaN, Infinity, -1, 0.5, Number.MAX_SAFE_INTEGER + 1])( + "rejects invalid accounting values: %s", + (value) => { + const state = ledger(); + state.openingNanoAiu = value; + expect(() => validateCreditLedger(state)).toThrow( + "Invalid credit ledger", + ); + }, + ); + + it("rejects cap increases and unbounded requests", () => { + const state = ledger(); + state.capNanoAiu = 20_000_000_000_001; + expect(() => validateCreditLedger(state)).toThrow("ceiling"); + state.capNanoAiu = 20_000; + state.requestMaximumNanoAiu = 0; + expect(() => validateCreditLedger(state)).toThrow("positive"); + }); + + it("fails closed if observed usage exceeds its reservation", () => { + const state = ledger(); + reserveCredits(state, "request"); + state.reservations[0].settledNanoAiu = 2_001; + expect(() => accountedNanoAiu(state)).toThrow("exceeded"); + }); + + it("reads only explicit Copilot billing fields, including zero", () => { + expect(extractNanoAiu({ usage: { total_tokens: 50 } })).toBeUndefined(); + expect(extractNanoAiu({ copilot_usage: { total_nano_aiu: 0 } })).toBe( + 0, + ); + expect( + extractNanoAiu({ + response: { copilot_usage: { totalNanoAiu: 123 } }, + }), + ).toBe(123); + expect(() => + extractNanoAiu({ copilot_usage: { total_nano_aiu: -1 } }), + ).toThrow("response usage"); + }); +}); diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts new file mode 100644 index 0000000000..76e9080ed9 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalPolicy.spec.ts @@ -0,0 +1,92 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { + assertGhcpEvalAction, + ghcpEvalExecutionStopped, + markGhcpEvalExecutionFailure, +} from "../src/execute/ghcpEvalPolicy.js"; + +describe("isolated GHCP evaluation action policy", () => { + it("stops subsequent execution after failure only in the isolated eval process", () => { + const original = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + try { + process.env.TYPEAGENT_GHCP_EVAL_FIXTURES = "fixture"; + markGhcpEvalExecutionFailure(); + expect(ghcpEvalExecutionStopped()).toBe(true); + expect(() => assertGhcpEvalAction("list", "addItems", {})).toThrow( + "stopped execution", + ); + delete process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + expect(ghcpEvalExecutionStopped()).toBe(false); + expect(() => + assertGhcpEvalAction("any", "normal", {}), + ).not.toThrow(); + } finally { + if (original === undefined) + delete process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + else process.env.TYPEAGENT_GHCP_EVAL_FIXTURES = original; + } + }); + it.each([ + ["github-cli", "issueClose"], + ["ipconfig", "releaseAddress"], + ["powershell", "executeScript"], + ])("blocks %s.%s before effects", (schema, action) => { + expect(() => + assertGhcpEvalAction(schema, action, {}, "fixture"), + ).toThrow("before execution"); + }); + it("permits the intended read-only external actions and disposable lists", () => { + expect(() => + assertGhcpEvalAction("github-cli", "prFiles", {}, "fixture"), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction( + "ipconfig", + "displayDNSResolverCacheContents", + {}, + "fixture", + ), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction("list", "clearList", {}, "fixture"), + ).not.toThrow(); + }); + it("permits only canonical registered fixture paths", () => { + const folder = fs.mkdtempSync( + path.join(os.tmpdir(), "ghcp-policy-test-"), + ); + try { + for (const file of [ + "report-a.txt", + "report-b.txt", + "trip.txt", + "unrelated.txt", + ]) { + fs.writeFileSync(path.join(folder, file), ""); + } + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "readFile", + { path: path.join(folder, "report-a.txt") }, + folder, + ), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "readFile", + { path: path.join(folder, "unrelated.txt") }, + folder, + ), + ).toThrow("before execution"); + } finally { + fs.rmSync(folder, { recursive: true }); + } + }); +}); diff --git a/ts/packages/dispatcher/types/src/dispatcher.ts b/ts/packages/dispatcher/types/src/dispatcher.ts index f9b50bbb1c..ba24d13e04 100644 --- a/ts/packages/dispatcher/types/src/dispatcher.ts +++ b/ts/packages/dispatcher/types/src/dispatcher.ts @@ -368,6 +368,11 @@ export type ProcessCommandOptions = { * and TypeAgent should act as a pure action executor. */ noReasoning?: boolean; + /** + * Control only the unknown/clarification translation transition to + * reasoning. Unlike noReasoning, this does not disable explicit reasoning. + */ + translationReasoningFallback?: boolean; /** * Restrict translation and grammar matching to this subset of currently * active schemas. The request returns notHandled when any requested schema From bb7c863f82b7b7048ccde7e4a1e882ba7a064f4e Mon Sep 17 00:00:00 2001 From: George Ng Date: Thu, 24 Sep 2026 03:13:13 -0700 Subject: [PATCH 02/14] Separate eval setup and grading and satisfy repository ratchets Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin/README.md | 6 + .../scripts/ghcp-credit-probe.mjs | 2 +- .../scripts/ghcp-eval-preflight.mjs | 4 +- .../copilot-plugin/scripts/ghcp-eval.mjs | 206 +++++++++--------- 4 files changed, 114 insertions(+), 104 deletions(-) diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index daf6080de3..1f2e85d4d7 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -73,6 +73,12 @@ The four domain agents are lists, GitHub CLI, registered PowerShell file actions, and IP configuration. Fixtures are restored per trial. Shipped MCP domain schemas outside that scope are disabled only in the disposable session. The production internal reasoning toolset is unchanged. The +mixed candidates keep production routing guidance and the pinned native tools. +Auxiliary outer workspace/macro/skill MCP servers are omitted to keep the +declared entry interfaces in scope. After failed or uncertain execution, +the eval policy blocks replay (including internal error-triggered retries); +it does not repair the underlying product failure or substitute an action. +These controls are identical across the relevant candidate pairs. The `translationReasoningFallback` request option controls only the existing unknown/clarification translation-to-reasoning transition, not ordinary orchestration. Optional `TYPEAGENT_GHCP_EVAL_TRACE` records its actual decision, diff --git a/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs b/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs index 82cbe97e54..dd3eaf4198 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs @@ -73,5 +73,5 @@ try { path.join(outputDirectory, "result.json"), JSON.stringify(result, null, 2) + "\n", ); - console.log(JSON.stringify(result)); + process.stdout.write(JSON.stringify(result) + "\n"); } diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs index 3636350974..dd0fbb72b3 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs @@ -262,12 +262,12 @@ try { JSON.stringify(result, null, 2) + "\n", ); } - console.log( + process.stdout.write( JSON.stringify({ status: result.status, contractCount: result.contracts.length, missing: result.missing, error: result.error, - }), + }) + "\n", ); } diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs index 1752dc5d4e..02cc2bdb44 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs @@ -47,6 +47,8 @@ const [ outputDirectory, configDirectory, ledgerPath, + templateDirectory: path.resolve(template), + fixtureReset: "Copy the same catalog-only state, exclude stale lock directories, restore all seven lists and three files per trial", selection = "1,2,3,4,5,6,7", phase = "pilot", evidencePath, @@ -201,7 +203,72 @@ function collectObservations(result, tracePath) { }; } -async function trial(candidate, directory, testCase, workspace, evidence) { +function captureNetworkEvidence(includeDns) { + return { + capturedAt: new Date().toISOString(), + configuration: execFileSync("ipconfig.exe", ["/all"], { + encoding: "utf8", + timeout: 15_000, + }), + dns: includeDns + ? execFileSync("ipconfig.exe", ["/displaydns"], { + encoding: "utf8", + timeout: 15_000, + }) + : null, + }; +} + +function gradeCompletedTrial({ + result, + testCase, + store, + workspace, + evidence, + clarificationGiven, +}) { + const after = JSON.parse(fs.readFileSync(store, "utf8")); + const correctNames = Object.keys(fixtures).every((name) => + new RegExp(`\\b${name}\\b`, "i").test(result.answer), + ); + const expected = expectedLists(testCase.id, 2617, evidence?.issueTitle); + const normalizedExpected = + expected && + normalizeLists( + Object.entries(expected).map(([name, items]) => ({ name, items })), + ); + result.grade = { + listStateMatchesOracle: normalizedExpected + ? JSON.stringify(normalizeLists(after)) === + JSON.stringify(normalizedExpected) + : null, + filesUnchanged: Object.entries(fileFixture).every( + ([name, content]) => + fs.readFileSync(path.join(workspace, name), "utf8") === content, + ), + containsAllNames: testCase.id === "S1" ? correctNames : null, + clarificationRequested: testCase.clarification + ? clarificationGiven + : null, + noPrematureListMutation: testCase.clarification + ? JSON.stringify(result.stateAtClarification) === + JSON.stringify(normalizeLists(seededLists)) + : null, + requiresManualFaithfulnessCheck: true, + }; + result.finalLists = normalizeLists(after); + result.status = "completed_ungraded"; + if ( + phase === "pilot" && + result.candidate !== 7 && + ((testCase.id === "S1" && !correctNames) || + result.grade.listStateMatchesOracle === false || + !result.grade.filesUnchanged) + ) + result.status = "pilot_needs_review"; +} + +function prepareTrial(candidate, directory, workspace) { fs.mkdirSync(directory); const { env, mcp } = makeConfiguration( directory, @@ -275,6 +342,26 @@ async function trial(candidate, directory, testCase, workspace, evidence) { config.env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK = env.TYPEAGENT_TRANSLATION_REASONING_FALLBACK; } + return { env, config, stores, sessionDataPath, sessionData, sessionId }; +} + +function persistTrial({ result, env, executionStopped, sessionData, sessionDataPath, evidence, network, directory, started }) { + result.totalIncludingSetupMs = performance.now() - started; + collectObservations(result, env.TYPEAGENT_GHCP_EVAL_TRACE); + result.terminalExecutionFailure = executionStopped; + result.providerUsage = { + before: sessionData.tokens ?? null, + after: JSON.parse(fs.readFileSync(sessionDataPath, "utf8")).tokens ?? null, + coverage: "Persisted TypeAgent token counters only; unflushed calls and embedding usage may be absent. Not Copilot credits.", + }; + result.preliminaryGrade = preliminaryGrade(result, evidence ?? {}); + if (network) fs.writeFileSync(path.join(directory, "private-network-evidence.json"), JSON.stringify(network, null, 2)); + fs.writeFileSync(path.join(directory, "result.json"), JSON.stringify(result, null, 2) + "\n"); +} + +async function trial(candidate, directory, testCase, workspace, evidence) { + const { env, config, stores, sessionDataPath, sessionData, sessionId } = + prepareTrial(candidate, directory, workspace); const result = { phase, candidate: candidate.id, @@ -335,7 +422,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { await waitForServer( server, port, - 30, + 90, new AbortController().signal, stderrPath, ); @@ -547,20 +634,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { preparation = false; } if (network) { - network.before = { - capturedAt: new Date().toISOString(), - configuration: execFileSync("ipconfig.exe", ["/all"], { - encoding: "utf8", - timeout: 15_000, - }), - dns: - testCase.id === "M2" - ? execFileSync("ipconfig.exe", ["/displaydns"], { - encoding: "utf8", - timeout: 15_000, - }) - : null, - }; + network.before = captureNetworkEvidence(testCase.id === "M2"); } measuredStart = performance.now(); result.promptAcceptedAt = new Date().toISOString(); @@ -571,72 +645,23 @@ async function trial(candidate, directory, testCase, workspace, evidence) { result.e2eMs = performance.now() - measuredStart; result.finalResponseAt = new Date().toISOString(); result.answer = answer?.data.content ?? ""; - const after = JSON.parse(fs.readFileSync(stores[0], "utf8")); - const correctNames = Object.keys(fixtures).every((name) => - new RegExp(`\\b${name}\\b`, "i").test(result.answer), - ); - const expected = expectedLists(testCase.id, 2617, evidence?.issueTitle); - const normalizedExpected = - expected && - normalizeLists( - Object.entries(expected).map(([name, items]) => ({ - name, - items, - })), - ); - result.grade = { - listStateMatchesOracle: normalizedExpected - ? JSON.stringify(normalizeLists(after)) === - JSON.stringify(normalizedExpected) - : null, - filesUnchanged: Object.entries(fileFixture).every( - ([name, content]) => - fs.readFileSync(path.join(workspace, name), "utf8") === - content, - ), - containsAllNames: testCase.id === "S1" ? correctNames : null, - clarificationRequested: testCase.clarification - ? clarificationGiven - : null, - noPrematureListMutation: testCase.clarification - ? JSON.stringify(result.stateAtClarification) === - JSON.stringify(normalizeLists(seededLists)) - : null, - requiresManualFaithfulnessCheck: true, - }; - result.finalLists = normalizeLists(after); + gradeCompletedTrial({ + result, + testCase, + store: stores[0], + workspace, + evidence, + clarificationGiven, + }); if (testCase.id === "S5" || testCase.id === "M2") { network.answer = result.answer; - network.after = { - capturedAt: new Date().toISOString(), - configuration: execFileSync("ipconfig.exe", ["/all"], { - encoding: "utf8", - timeout: 15_000, - }), - dns: - testCase.id === "M2" - ? execFileSync("ipconfig.exe", ["/displaydns"], { - encoding: "utf8", - timeout: 15_000, - }) - : null, - }; + network.after = captureNetworkEvidence(testCase.id === "M2"); result.answerSha256 = createHash("sha256") .update(result.answer) .digest("hex"); result.answer = "[network response withheld from sanitized results]"; } - result.status = "completed_ungraded"; - if ( - phase === "pilot" && - candidate.id !== 7 && - ((testCase.id === "S1" && !correctNames) || - result.grade.listStateMatchesOracle === false || - !result.grade.filesUnchanged) - ) { - result.status = "pilot_needs_review"; - } } catch (error) { result.status = "failed"; result.error = error instanceof Error ? error.message : String(error); @@ -650,30 +675,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { try { if (server) await stopProcess(server); } finally { - result.totalIncludingSetupMs = performance.now() - started; - collectObservations(result, env.TYPEAGENT_GHCP_EVAL_TRACE); - result.terminalExecutionFailure = executionStopped; - result.providerUsage = { - before: sessionData.tokens ?? null, - after: - JSON.parse(fs.readFileSync(sessionDataPath, "utf8")) - .tokens ?? null, - coverage: - "Persisted TypeAgent token counters only; unflushed calls and embedding usage may be absent. Not Copilot credits.", - }; - result.preliminaryGrade = preliminaryGrade( - result, - evidence ?? {}, - ); - if (network) - fs.writeFileSync( - path.join(directory, "private-network-evidence.json"), - JSON.stringify(network, null, 2), - ); - fs.writeFileSync( - path.join(directory, "result.json"), - JSON.stringify(result, null, 2) + "\n", - ); + persistTrial({ result, env, executionStopped, sessionData, sessionDataPath, evidence, network, directory, started }); } } } @@ -790,6 +792,8 @@ const specification = internalTools: "Unmodified production TypeAgent reasoning toolset; effects and credits gated", disabledShippedMcpSchemas: excludedSchemas, + disabledAuxiliaryOuterMcpServers: ["typeagent-workspace", "typeagent-macros", "typeagent-skills"], + safety: "Normal confirmation retained; no replay after failed/denied/cancelled/uncertain execution, including internal error-triggered retries. Translation fallback toolset retained.", }, null, 2, @@ -820,18 +824,18 @@ for (const entry of order.slice(batchStart, batchStart + batchSize)) { path.join(outputDirectory, "results.json"), JSON.stringify(results, null, 2) + "\n", ); - console.log( + process.stdout.write( JSON.stringify({ candidate: candidate.id, caseId: testCase.id, repetition: entry.repetition, status: result.status, error: result.error, - }), + }) + "\n", ); if ( (phase === "pilot" && result.status !== "completed_ungraded") || - /credit|budget|reservation|Agent server exited|permission orchestrator/i.test( + /credit|budget|reservation|Agent server|permission orchestrator/i.test( result.error ?? "", ) ) { From f1c332b68ffc2de1d75b03b0103f30d20d0fb429 Mon Sep 17 00:00:00 2001 From: George Ng Date: Thu, 24 Sep 2026 03:13:45 -0700 Subject: [PATCH 03/14] Correct frozen run metadata placement Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../copilot-plugin/scripts/ghcp-eval.mjs | 52 +++++++++++++++---- 1 file changed, 43 insertions(+), 9 deletions(-) diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs index 02cc2bdb44..ec8a876cb2 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs @@ -47,8 +47,6 @@ const [ outputDirectory, configDirectory, ledgerPath, - templateDirectory: path.resolve(template), - fixtureReset: "Copy the same catalog-only state, exclude stale lock directories, restore all seven lists and three files per trial", selection = "1,2,3,4,5,6,7", phase = "pilot", evidencePath, @@ -345,18 +343,37 @@ function prepareTrial(candidate, directory, workspace) { return { env, config, stores, sessionDataPath, sessionData, sessionId }; } -function persistTrial({ result, env, executionStopped, sessionData, sessionDataPath, evidence, network, directory, started }) { +function persistTrial({ + result, + env, + executionStopped, + sessionData, + sessionDataPath, + evidence, + network, + directory, + started, +}) { result.totalIncludingSetupMs = performance.now() - started; collectObservations(result, env.TYPEAGENT_GHCP_EVAL_TRACE); result.terminalExecutionFailure = executionStopped; result.providerUsage = { before: sessionData.tokens ?? null, - after: JSON.parse(fs.readFileSync(sessionDataPath, "utf8")).tokens ?? null, - coverage: "Persisted TypeAgent token counters only; unflushed calls and embedding usage may be absent. Not Copilot credits.", + after: + JSON.parse(fs.readFileSync(sessionDataPath, "utf8")).tokens ?? null, + coverage: + "Persisted TypeAgent token counters only; unflushed calls and embedding usage may be absent. Not Copilot credits.", }; result.preliminaryGrade = preliminaryGrade(result, evidence ?? {}); - if (network) fs.writeFileSync(path.join(directory, "private-network-evidence.json"), JSON.stringify(network, null, 2)); - fs.writeFileSync(path.join(directory, "result.json"), JSON.stringify(result, null, 2) + "\n"); + if (network) + fs.writeFileSync( + path.join(directory, "private-network-evidence.json"), + JSON.stringify(network, null, 2), + ); + fs.writeFileSync( + path.join(directory, "result.json"), + JSON.stringify(result, null, 2) + "\n", + ); } async function trial(candidate, directory, testCase, workspace, evidence) { @@ -675,7 +692,17 @@ async function trial(candidate, directory, testCase, workspace, evidence) { try { if (server) await stopProcess(server); } finally { - persistTrial({ result, env, executionStopped, sessionData, sessionDataPath, evidence, network, directory, started }); + persistTrial({ + result, + env, + executionStopped, + sessionData, + sessionDataPath, + evidence, + network, + directory, + started, + }); } } } @@ -783,6 +810,9 @@ const specification = requestCreditReservation: 2118, sessionCreditSoftLimit: 60, ledgerPath, + templateDirectory: path.resolve(template), + fixtureReset: + "Copy catalog-only state, exclude stale locks, restore seven lists and three files per trial", gradingStatus: "independent fixture oracles; explicit final-answer review required", nativeTools, @@ -792,7 +822,11 @@ const specification = internalTools: "Unmodified production TypeAgent reasoning toolset; effects and credits gated", disabledShippedMcpSchemas: excludedSchemas, - disabledAuxiliaryOuterMcpServers: ["typeagent-workspace", "typeagent-macros", "typeagent-skills"], + disabledAuxiliaryOuterMcpServers: [ + "typeagent-workspace", + "typeagent-macros", + "typeagent-skills", + ], safety: "Normal confirmation retained; no replay after failed/denied/cancelled/uncertain execution, including internal error-triggered retries. Translation fallback toolset retained.", }, null, From 73fc67fa2aa455f39867410120b76e88f75593c4 Mon Sep 17 00:00:00 2001 From: George Ng Date: Thu, 24 Sep 2026 03:16:46 -0700 Subject: [PATCH 04/14] Explicitly pin mixed routing in isolated evaluation profiles Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../copilot-plugin/scripts/ghcp-eval.mjs | 23 ++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs index ec8a876cb2..471defa97f 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs @@ -293,6 +293,22 @@ function prepareTrial(candidate, directory, workspace) { fs.cpSync(path.join(template, "plugin-data"), env.TYPEAGENT_PLUGIN_DATA, { recursive: true, }); + const routingPath = path.join(env.TYPEAGENT_PLUGIN_DATA, "config.json"); + const routingConfig = fs.existsSync(routingPath) + ? JSON.parse(fs.readFileSync(routingPath, "utf8")) + : {}; + fs.writeFileSync( + routingPath, + JSON.stringify( + { + ...routingConfig, + mode: "mcp", + mcpRouting: candidate.id >= 5 ? "mixed" : "delegate", + }, + null, + 2, + ), + ); const stores = findListStores(env.TYPEAGENT_USER_DATA_DIR); if (stores.length !== 1) throw new Error("Expected exactly one disposable list store"); @@ -453,7 +469,12 @@ async function trial(candidate, directory, testCase, workspace, evidence) { candidate.id === 7 ? Object.fromEntries( Object.entries(env).filter( - ([key]) => !key.startsWith("TYPEAGENT_"), + ([key]) => + !key.startsWith("TYPEAGENT_") && + ![ + "CLAUDE_PLUGIN_DATA", + "INSTANCE_NAME", + ].includes(key), ), ) : env, From fc604fa56d70b1369be2bca1e8a3204855859fd4 Mon Sep 17 00:00:00 2001 From: George Ng Date: Thu, 24 Sep 2026 04:21:18 -0700 Subject: [PATCH 05/14] Document measured GHCP evaluation blockers Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin/README.md | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index 1f2e85d4d7..14cd41f101 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -91,6 +91,17 @@ sanitized results contain hashes. Unobserved internal stage durations are null, not zero. Keep pilot/harness failures separate from measured outcomes, and do not pool fast refusals with successful-completion latency. +**Live-run limitations:** the initial measured pass was paused, not completed. +The fixture-only confirmation policy does not authorize SDK-created oversized +tool-output files or intermediate `list.startEditList` interactions. The runner +also answers clarification callbacks but does not continue a conversation when +a candidate returns its clarification as final-response text. These are harness +limitations, not evidence of inferior candidate accuracy. Do not use affected +paired cases for comparative scores or resume paid execution until these paths +have safe, scoped support and representative live validation. A correction +requires a newly frozen protocol; preserve the original results and cumulative +credit ledger rather than rewriting or selectively replaying them. + ## Structured actions in Direct and MCP modes ### One-command discovery E2E session (Windows) From e847c7a00907a1f4e2c6c4426c935b964fd9997c Mon Sep 17 00:00:00 2001 From: George Ng Date: Thu, 24 Sep 2026 11:18:16 -0700 Subject: [PATCH 06/14] Repair scoped eval artifacts and scripted clarification continuation Preserve original trials and enforce the user-amended cumulative 50000-credit ceiling. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin/README.md | 29 +++-- .../scripts/ghcp-eval-corpus.mjs | 37 ++++++ .../copilot-plugin/scripts/ghcp-eval.mjs | 81 ++++++++++-- .../scripts/test/ghcp-eval.spec.mjs | 71 +++++++++++ .../src/execute/ghcpEvalArtifacts.ts | 116 ++++++++++++++++++ .../dispatcher/src/execute/ghcpEvalPolicy.ts | 4 +- .../src/reasoning/copilotCreditBudget.ts | 4 +- .../test/copilotCreditBudget.spec.ts | 6 +- .../dispatcher/test/ghcpEvalArtifacts.spec.ts | 101 +++++++++++++++ 9 files changed, 422 insertions(+), 27 deletions(-) create mode 100644 ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts create mode 100644 ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index 14cd41f101..2a19e723b3 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -91,16 +91,25 @@ sanitized results contain hashes. Unobserved internal stage durations are null, not zero. Keep pilot/harness failures separate from measured outcomes, and do not pool fast refusals with successful-completion latency. -**Live-run limitations:** the initial measured pass was paused, not completed. -The fixture-only confirmation policy does not authorize SDK-created oversized -tool-output files or intermediate `list.startEditList` interactions. The runner -also answers clarification callbacks but does not continue a conversation when -a candidate returns its clarification as final-response text. These are harness -limitations, not evidence of inferior candidate accuracy. Do not use affected -paired cases for comparative scores or resume paid execution until these paths -have safe, scoped support and representative live validation. A correction -requires a newly frozen protocol; preserve the original results and cumulative -credit ledger rather than rewriting or selectively replaying them. +**Protocol 2:** the original partial pass remains historical, with affected +paired cases excluded rather than scored as candidate failures. Each new trial +gets a private SDK temp directory. Overflow files are registered only from SDK +completion notices, must be regular single-link direct children with the SDK +filename pattern, and are SHA256-checked again before registered file reads. +This does not authorize arbitrary temp files, native replacement actions, or +retrying failed execution. Provenance failures stop the run as harness failures. + +Fixture confirmations also permit `list.startEditList` only for the specific +case's disposable target list. A final-text clarification receives the same +single scripted answer as a callback, within the original end-to-end deadline +and never after a terminal execution failure. The internal fallback toolset +and candidate entry interfaces are unchanged. Freeze a new run after validating +these paths; do not overwrite or selectively replay the historical run. + +The user-authorized cumulative ceiling is now 50,000 Copilot AI credits. This +raises the maximum accepted ledger cap, not the balance of any existing run: +reconcile all prior charges, preserve the original ledger and budget amendment, +and retain reporting headroom before admitting new work. ## Structured actions in Direct and MCP modes diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs index 80e1cb9134..b050a5593b 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs @@ -173,6 +173,18 @@ export function fixtureConfirmationAllowed( } if (schemaName !== "list") return false; + if (actionName === "startEditList") { + const target = { + S4: "grocery", + M3: "grocery", + M5: "errand", + R4: "packing", + R5: "errand", + A1: "grocery", + A4: "grocery", + }[id]; + return target !== undefined && parameters.listName === target; + } if (actionName === "clearList") { return ["M3", "A4"].includes(id) && parameters.listName === "grocery"; } @@ -210,6 +222,31 @@ export function isClarificationQuestion(id, question) { ); } +export async function sendWithClarification({ + session, + prompt, + timeoutMs, + testCase, + canClarify, + clarify, +}) { + const start = performance.now(); + const first = await session.sendAndWait({ prompt }, timeoutMs); + const text = first?.data.content ?? ""; + if ( + canClarify() && + testCase.clarification && + isClarificationQuestion(testCase.id, text) + ) { + const remaining = timeoutMs - (performance.now() - start); + if (remaining <= 0) + throw new Error("Clarification exhausted trial timeout"); + const answer = clarify(text, "final_text"); + return session.sendAndWait({ prompt: answer }, remaining); + } + return first; +} + export function shuffled(values, seed) { const result = [...values]; let state = seed >>> 0; diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs index 471defa97f..3bdbe622de 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs @@ -11,6 +11,10 @@ import { execFileSync } from "node:child_process"; import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; import { CopilotCreditBudget } from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; import { getCopilotPermissionDefault } from "../../dispatcher/dispatcher/dist/reasoning/copilot.js"; +import { + registerGhcpEvalArtifact, + isGhcpEvalArtifact, +} from "../../dispatcher/dispatcher/dist/execute/ghcpEvalArtifacts.js"; import { balancedOrder, buildCorpus, @@ -21,6 +25,7 @@ import { listFixture, normalizeLists, shuffled, + sendWithClarification, } from "./ghcp-eval-corpus.mjs"; import { stageCopilotPlugin } from "../../../tools/scripts/stageCopilotPlugin.mjs"; import { @@ -286,6 +291,14 @@ function prepareTrial(candidate, directory, workspace) { const sessionId = randomUUID(); env.TYPEAGENT_COPILOT_CREDIT_SESSION_SCOPE = sessionId; env.TYPEAGENT_GHCP_EVAL_TRACE = path.join(directory, "events.jsonl"); + const temporaryRoot = path.resolve(directory, "sdk-temp"); + fs.mkdirSync(temporaryRoot); + env.TEMP = env.TMP = env.TMPDIR = temporaryRoot; + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS = path.join(directory, "artifacts.json"); + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, + JSON.stringify({ root: temporaryRoot, artifacts: [] }), + ); fs.cpSync(path.join(template, "data"), env.TYPEAGENT_USER_DATA_DIR, { recursive: true, filter: (entry) => !entry.endsWith(".lock"), @@ -370,6 +383,10 @@ function persistTrial({ directory, started, }) { + if (result.harnessError) { + result.status = "harness_failed"; + result.error = result.harnessError; + } result.totalIncludingSetupMs = performance.now() - started; collectObservations(result, env.TYPEAGENT_GHCP_EVAL_TRACE); result.terminalExecutionFailure = executionStopped; @@ -424,6 +441,17 @@ async function trial(candidate, directory, testCase, workspace, evidence) { ? { toolResults: [] } : undefined; const approvedInteractions = new Set(); + const clarify = (question, source) => { + if (executionStopped || clarificationGiven) + throw new Error("Clarification cannot replay stopped work"); + clarificationGiven = true; + result.stateAtClarification = normalizeLists( + JSON.parse(fs.readFileSync(stores[0], "utf8")), + ); + result.clarificationSource = source; + result.clarificationQuestion = question; + return testCase.clarification; + }; const started = performance.now(); try { if (candidate.id !== 7) { @@ -551,12 +579,8 @@ async function trial(candidate, directory, testCase, workspace, evidence) { "Clarification is required before effect confirmation.", ); } - clarificationGiven = true; - result.stateAtClarification = normalizeLists( - JSON.parse(fs.readFileSync(stores[0], "utf8")), - ); return { - answer: testCase.clarification, + answer: clarify(request.question, "callback"), wasFreeform: true, }; } @@ -567,14 +591,22 @@ async function trial(candidate, directory, testCase, workspace, evidence) { )?.result.structuredContent; const action = pending?.prompt?.action; if ( - confirmationCount < 4 && + !executionStopped && + confirmationCount < 8 && pending?.prompt?.type === "confirmation" && - fixtureConfirmationAllowed( + (fixtureConfirmationAllowed( testCase.id, action, workspace, evidence?.issueTitle, - ) + ) || + (action?.schemaName === "powershell.powershell-files" && + action.actionName === "readFile" && + typeof action.parameters?.path === "string" && + isGhcpEvalArtifact( + action.parameters.path, + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, + ))) ) { const yes = request.choices?.find((choice) => /^(yes|approve|confirm|proceed|allow)\b/i.test(choice), @@ -637,6 +669,20 @@ async function trial(candidate, directory, testCase, workspace, evidence) { }), ); session.on("tool.execution_complete", (event) => { + try { + const artifact = registerGhcpEvalArtifact( + event.data.result, + env.TEMP, + env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, + ); + if (artifact) { + result.outputArtifacts ??= []; + result.outputArtifacts.push(artifact); + } + } catch (error) { + result.harnessError = `Artifact provenance failed: ${String(error)}`; + executionStopped = true; + } if (network) network.toolResults.push(event.data); const tool = result.tools.find( (tool) => tool.toolCallId === event.data.toolCallId, @@ -676,10 +722,14 @@ async function trial(candidate, directory, testCase, workspace, evidence) { } measuredStart = performance.now(); result.promptAcceptedAt = new Date().toISOString(); - const answer = await session.sendAndWait( - { prompt: result.prompt }, - 90_000, - ); + const answer = await sendWithClarification({ + session, + prompt: result.prompt, + timeoutMs: 90_000, + testCase, + canClarify: () => !clarificationGiven && !executionStopped, + clarify, + }); result.e2eMs = performance.now() - measuredStart; result.finalResponseAt = new Date().toISOString(); result.answer = answer?.data.content ?? ""; @@ -804,6 +854,7 @@ const order = balancedOrder( const specification = JSON.stringify( { + protocolVersion: 2, runnerSha256: createHash("sha256") .update(fs.readFileSync(fileURLToPath(import.meta.url))) .digest("hex"), @@ -829,6 +880,11 @@ const specification = perSessionRequestLimit: 24, cumulativeRequestLimit: 2000, requestCreditReservation: 2118, + cumulativeCreditCap: 50000, + overflowPolicy: + "Trial-private temp artifacts registered from SDK completion notices; canonical direct child, regular unlinked file, SHA256 rechecked before registered reads.", + clarificationPolicy: + "One scripted corpus answer through callback or final text, same 90-second end-to-end deadline; no continuation after execution failure.", sessionCreditSoftLimit: 60, ledgerPath, templateDirectory: path.resolve(template), @@ -889,6 +945,7 @@ for (const entry of order.slice(batchStart, batchStart + batchSize)) { }) + "\n", ); if ( + result.status === "harness_failed" || (phase === "pilot" && result.status !== "completed_ungraded") || /credit|budget|reservation|Agent server|permission orchestrator/i.test( result.error ?? "", diff --git a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs index f0aa607c3f..1b4f8e2718 100644 --- a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs +++ b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs @@ -18,9 +18,80 @@ import { listFixture, normalizeLists, shuffled, + sendWithClarification, } from "../ghcp-eval-corpus.mjs"; const corpus = buildCorpus("C:\\fixtures", "owner/repo", 10, 20, 30); +test("intermediate edit confirmation is limited to the case's disposable list", () => { + const action = { + schemaName: "list", + actionName: "startEditList", + parameters: { listName: "errand" }, + }; + assert.equal( + fixtureConfirmationAllowed("R5", action, "C:\\fixtures"), + true, + ); + assert.equal( + fixtureConfirmationAllowed("S1", action, "C:\\fixtures"), + false, + ); + assert.equal( + fixtureConfirmationAllowed("A1", action, "C:\\fixtures"), + false, + ); +}); +test("final-text clarification gets exactly one answer within the same deadline", async () => { + const calls = []; + const session = { + sendAndWait: async (input, timeout) => { + calls.push({ input, timeout }); + return { + data: { + content: + calls.length === 1 + ? "Which list should receive apples?" + : "Done", + }, + }; + }, + }; + const testCase = corpus.find((entry) => entry.id === "A1"); + const result = await sendWithClarification({ + session, + prompt: testCase.prompt, + timeoutMs: 1000, + testCase, + canClarify: () => true, + clarify: () => testCase.clarification, + }); + assert.equal(result.data.content, "Done"); + assert.equal(calls.length, 2); + assert.equal(calls[1].input.prompt, "The grocery list."); + assert.ok(calls[1].timeout <= calls[0].timeout); +}); +test("text continuation never replays stopped work or confirms a guessed target", async () => { + for (const [text, allowed] of [ + ["Which list?", false], + ["Confirm adding apples to grocery?", true], + ]) { + let calls = 0; + await sendWithClarification({ + session: { + sendAndWait: async () => { + calls++; + return { data: { content: text } }; + }, + }, + prompt: "Add apples to my list.", + timeoutMs: 1000, + testCase: corpus.find((entry) => entry.id === "A1"), + canClarify: () => allowed, + clarify: () => assert.fail("must not answer"), + }); + assert.equal(calls, 1); + } +}); test("failure detection preserves structured status and NL errors, not check-result words", () => { assert.equal( terminalExecutionFailure( diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts new file mode 100644 index 0000000000..3789e3f3d5 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts @@ -0,0 +1,116 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { createHash } from "node:crypto"; + +type Artifact = { path: string; sha256: string }; +type Manifest = { root: string; artifacts: Artifact[] }; + +function originalContent(result: object): string[] { + const expected: string[] = []; + if ( + "detailedContent" in result && + typeof result.detailedContent === "string" + ) + expected.push(result.detailedContent); + const chunks = + "contents" in result && Array.isArray(result.contents) + ? result.contents + : []; + const texts = chunks.flatMap((chunk: unknown) => + typeof chunk === "object" && + chunk !== null && + "type" in chunk && + chunk.type === "text" && + "text" in chunk && + typeof chunk.text === "string" + ? [chunk.text] + : [], + ); + if (texts.length) expected.push(texts.join("\n"), texts.join("\n\n")); + if ( + "structuredContent" in result && + result.structuredContent !== undefined + ) { + const structured = JSON.stringify(result.structuredContent); + expected.push( + ...expected.map((text) => `${text}\n\n${structured}`), + structured, + JSON.stringify(result.structuredContent, null, 2), + ); + } + return expected; +} + +function fingerprint(file: string, root: string): Artifact { + const canonicalRoot = fs.realpathSync(root); + const canonical = fs.realpathSync(file); + const stat = fs.lstatSync(file); + if ( + !stat.isFile() || + stat.isSymbolicLink() || + stat.nlink !== 1 || + stat.size > 10 * 1024 * 1024 || + path.dirname(canonical).toLowerCase() !== canonicalRoot.toLowerCase() || + path.resolve(file).toLowerCase() !== canonical.toLowerCase() || + !/^(?:\d+-)?copilot-tool-output-[\w-]+\.txt$/.test(path.basename(file)) + ) { + throw new Error("Untrusted GHCP evaluation output artifact"); + } + return { + path: canonical, + sha256: createHash("sha256") + .update(fs.readFileSync(file)) + .digest("hex"), + }; +} + +/** Register only an SDK completion's overflow notice inside this trial's private temp directory. */ +export function registerGhcpEvalArtifact( + result: unknown, + root: string, + manifestFile: string, +): Artifact | undefined { + if ( + !result || + typeof result !== "object" || + !("content" in result) || + typeof result.content !== "string" + ) + return undefined; + const match = result.content.match( + /^Output too large to read at once[^\r\n]*?Saved to:\s*([^\r\n]+)/, + ); + if (!match) return undefined; + const artifact = fingerprint(match[1].trim(), root); + const text = fs.readFileSync(artifact.path, "utf8"); + if (!originalContent(result).includes(text)) + throw new Error("Output artifact does not match the SDK result"); + const manifest: Manifest = JSON.parse( + fs.readFileSync(manifestFile, "utf8"), + ); + if (manifest.root !== root) throw new Error("Artifact scope mismatch"); + if (!manifest.artifacts.some((entry) => entry.path === artifact.path)) + manifest.artifacts.push(artifact); + fs.writeFileSync(manifestFile, JSON.stringify(manifest)); + return artifact; +} + +export function isGhcpEvalArtifact( + file: string, + manifestFile = process.env.TYPEAGENT_GHCP_EVAL_ARTIFACTS, +): boolean { + if (!manifestFile) return false; + const manifest: Manifest = JSON.parse( + fs.readFileSync(manifestFile, "utf8"), + ); + const expected = manifest.artifacts.find( + (entry) => + entry.path.toLowerCase() === path.resolve(file).toLowerCase(), + ); + if (!expected) return false; + const actual = fingerprint(file, manifest.root); + return actual.sha256 === expected.sha256; +} diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts index 95271d487b..e4053554b5 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts @@ -3,6 +3,7 @@ import fs from "node:fs"; import path from "node:path"; +import { isGhcpEvalArtifact } from "./ghcpEvalArtifacts.js"; let executionFailureObserved = false; @@ -77,7 +78,8 @@ export function assertGhcpEvalAction( (file) => fs.realpathSync(path.join(fixtureRoot, file)).toLowerCase(), ); - if (allowed.includes(requested)) return; + if (allowed.includes(requested) || isGhcpEvalArtifact(parameters.path)) + return; } recordGhcpEvalEvent("action.denied", { schemaName, actionName }); markGhcpEvalExecutionFailure(); diff --git a/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts b/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts index 9590d0b537..58accaf93b 100644 --- a/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts +++ b/ts/packages/dispatcher/dispatcher/src/reasoning/copilotCreditBudget.ts @@ -51,8 +51,8 @@ export function validateCreditLedger(ledger: CopilotCreditLedger): void { ] as const) { requireAmount(ledger[name], name); } - if (ledger.capNanoAiu > 20_000_000_000_000) { - throw new Error("Credit ledger exceeds the 20,000-credit ceiling"); + if (ledger.capNanoAiu > 50_000_000_000_000) { + throw new Error("Credit ledger exceeds the 50,000-credit ceiling"); } if (ledger.requestMaximumNanoAiu === 0) { throw new Error("A positive request reservation is required"); diff --git a/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts b/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts index 09f74210c0..c26c7425c6 100644 --- a/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts +++ b/ts/packages/dispatcher/dispatcher/test/copilotCreditBudget.spec.ts @@ -197,9 +197,11 @@ describe("Copilot credit admission", () => { }, ); - it("rejects cap increases and unbounded requests", () => { + it("accepts the amended ceiling but rejects excess and unbounded requests", () => { const state = ledger(); - state.capNanoAiu = 20_000_000_000_001; + state.capNanoAiu = 50_000_000_000_000; + expect(() => validateCreditLedger(state)).not.toThrow(); + state.capNanoAiu = 50_000_000_000_001; expect(() => validateCreditLedger(state)).toThrow("ceiling"); state.capNanoAiu = 20_000; state.requestMaximumNanoAiu = 0; diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts new file mode 100644 index 0000000000..13dcecd720 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts @@ -0,0 +1,101 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { + isGhcpEvalArtifact, + registerGhcpEvalArtifact, +} from "../src/execute/ghcpEvalArtifacts.js"; + +describe("isolated output artifact provenance", () => { + let root: string; + let manifest: string; + let artifact: string; + const notice = (file: string) => ({ + content: `Output too large to read at once (30 KB). Saved to:\n${file}\nPreview`, + detailedContent: "tool evidence", + }); + beforeEach(() => { + root = fs.mkdtempSync(path.join(os.tmpdir(), "ghcp-artifacts-")); + manifest = path.join(root, "manifest.json"); + artifact = path.join(root, "123-copilot-tool-output-abc.txt"); + fs.writeFileSync(manifest, JSON.stringify({ root, artifacts: [] })); + fs.writeFileSync(artifact, "tool evidence"); + }); + afterEach(() => fs.rmSync(root, { recursive: true, force: true })); + it("requires a trusted completion notice and exact content at read time", () => { + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(false); + expect( + registerGhcpEvalArtifact({ content: "ordinary" }, root, manifest), + ).toBeUndefined(); + registerGhcpEvalArtifact(notice(artifact), root, manifest); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(true); + fs.writeFileSync(artifact, "changed"); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(false); + }); + it("rejects arbitrary names, subdirectories and hard-link aliases", () => { + const other = path.join(root, "private.txt"); + fs.writeFileSync(other, "private"); + expect(() => + registerGhcpEvalArtifact(notice(other), root, manifest), + ).toThrow("Untrusted"); + const child = path.join(root, "child"); + fs.mkdirSync(child); + const nested = path.join(child, path.basename(artifact)); + fs.writeFileSync(nested, "nested"); + expect(() => + registerGhcpEvalArtifact(notice(nested), root, manifest), + ).toThrow("Untrusted"); + const alias = path.join(root, "copilot-tool-output-alias.txt"); + fs.linkSync(artifact, alias); + expect(() => + registerGhcpEvalArtifact(notice(alias), root, manifest), + ).toThrow("Untrusted"); + }); + it("rejects another trial's scope even for a correctly named output", () => { + const child = path.join(root, "other-trial"); + fs.mkdirSync(child); + expect(() => + registerGhcpEvalArtifact(notice(artifact), child, manifest), + ).toThrow("Untrusted"); + expect(isGhcpEvalArtifact(artifact, undefined)).toBe(false); + }); + it("accepts same-line SDK notices only when content matches the structured result", () => { + fs.writeFileSync( + artifact, + JSON.stringify({ status: "completed", output: ["evidence"] }), + ); + const result = { + content: `Output too large to read at once (30 KB). Saved to: ${artifact}\nPreview`, + structuredContent: { status: "completed", output: ["evidence"] }, + }; + registerGhcpEvalArtifact(result, root, manifest); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(true); + fs.writeFileSync( + artifact, + JSON.stringify({ secret: "not the result" }), + ); + expect(() => registerGhcpEvalArtifact(result, root, manifest)).toThrow( + "does not match", + ); + }); + it("verifies SDK text plus structured content without accepting suffixes", () => { + const structuredContent = { status: "completed", output: ["evidence"] }; + const text = JSON.stringify(structuredContent, null, 2); + const body = `${text}\n\n${JSON.stringify(structuredContent)}`; + const result = { + ...notice(artifact), + structuredContent, + contents: [{ type: "text", text }], + }; + fs.writeFileSync(artifact, body); + registerGhcpEvalArtifact(result, root, manifest); + expect(isGhcpEvalArtifact(artifact, manifest)).toBe(true); + fs.writeFileSync(artifact, body + "\nunrelated secret"); + expect(() => registerGhcpEvalArtifact(result, root, manifest)).toThrow( + "does not match", + ); + }); +}); From 4275a7ca7743b761b7a108971de34184ebf6e289 Mon Sep 17 00:00:00 2001 From: George Ng Date: Thu, 24 Sep 2026 13:31:37 -0700 Subject: [PATCH 07/14] Close native no-replay gap for future GHCP eval runs Preserve measured protocol-two evidence; bump future specifications to protocol three and stop after failed native domain tools even without SDK error details. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin/README.md | 8 ++++ .../scripts/ghcp-eval-grade.mjs | 6 +++ .../copilot-plugin/scripts/ghcp-eval.mjs | 2 +- .../scripts/test/ghcp-eval.spec.mjs | 38 +++++++++++++++++++ 4 files changed, 53 insertions(+), 1 deletion(-) diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index 2a19e723b3..172835062a 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -106,6 +106,14 @@ and never after a terminal execution failure. The internal fallback toolset and candidate entry interfaces are unchanged. Freeze a new run after validating these paths; do not overwrite or selectively replay the historical run. +**Protocol 3:** native domain tool failures now enter the same terminal +no-replay gate as failed TypeAgent execution, including SDK failures with no +error payload. Failed `ask_user` interactions are not domain effects. This +prospective guard correction does not relabel or rerun protocol-two evidence: +audit earlier native/mixed traces for operations after failed native calls, +and do not count those recoveries as policy-valid successes merely because +their final answers are correct. + The user-authorized cumulative ceiling is now 50,000 Copilot AI credits. This raises the maximum accepted ledger cap, not the balance of any existing run: reconcile all prior charges, preserve the original ledger and budget amendment, diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs index 42fc2d24c2..33a4380f3e 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs @@ -2,6 +2,12 @@ // Licensed under the MIT License. export function terminalExecutionFailure(toolName, result, success) { + if ( + /^(?:functions[.-])?(?:powershell|view|glob|rg|web_fetch)$/.test( + toolName, + ) + ) + return success === false; if (!/processCommand|executeAction|continueAction/.test(toolName)) return false; return ( diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs index 3bdbe622de..9c1c3b404e 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs @@ -854,7 +854,7 @@ const order = balancedOrder( const specification = JSON.stringify( { - protocolVersion: 2, + protocolVersion: 3, runnerSha256: createHash("sha256") .update(fs.readFileSync(fileURLToPath(import.meta.url))) .digest("hex"), diff --git a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs index 1b4f8e2718..4233229e89 100644 --- a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs +++ b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs @@ -144,6 +144,44 @@ test("nested/parallel tool durations are not double-counted and empty tails are assert.equal(percentile([9, 1, 3], 0.5), 3); assert.equal(percentile([9, 1, 3], 0.95), 9); }); +test("native domain failures stop execution even without an SDK error payload", () => { + for (const tool of [ + "powershell", + "view", + "glob", + "rg", + "web_fetch", + "functions.powershell", + "functions-web_fetch", + ]) { + assert.equal(terminalExecutionFailure(tool, undefined, false), true); + assert.equal(terminalExecutionFailure(tool, {}, true), false); + } + assert.equal(terminalExecutionFailure("ask_user", undefined, false), false); +}); +test("failed native execution cannot trigger a scripted continuation", async () => { + let stopped = false; + let calls = 0; + await sendWithClarification({ + session: { + sendAndWait: async () => { + calls++; + stopped = terminalExecutionFailure( + "powershell", + undefined, + false, + ); + return { data: { content: "Which file?" } }; + }, + }, + prompt: "Read that file.", + timeoutMs: 1000, + testCase: corpus.find((entry) => entry.id === "A5"), + canClarify: () => !stopped, + clarify: () => assert.fail("cannot continue after native failure"), + }); + assert.equal(calls, 1); +}); test("independent PR file evidence must be complete", () => { const snapshot = { status: "passed", From b47f8bacd5073aa226392cc6ccb47a9a7e4caac7 Mon Sep 17 00:00:00 2001 From: George Ng Date: Fri, 25 Sep 2026 12:20:24 -0700 Subject: [PATCH 08/14] Fix cross-platform GHCP evaluation artifact and fixture paths Accept the configured temporary-directory alias and its canonical spelling without admitting other aliases, nested files, or modified artifacts. Use host-native paths in confirmation fixtures. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../scripts/test/ghcp-eval.spec.mjs | 46 +++++++++++-------- .../src/execute/ghcpEvalArtifacts.ts | 11 +++-- .../dispatcher/test/ghcpEvalArtifacts.spec.ts | 30 ++++++++++++ 3 files changed, 64 insertions(+), 23 deletions(-) diff --git a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs index 4233229e89..67898ae1dc 100644 --- a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs +++ b/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs @@ -2,6 +2,7 @@ // Licensed under the MIT License. import assert from "node:assert/strict"; +import path from "node:path"; import { test } from "node:test"; import { externalOracle, @@ -21,25 +22,17 @@ import { sendWithClarification, } from "../ghcp-eval-corpus.mjs"; -const corpus = buildCorpus("C:\\fixtures", "owner/repo", 10, 20, 30); +const fixtures = path.resolve("fixtures"); +const corpus = buildCorpus(fixtures, "owner/repo", 10, 20, 30); test("intermediate edit confirmation is limited to the case's disposable list", () => { const action = { schemaName: "list", actionName: "startEditList", parameters: { listName: "errand" }, }; - assert.equal( - fixtureConfirmationAllowed("R5", action, "C:\\fixtures"), - true, - ); - assert.equal( - fixtureConfirmationAllowed("S1", action, "C:\\fixtures"), - false, - ); - assert.equal( - fixtureConfirmationAllowed("A1", action, "C:\\fixtures"), - false, - ); + assert.equal(fixtureConfirmationAllowed("R5", action, fixtures), true); + assert.equal(fixtureConfirmationAllowed("S1", action, fixtures), false); + assert.equal(fixtureConfirmationAllowed("A1", action, fixtures), false); }); test("final-text clarification gets exactly one answer within the same deadline", async () => { const calls = []; @@ -240,13 +233,13 @@ test("scripted confirmations are limited to exact disposable fixture actions", ( actionName: "addItems", parameters: { listName: "grocery", items: ["apples"] }, }; - assert.equal(fixtureConfirmationAllowed("S4", add, "C:\\fixtures"), true); - assert.equal(fixtureConfirmationAllowed("S1", add, "C:\\fixtures"), false); + assert.equal(fixtureConfirmationAllowed("S4", add, fixtures), true); + assert.equal(fixtureConfirmationAllowed("S1", add, fixtures), false); assert.equal( fixtureConfirmationAllowed( "S4", { ...add, schemaName: "github-cli" }, - "C:\\fixtures", + fixtures, ), false, ); @@ -257,17 +250,30 @@ test("scripted confirmations are limited to exact disposable fixture actions", ( ...add, parameters: { listName: "errand", items: ["guessed title"] }, }, - "C:\\fixtures", + fixtures, ), false, ); const read = { schemaName: "powershell.powershell-files", actionName: "readFile", - parameters: { path: "C:\\fixtures\\report-a.txt" }, + parameters: { path: path.join(fixtures, "report-a.txt") }, }; - assert.equal(fixtureConfirmationAllowed("S2", read, "C:\\fixtures"), true); - assert.equal(fixtureConfirmationAllowed("A2", read, "C:\\fixtures"), false); + assert.equal(fixtureConfirmationAllowed("S2", read, fixtures), true); + assert.equal(fixtureConfirmationAllowed("A2", read, fixtures), false); + assert.equal( + fixtureConfirmationAllowed( + "S2", + { + ...read, + parameters: { + path: path.resolve(fixtures, "..", "report-a.txt"), + }, + }, + fixtures, + ), + false, + ); }); test("seeded ordering is reproducible without dropping examples", () => { assert.deepEqual(shuffled(corpus, 42), shuffled(corpus, 42)); diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts index 3789e3f3d5..b839055f29 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalArtifacts.ts @@ -53,8 +53,9 @@ function fingerprint(file: string, root: string): Artifact { stat.isSymbolicLink() || stat.nlink !== 1 || stat.size > 10 * 1024 * 1024 || - path.dirname(canonical).toLowerCase() !== canonicalRoot.toLowerCase() || - path.resolve(file).toLowerCase() !== canonical.toLowerCase() || + path.relative(canonicalRoot, path.dirname(canonical)) !== "" || + (path.relative(root, path.dirname(file)) !== "" && + path.relative(canonicalRoot, path.dirname(file)) !== "") || !/^(?:\d+-)?copilot-tool-output-[\w-]+\.txt$/.test(path.basename(file)) ) { throw new Error("Untrusted GHCP evaluation output artifact"); @@ -108,7 +109,11 @@ export function isGhcpEvalArtifact( ); const expected = manifest.artifacts.find( (entry) => - entry.path.toLowerCase() === path.resolve(file).toLowerCase(), + path.relative(entry.path, file) === "" || + path.relative( + path.join(manifest.root, path.basename(entry.path)), + file, + ) === "", ); if (!expected) return false; const actual = fingerprint(file, manifest.root); diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts index 13dcecd720..49d8ccd334 100644 --- a/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalArtifacts.spec.ts @@ -25,6 +25,36 @@ describe("isolated output artifact provenance", () => { fs.writeFileSync(artifact, "tool evidence"); }); afterEach(() => fs.rmSync(root, { recursive: true, force: true })); + it("accepts a trial beneath an aliased temp parent and verifies both path spellings", () => { + const parent = path.join(root, "actual"); + const alias = path.join(root, "alias"); + fs.mkdirSync(parent); + fs.symlinkSync(parent, alias, "junction"); + const trial = path.join(alias, "trial"); + fs.mkdirSync(trial); + const output = path.join(trial, path.basename(artifact)); + fs.writeFileSync(output, "tool evidence"); + fs.writeFileSync( + manifest, + JSON.stringify({ root: trial, artifacts: [] }), + ); + expect( + registerGhcpEvalArtifact(notice(output), trial, manifest)?.path, + ).toBe(fs.realpathSync(output)); + expect(isGhcpEvalArtifact(output, manifest)).toBe(true); + expect(isGhcpEvalArtifact(fs.realpathSync(output), manifest)).toBe( + true, + ); + const outsideAlias = path.join(root, "other-alias"); + fs.symlinkSync(fs.realpathSync(trial), outsideAlias, "junction"); + const outsideOutput = path.join(outsideAlias, path.basename(output)); + expect(isGhcpEvalArtifact(outsideOutput, manifest)).toBe(false); + expect(() => + registerGhcpEvalArtifact(notice(outsideOutput), trial, manifest), + ).toThrow("Untrusted"); + fs.writeFileSync(output, "changed"); + expect(isGhcpEvalArtifact(output, manifest)).toBe(false); + }); it("requires a trusted completion notice and exact content at read time", () => { expect(isGhcpEvalArtifact(artifact, manifest)).toBe(false); expect( From bb09c7930ba8773e48e0b738036c2142836411c2 Mon Sep 17 00:00:00 2001 From: George Ng Date: Fri, 25 Sep 2026 15:19:22 -0700 Subject: [PATCH 09/14] Move Copilot evaluation harness into a dedicated package Promote the updated methodology to the eval README, document initial ballpark scope and isolation limitations, and pin future outer/nested Copilot evaluation to Luna 5.6 with fail-fast ledger identity checks. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin-eval/README.md | 305 ++++++++++++++++++ ts/packages/copilot-plugin-eval/package.json | 23 ++ .../scripts/ghcp-credit-probe.mjs | 5 +- .../scripts/ghcp-eval-config.mjs | 12 + .../scripts/ghcp-eval-corpus.mjs | 0 .../scripts/ghcp-eval-grade.mjs | 0 .../scripts/ghcp-eval-preflight.mjs | 6 +- .../scripts/ghcp-eval.mjs | 20 +- .../scripts/test/entrypoints.spec.mjs | 53 +++ .../scripts/test/ghcp-eval.spec.mjs | 11 + ts/packages/copilot-plugin/README.md | 89 +---- ts/packages/copilot-plugin/package.json | 3 +- ts/pnpm-lock.yaml | 21 ++ 13 files changed, 452 insertions(+), 96 deletions(-) create mode 100644 ts/packages/copilot-plugin-eval/README.md create mode 100644 ts/packages/copilot-plugin-eval/package.json rename ts/packages/{copilot-plugin => copilot-plugin-eval}/scripts/ghcp-credit-probe.mjs (94%) create mode 100644 ts/packages/copilot-plugin-eval/scripts/ghcp-eval-config.mjs rename ts/packages/{copilot-plugin => copilot-plugin-eval}/scripts/ghcp-eval-corpus.mjs (100%) rename ts/packages/{copilot-plugin => copilot-plugin-eval}/scripts/ghcp-eval-grade.mjs (100%) rename ts/packages/{copilot-plugin => copilot-plugin-eval}/scripts/ghcp-eval-preflight.mjs (97%) rename ts/packages/{copilot-plugin => copilot-plugin-eval}/scripts/ghcp-eval.mjs (98%) create mode 100644 ts/packages/copilot-plugin-eval/scripts/test/entrypoints.spec.mjs rename ts/packages/{copilot-plugin => copilot-plugin-eval}/scripts/test/ghcp-eval.spec.mjs (95%) diff --git a/ts/packages/copilot-plugin-eval/README.md b/ts/packages/copilot-plugin-eval/README.md new file mode 100644 index 0000000000..dbc159e8f0 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/README.md @@ -0,0 +1,305 @@ +# TypeAgent Copilot end-to-end evaluation + +Updated: 2026-09-25. This is the maintained version of the original GHCP +evaluation methodology, alongside its harness, corpus, grading, and tests. + +## Purpose and limitations + +**This is a simple initial evaluation for ballpark estimates**, not a +production-quality benchmark or a statistically powered comparison. It samples +twenty tasks across four domains to explore task completion, user-visible +latency, and workflow overhead. Five cases per cohort and one historical +balanced repetition cannot establish precise tail percentiles, non-inferiority, +or a general winning strategy. + +**We acknowledge potential environment-isolation risks from implementing and +running this evaluation alongside the Copilot plugin.** Moving its code into +this separate directory prevents it from being part of the plugin source +layout; it does not create an OS security boundary or prove full isolation. +The harness still reuses plugin staging/discovery infrastructure, the same +host, authenticated CLI, runtime dependencies, model configuration, and +read-only external services. Native tools and inherited process configuration +can expose environmental differences. Shared caches, provider behavior and +changing GitHub/network data can confound comparisons. This implementation is +aware of these limitations; isolated fixtures, scoped permissions, private +data/temp directories and trace audits mitigate them but do not eliminate them. +Never describe this implementation as a sandbox or isolation certification. + +Only actual Copilot SDK conversations count as end-to-end trials. A supplied +correct action, mocked model selection, discovery smoke test or dispatcher +microbenchmark is not a substitute. Throughput/load tests, cold-start campaigns, +Direct-hook comparisons and a broad adversarial suite are outside this initial +scope. Weather is removed and calendar is deferred. + +## Implementation and protocol history + +| Version | Meaning | +| ---------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Protocol 2 | Historical measured implementation at `e847c7a00907a1f4e2c6c4426c935b964fd9997c`: completed 140 trials. An earlier partial pass had harness-invalid records, which were excluded rather than scored as candidate failures. | +| Protocol 3 | Strict native failure/no-replay correction at `4275a7ca7743b761b7a108971de34184ebf6e289`; tested offline, not live measured. This base runner still uses that policy. | +| Portability | `b47f8bacd5073aa226392cc6ccb47a9a7e4caac7` accepts configured temp-parent aliases while retaining artifact provenance checks, and uses platform-native test paths. | +| Protocol 4 follow-up (#3077) | Prospective positive-evidence safe-read recovery, frozen native applicability and replacement A4 oracle. Kept in its separate dependent PR; do not infer its runtime behavior from this base runner. | +| Model and layout revision | Eval-only scripts now reside here. Future Copilot sessions explicitly use Luna 5.6 (`gpt-5.6-luna`), not the historical `gpt-5.6-sol`. No Luna rerun or improved measured outcome is claimed. | + +Preserve original results, grades, run specifications and safety audits. +Retrospective reporting amendments do not rewrite observations or prove what +a stopped trial would have done. Result-entity product fixes are separate +from the evaluation harness and do not establish new measured success rates. + +## Running and package boundaries + +From `ts`, after normal worktree dependency provisioning: + +```powershell +pnpm exec fluid-build '^@typeagent/copilot-plugin-eval$' -t build --dep +pnpm --filter @typeagent/copilot-plugin-eval test +``` + +The package is private and JavaScript-only. Its build checks executable syntax; +dependency-aware builds prepare the plugin, dispatcher and agent-server. +Tests are offline. Plugin installation/bundling does not include this harness. +`copilot-plugin/scripts/discovery-e2e.mjs` remains shared plugin test +infrastructure; dispatcher-side permission, credit and artifact guards stay +at their actual runtime enforcement boundaries rather than moving into a +client-only package. + +Live commands below require separate authorization, an authenticated Copilot +executable, existing model configuration, a reconciled model-specific credit +ledger, and verified handlers/catalog. They can incur model usage; building or +testing this package does not run them. + +```text +node packages\copilot-plugin-eval\scripts\ghcp-eval-preflight.mjs --external-evidence +node packages\copilot-plugin-eval\scripts\ghcp-eval.mjs pilot +node packages\copilot-plugin-eval\scripts\ghcp-eval.mjs 1,2,3,4,5,6,7 measured S1 7 1 +node packages\copilot-plugin-eval\scripts\ghcp-credit-probe.mjs +``` + +Use nonsynchronized local directories for live databases and locks. Preserve +sanitized results and specifications in durable storage afterwards. The oracle +JSON pins issue/PR evidence and a relative `readinessFile` naming a successful +preflight. The current concrete GitHub inputs are PRs 3058/3067 and issue 2617 +in microsoft/TypeAgent; verify availability and contemporaneous evidence before +a new run. Do not silently replace targets during a frozen run. + +## Explicit model and admission + +`scripts/ghcp-eval-config.mjs` pins **Luna 5.6, `gpt-5.6-luna`**. The measured +outer conversation, same-binding contract preparation, and TypeAgent nested +Copilot reasoning use this identity. Preflight and the optional credit probe +also enforce it. Main trials keep high reasoning effort; the small accounting +probe uses low effort. TypeAgent translation and embedding providers remain +separately configured and must be recorded; this pin does not silently change +their model identities. + +Every entry point rejects a mismatched ledger before starting services or model +work. Do not relabel a historical Sol ledger or assume its per-request bounds +apply to Luna. Verify current model availability, credit rates, context/output +bounds and nested-request accounting; reconcile cumulative prior charges in a +new run specification. An unavailable model is a blocker, not permission to +substitute one. No live availability/pricing probe is part of this migration. + +The authorized ceiling is **50,000 cumulative Copilot AI credits**, superseding +the original 20,000 and interim 40,000 limits, not a fresh allowance per run. +Include planning, implementation, preparation, pilots, failed/cancelled work, +nested reasoning, grading and reporting. Reserve report headroom, retain +unsettled maximum reservations and stop new admissions if accounting is unclear. +Externally billed translation/embedding usage is reported separately; unknown +usage is not zero. This is not a general-purpose pricing or billing-hard-stop +service. + +The proxy reserves before forwarding, settles explicit billing fields, rejects +model mismatches/WebSockets and retains unknown charges. Limits include 24 +requests per scoped session and 2,000 cumulative ledger requests. The SDK +60-credit session limit is additional and **soft**, not the hard admission +mechanism. Historical request reservations are not certified bounds for Luna. +Moving code or changing models never reopens a closed ledger. + +## Seven candidates and fallback + +| # | Candidate | Allowed outer entry | +| --- | ----------------------------------- | ---------------------------------------------------------------------------------------------- | +| 1 | NL-MCP, fallback disabled | `typeagent-processCommand` only | +| 2 | NL-MCP, fallback enabled | Same as 1 | +| 3 | Structured discovery | `searchActions`, `executeAction`, structured continuation/cancellation; never `processCommand` | +| 4 | Structured current-contract reuse | Same structured surface; contracts earned by discovery earlier in the same binding | +| 5 | Production mixed, fallback disabled | NL and structured interfaces under production routing guidance | +| 6 | Production mixed, fallback enabled | Same as 5 | +| 7 | Native Copilot only | Pinned native tools; no TypeAgent plugin, hooks, tools, guidance or special storage access | + +All TypeAgent candidates use MCP. Direct-hook execution is excluded to keep +transport and the Copilot agent loop comparable; structured tools also exist +in Direct mode but are still MCP tools. Pure candidates neutralize conflicting +plugin guidance and enforce interfaces at the tool boundary. Mixed candidates +retain production guidance. Auxiliary outer workspace/macro/skill MCP servers +are omitted; do not silently alter the internal reasoning toolset. + +Fallback means TypeAgent's **failed translation to Copilot reasoning** +transition, controlled by `translationReasoningFallback`. Grammar/cache misses +proceeding to LLM translation, normal action selection, successful-result +reasoning, multi-tool orchestration and outer tool-error recovery are not this +fallback. Preserve grammar/cache/LLM translation in each toggle pair and trace +the actual decision, entry and outcome (`TYPEAGENT_GHCP_EVAL_TRACE`). + +Compare 1/3/4 for resolution strategies, 1/2 and 5/6 for fallback, and pure +versus mixed separately. Reuse must not preload answers, exact action arguments +or resolved ambiguous referents. Report discovery preparation separately and +include it in whole-workflow amortization, not just subsequent-turn savings. + +## Fixtures, domains and twenty cases + +Four cohorts contain five cases each. Domain operations must map to existing +list, GitHub CLI, registered PowerShell-file and read-only IP configuration +actions. PowerShell uses `powershell.powershell-files.readFile`, not generated +script replacements for structured candidates. GitHub and network actions are +read-only; no renewal, cache flush, write or real external effect is authorized. + +Seed lists before each trial: + +| List | Items | +| ------- | ------------------------ | +| grocery | milk, eggs, rice | +| pantry | rice, beans | +| packing | passport, charger, socks | +| travel | charger, adapter | +| office | notebook, pen, charger | +| errand | pharmacy, post office | +| weekend | empty | + +Fixture files are UTF-8 with LF and a final newline: `report-a.txt` contains +passport, charger, socks (three lines); `report-b.txt` contains charger, adapter +(two lines); `trip.txt` contains `destination: mountain` and `jacket: required`. +No other fixture files exist. Explicit absolute paths are equal user inputs +for every candidate; contents are hidden until read. Preserve all unrelated +lists, items and files. Dynamic GitHub/network answers are graded against +independently captured contemporaneous evidence, not the assistant's claims. + +The source corpus holds exact prompts and scripted answers; these summaries +define intent and outcomes without prescribing a single reasoning trace. + +| ID | Request | Independent success requirement | +| -------------------- | -------------------------------------------------------- | -------------------------------------------------------------------- | +| S1 | Show my lists. | Exactly seven seeded names | +| S2 | Read report-a.txt. | All three lines in order | +| S3 | Show files changed by PR A. | Complete observed file set | +| S4 | Add apples to grocery. | Apples added, previous items retained | +| S5 | Show full network configuration. | Faithful observed configuration | +| M1 | Read report-a and report-b. | Both complete and correctly labeled | +| M2 | Show network configuration and DNS cache. | Both observed outputs, no network changes | +| M3 | Empty grocery, then add bread and oranges. | Existing grocery contains exactly those two | +| M4 | Show PR A files and checks. | Both complete, pending/absent checks explicit | +| M5 | Show issue A, then add "review issue A" to errand. | Correct issue and literal item, previous items retained | +| R1 | Which items are in grocery and pantry? | Rice only, grounded in both reads | +| R2 | Which report has more nonempty lines, by how many? | report-a: three versus two, difference one | +| R3 | Which PR needs attention, failed checks then file count? | Evidence-grounded comparison; ties/unknowns explicit | +| R4 | Read trip; add jacket to packing only if required. | Jacket added, other state preserved | +| R5 | Add the retrieved issue title to errand only if absent. | Exact title, conditional addition, no duplicate | +| A1 | Add apples to my list. | Ask which list; answer grocery; then add | +| A2 | Read the report. | Ask which report; answer report-b; then read | +| A3 | Show files changed by that PR. | Ask which PR; answer PR A; then read | +| A4 (historical/base) | Clean up my grocery list. | Clarify operation; answer remove all items but keep list; then clear | +| A5 | Read that file. | Ask which file; answer trip; then read | + +Ambiguous cases start without antecedents/defaults. Scripted user answers are +only for these disposable fixtures, never approval of real effects. One answer +can arrive through a callback or a final-text clarification within the original +90-second deadline. Confirming a guessed referent is not clarification. +Final state alone cannot excuse premature mutation. + +**Latest A4 follow-up:** protocol 4 replaces only the vague clean-up verb with +an unresolved-item-removal request, then supplies the item after clarification. +Its independent oracle rejects guessed/wrong-item confirmation and premature +mutation even if the final state is correct, and preserves unrelated state. +This avoids conflating verb interpretation with referent resolution; it does +not claim the original product ambiguity is fixed. Original A4 evidence remains +historical. M1/M5/S3/A3 routing issues, timeouts and network presentation failures +are retained as valid failures, not replaced with easier tests. + +## Applicability and recovery amendments + +Native list-dependent cases **S1, S4, M3, M5, R1, R4, R5, A1, A4** are N/A, +including cross-domain tasks that require lists. Native's applicable denominator +is 11; candidates 1-6 retain 20. Retain original twenty-case native observations +and safety findings as historical evidence. Different denominators are not a +matched-workload ranking. Native receives no equivalent list adapter or hidden +fixture-storage coaching. + +Protocol 4 freezes this applicability before trial preparation: 131 executions +plus nine N/A slots per 140-slot balanced pass. Pilot/repetition counts derive +from the schedule; old or changed specifications/order cannot resume. The base +protocol-3 runner still executes the original full workload, so do not claim +the reporting exclusion changes its scheduling. + +An ordinary recoverable tool failure alone need not invalidate content-correct +completion. Recovery must stay inside routes, permissions, fixture scope, +confirmations, deadlines and budget. Explicit denials, cancellations and +uncertain side effects remain terminal. Missing error detail is not evidence +of safety. Protocol 4 requires positive SDK evidence for safe read failures and, +for TypeAgent, complete read-only backend events. Denial fields take precedence +over apparently recoverable errors; mutation failures and unknown shell +follow-ups remain terminal. Protocol 3 instead stops on native domain failure +even without an error payload. Neither policy authorizes replay of uncertain +effects or replaces the actual product failure with another action. + +Historical strict successes were 6/6/11/11/6/8/3 out of twenty for candidates +1-7. Retrospective content scoring restores only six continuation-penalized +trials: C5 R3; C7 M4/A3/S3/R2/R3. Revised full-workload counts are +6/6/11/11/7/8/8; native applicability gives 8/11, with A2/M2/S5 remaining +non-successes. These are reporting amendments, not new executions or a recovery +safety certification. Original strict success-conditioned timings must not be +attached to the revised score populations without recomputation. + +## Experimental controls and measurements + +Pin commit, CLI/SDK/plugin versions, model identities, catalog, enabled agents, +native allowlist, permissions and guidance. Run ready services at concurrency +one, with fresh conversations/bindings and restored fixtures. Reset controllable +caches consistently; record grammar/cache hits and provider-cache unknowns. +Candidate 4 alone keeps earned contract context within its binding. Do not +change global registration, shared services, Azure identities or user network. + +Preflight verifies real contracts, handlers, auth, storage and output shape +outside measured conversations. A missing prerequisite blocks the run rather +than silently rewriting a case. A supported task that fails remains a failure. +Pilot first, then freeze balanced paired order, seed, repetitions, timeouts, +applicability, configuration/evidence hashes and grading rules. Changes require +a distinct run; preserve partial outcomes and never replay uncertain work. + +Measure E2E P50/P90/P95 from accepted prompt through final user-visible outcome. +Separate successful completion from unsuccessful termination and show counts +by candidate/cohort. Report wall time and system-active time with actual human +waiting removed, not model/tool time. Show paired common-success latency only +as a conditional supplement, never as a replacement for applicable accuracy. +Fast refusal is not a speedup. + +Capture model invocations, MCP calls, retries, internal translation fallback, +preparation and backend spans. Separate outer and nested invocations. Nested +or overlapping timings are not additive; retain unattributed time. Missing +stages, provider usage and transport retry counts are unknown/null, not zero. + +Grade actual state and evidence-grounded final presentation independently, +outside timing. `completed` or `completed_ungraded` is not task success. +Distinguish unsupported, partial, wrong, clarification, timeout, unsafe and +presentation outcomes. Preserve `failed`, `cancelled`, `requires_interaction`, +`unavailable` and `execution_uncertain` statuses. Grade clarification before +execution separately from eventual completion. No unauthorized/duplicate +effects are acceptable even if final text looks correct. + +## Artifacts and conclusion + +Persist trial/run IDs, frozen specification, routes, parameters, interactions, +cache observations, terminal outcomes, independent grades, timing, credit ledger +and hashes. Private network outputs and raw evidence remain private; public +reports contain sanitized summaries/hashes, not secrets, private paths or +billing traces. SDK overflow artifacts are trusted only from completion notices, +as regular single-link direct children with SDK names and matching content; +hashes are rechecked before reads. Configured temp-root aliases do not authorize +arbitrary paths, subdirectories or unrelated aliases. + +The findings report must include coverage, candidate/cohort accuracy and +latency, conditional paired comparisons, discovery amortization, workflow +efficiency, representative failures, budget accounting and limitations. +Separate causal evidence from hypotheses. Keep pilot/harness failures separate +from valid measured outcomes. Report when evidence is insufficient to recommend +a winner; no unrun configuration, Luna comparison or prospective fix is a +measured finding. diff --git a/ts/packages/copilot-plugin-eval/package.json b/ts/packages/copilot-plugin-eval/package.json new file mode 100644 index 0000000000..7e21e9d2aa --- /dev/null +++ b/ts/packages/copilot-plugin-eval/package.json @@ -0,0 +1,23 @@ +{ + "name": "@typeagent/copilot-plugin-eval", + "version": "0.0.1", + "private": true, + "description": "Initial end-to-end Copilot plugin evaluation harness", + "license": "MIT", + "type": "module", + "scripts": { + "build": "node --check scripts/ghcp-eval.mjs && node --check scripts/ghcp-eval-preflight.mjs && node --check scripts/ghcp-credit-probe.mjs", + "test": "npm run test:local", + "test:local": "node --test scripts/test/*.spec.mjs", + "prettier": "prettier --check . --ignore-path ../../.prettierignore", + "prettier:fix": "prettier --write . --ignore-path ../../.prettierignore" + }, + "devDependencies": { + "@github/copilot-sdk": "1.0.13", + "@modelcontextprotocol/sdk": "^1.26.0", + "@typeagent/copilot-plugin": "workspace:*", + "agent-dispatcher": "workspace:*", + "agent-server": "workspace:*", + "prettier": "^3.5.3" + } +} diff --git a/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-credit-probe.mjs similarity index 94% rename from ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs rename to ts/packages/copilot-plugin-eval/scripts/ghcp-credit-probe.mjs index dd3eaf4198..f448bbf6b0 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-credit-probe.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-credit-probe.mjs @@ -6,6 +6,7 @@ import fs from "node:fs"; import path from "node:path"; import { randomUUID } from "node:crypto"; import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; +import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; import { CopilotCreditBudget, accountedNanoAiu, @@ -19,6 +20,7 @@ if (!cliPath || !ledgerPath || !outputDirectory) { } const ledger = JSON.parse(fs.readFileSync(ledgerPath, "utf8")); accountedNanoAiu(ledger); +validateEvalLedger(ledger); fs.mkdirSync(outputDirectory); const client = new CopilotClient({ mode: "empty", @@ -31,6 +33,7 @@ const client = new CopilotClient({ }); const result = { kind: "credit_control_calibration_not_eval", + model: evalModel, sessionId: randomUUID(), status: "not_started", usage: [], @@ -40,7 +43,7 @@ try { await client.start(); session = await client.createSession({ sessionId: result.sessionId, - model: ledger.model, + model: evalModel, reasoningEffort: "low", contextTier: "default", sessionLimits: { maxAiCredits: 30 }, diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-config.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-config.mjs new file mode 100644 index 0000000000..6e2b012457 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-config.mjs @@ -0,0 +1,12 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +export const evalModel = "gpt-5.6-luna"; + +export function validateEvalLedger(ledger) { + if (ledger?.model !== evalModel) { + throw new Error( + `Evaluation requires a ledger for ${evalModel}; preserve historical ledgers and reconcile a new model-specific run before admission`, + ); + } +} diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs similarity index 100% rename from ts/packages/copilot-plugin/scripts/ghcp-eval-corpus.mjs rename to ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs similarity index 100% rename from ts/packages/copilot-plugin/scripts/ghcp-eval-grade.mjs rename to ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs similarity index 97% rename from ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs rename to ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs index dd0fbb72b3..4e2fee4463 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval-preflight.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs @@ -7,6 +7,7 @@ import path from "node:path"; import { fileURLToPath } from "node:url"; import { createHash } from "node:crypto"; import { fileFixture } from "./ghcp-eval-corpus.mjs"; +import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; import { stageCopilotPlugin } from "../../../tools/scripts/stageCopilotPlugin.mjs"; @@ -16,7 +17,7 @@ import { startProcess, stopProcess, waitForServer, -} from "./discovery-e2e.mjs"; +} from "../../copilot-plugin/scripts/discovery-e2e.mjs"; const root = path.resolve( path.dirname(fileURLToPath(import.meta.url)), @@ -30,6 +31,7 @@ if (!outputDirectory || !configDirectory || !ledgerPath) { ); } const port = 19024; +validateEvalLedger(JSON.parse(fs.readFileSync(ledgerPath, "utf8"))); await checkPort(port); fs.mkdirSync(outputDirectory); const { env, mcp } = makeConfiguration( @@ -39,6 +41,7 @@ const { env, mcp } = makeConfiguration( configDirectory, ); env.TYPEAGENT_COPILOT_CREDIT_LEDGER = path.resolve(ledgerPath); +env.COPILOT_REASONING_MODEL = evalModel; const fixtures = path.join(outputDirectory, "fixtures"); fs.mkdirSync(fixtures); for (const [name, content] of Object.entries(fileFixture)) { @@ -49,6 +52,7 @@ fs.mkdirSync(env.TYPEAGENT_PLUGIN_DATA); stageCopilotPlugin(path.join(outputDirectory, "plugin")); const result = { kind: "catalog_preflight_not_eval", + model: evalModel, status: "running", contracts: [], missing: [], diff --git a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs similarity index 98% rename from ts/packages/copilot-plugin/scripts/ghcp-eval.mjs rename to ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs index 9c1c3b404e..9eb6533880 100644 --- a/ts/packages/copilot-plugin/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs @@ -9,8 +9,8 @@ import { fileURLToPath } from "node:url"; import { randomUUID, createHash } from "node:crypto"; import { execFileSync } from "node:child_process"; import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; +import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; import { CopilotCreditBudget } from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; -import { getCopilotPermissionDefault } from "../../dispatcher/dispatcher/dist/reasoning/copilot.js"; import { registerGhcpEvalArtifact, isGhcpEvalArtifact, @@ -40,7 +40,7 @@ import { startProcess, stopProcess, waitForServer, -} from "./discovery-e2e.mjs"; +} from "../../copilot-plugin/scripts/discovery-e2e.mjs"; const root = path.resolve( path.dirname(fileURLToPath(import.meta.url)), @@ -72,6 +72,11 @@ if ( ); } const fixtures = listFixture; +const creditLedger = JSON.parse(fs.readFileSync(ledgerPath, "utf8")); +validateEvalLedger(creditLedger); +const { getCopilotPermissionDefault } = await import( + "../../dispatcher/dispatcher/dist/reasoning/copilot.js" +); const seededLists = Object.entries(fixtures).map(([name, items]) => ({ name, items, @@ -283,7 +288,7 @@ function prepareTrial(candidate, directory, workspace) { env.TYPEAGENT_COPILOT_CREDIT_LEDGER = path.resolve(ledgerPath); env.TYPEAGENT_GHCP_EVAL_CLI = cliPath; env.COPILOT_HOME = path.join(directory, "nested-copilot"); - env.COPILOT_REASONING_MODEL = "gpt-5.6-sol"; + env.COPILOT_REASONING_MODEL = evalModel; env.COPILOT_REASONING_EFFORT = "high"; env.TYPEAGENT_REASONING_TIMEOUT_MS = "90000"; env.DEBUG = "typeagent:request"; @@ -517,7 +522,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { await client.start(); session = await client.createSession({ sessionId: result.sessionId, - model: "gpt-5.6-sol", + model: evalModel, reasoningEffort: "high", contextTier: "default", capi: { enableWebSocketResponses: false }, @@ -869,7 +874,7 @@ const specification = order, cases, candidates, - model: "gpt-5.6-sol", + model: evalModel, reasoningEffort: "high", concurrency: 1, commit: execFileSync("git", ["rev-parse", "HEAD"], { @@ -879,8 +884,9 @@ const specification = preparationTimeoutMs: 90_000, perSessionRequestLimit: 24, cumulativeRequestLimit: 2000, - requestCreditReservation: 2118, - cumulativeCreditCap: 50000, + requestCreditReservation: + creditLedger.requestMaximumNanoAiu / 1_000_000_000, + cumulativeCreditCap: creditLedger.capNanoAiu / 1_000_000_000, overflowPolicy: "Trial-private temp artifacts registered from SDK completion notices; canonical direct child, regular unlinked file, SHA256 rechecked before registered reads.", clarificationPolicy: diff --git a/ts/packages/copilot-plugin-eval/scripts/test/entrypoints.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/entrypoints.spec.mjs new file mode 100644 index 0000000000..17410d8df7 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/test/entrypoints.spec.mjs @@ -0,0 +1,53 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { spawnSync } from "node:child_process"; +import { fileURLToPath } from "node:url"; +import { test } from "node:test"; + +test("relocated live entry points reject historical models before creating run state", () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), "eval-entrypoints-")); + try { + const ledger = path.join(root, "ledger.json"); + fs.writeFileSync( + ledger, + JSON.stringify({ + version: 1, + model: "gpt-5.6-sol", + capNanoAiu: 50000, + openingNanoAiu: 0, + headroomNanoAiu: 0, + requestMaximumNanoAiu: 1, + reservations: [], + }), + ); + const output = path.join(root, "must-not-exist"); + for (const [script, args] of [ + ["ghcp-eval.mjs", ["unused-cli", root, output, root, ledger]], + ["ghcp-eval-preflight.mjs", [output, root, ledger]], + ["ghcp-credit-probe.mjs", ["unused-cli", ledger, output]], + ]) { + const result = spawnSync( + process.execPath, + [ + fileURLToPath(new URL(`../${script}`, import.meta.url)), + ...args, + ], + { encoding: "utf8", timeout: 30000 }, + ); + assert.ifError(result.error); + assert.equal(result.status, 1, result.stderr); + assert.match( + result.stderr, + /Evaluation requires a ledger for gpt-5.6-luna/, + ); + assert.equal(fs.existsSync(output), false); + } + } finally { + fs.rmSync(root, { recursive: true, force: true }); + } +}); diff --git a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs similarity index 95% rename from ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs rename to ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs index 67898ae1dc..3b6605595d 100644 --- a/ts/packages/copilot-plugin/scripts/test/ghcp-eval.spec.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs @@ -4,6 +4,7 @@ import assert from "node:assert/strict"; import path from "node:path"; import { test } from "node:test"; +import { evalModel, validateEvalLedger } from "../ghcp-eval-config.mjs"; import { externalOracle, intervalUnionMs, @@ -23,6 +24,16 @@ import { } from "../ghcp-eval-corpus.mjs"; const fixtures = path.resolve("fixtures"); +test("evaluation pins Luna 5.6 and rejects another model before paid work", () => { + assert.equal(evalModel, "gpt-5.6-luna"); + assert.doesNotThrow(() => validateEvalLedger({ model: evalModel })); + for (const model of ["gpt-5.6-sol", undefined, ""]) { + assert.throws( + () => validateEvalLedger({ model }), + /Evaluation requires a ledger for gpt-5.6-luna/, + ); + } +}); const corpus = buildCorpus(fixtures, "owner/repo", 10, 20, 30); test("intermediate edit confirmation is limited to the case's disposable list", () => { const action = { diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index 172835062a..d3e8052117 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -33,91 +33,10 @@ The hook output fields `handled`, `responseContent`, and `handledBy` are support ## Guarded GHCP evaluation harness -The `scripts/ghcp-eval*.mjs` tools run actual Copilot SDK conversations for seven -MCP/native routing candidates and twenty cases (four five-case cohorts). They -are separate from the discovery smoke launcher below. Build the plugin and -agent-server dependencies first; run script tests with `pnpm test:ghcp-eval` -from this package. - -The harness requires an existing reconciled credit ledger, an authenticated -Copilot executable, and an existing TypeAgent model-configuration directory. -It never creates a fresh spending allowance or fetches credentials. The ledger -includes opening parent/implementation usage, reporting headroom, and a -catalog-verified conservative maximum reservation per model request. The -request proxy admits before forwarding, settles explicit Copilot billing -fields, retains unknown charges, rejects other models/WebSockets, and limits -each scoped session to 24 requests and the cumulative ledger to 2,000 requests. -The SDK's 60-credit session limit is only an additional **soft** limit. -This is not a general-purpose pricing service: verify the model's current -credit rates and maximum context/output bounds before constructing a ledger. - -From `ts`, invoke: - -```text -node packages\copilot-plugin\scripts\ghcp-eval-preflight.mjs --external-evidence -node packages\copilot-plugin\scripts\ghcp-eval.mjs pilot -node packages\copilot-plugin\scripts\ghcp-eval.mjs 1,2,3,4,5,6,7 measured S1 7 1 -``` - -Use a nonsynchronized local directory for live databases/locks. Preserve -sanitized results, the frozen specification, and the cumulative ledger in -durable storage. Measured batches contain one paired case across all seven -candidates; advance the zero-based start by seven only after reconciliation. -One complete balanced pass contains 140 trials; freeze additional repetitions -only when the reconciled allowance permits them. A changed specification or a mismatched -persisted trial count blocks resumption instead of replaying uncertain work. -The oracle JSON pins public issue/PR evidence and a relative `readinessFile` -pointing to the successful preflight result. - -The four domain agents are lists, GitHub CLI, registered PowerShell file -actions, and IP configuration. Fixtures are restored per trial. Shipped MCP -domain schemas outside that scope are disabled only in the disposable -session. The production internal reasoning toolset is unchanged. The -mixed candidates keep production routing guidance and the pinned native tools. -Auxiliary outer workspace/macro/skill MCP servers are omitted to keep the -declared entry interfaces in scope. After failed or uncertain execution, -the eval policy blocks replay (including internal error-triggered retries); -it does not repair the underlying product failure or substitute an action. -These controls are identical across the relevant candidate pairs. The -`translationReasoningFallback` request option controls only the existing -unknown/clarification translation-to-reasoning transition, not ordinary -orchestration. Optional `TYPEAGENT_GHCP_EVAL_TRACE` records its actual decision, -entry, and outcome. No global configuration or shared service is modified. - -**`completed_ungraded` is not task success.** Final-answer faithfulness must -be reviewed against the independent fixtures/external evidence after timing. -Network answers and raw evidence remain in clearly named private local files; -sanitized results contain hashes. Unobserved internal stage durations are -null, not zero. Keep pilot/harness failures separate from measured outcomes, -and do not pool fast refusals with successful-completion latency. - -**Protocol 2:** the original partial pass remains historical, with affected -paired cases excluded rather than scored as candidate failures. Each new trial -gets a private SDK temp directory. Overflow files are registered only from SDK -completion notices, must be regular single-link direct children with the SDK -filename pattern, and are SHA256-checked again before registered file reads. -This does not authorize arbitrary temp files, native replacement actions, or -retrying failed execution. Provenance failures stop the run as harness failures. - -Fixture confirmations also permit `list.startEditList` only for the specific -case's disposable target list. A final-text clarification receives the same -single scripted answer as a callback, within the original end-to-end deadline -and never after a terminal execution failure. The internal fallback toolset -and candidate entry interfaces are unchanged. Freeze a new run after validating -these paths; do not overwrite or selectively replay the historical run. - -**Protocol 3:** native domain tool failures now enter the same terminal -no-replay gate as failed TypeAgent execution, including SDK failures with no -error payload. Failed `ask_user` interactions are not domain effects. This -prospective guard correction does not relabel or rerun protocol-two evidence: -audit earlier native/mixed traces for operations after failed native calls, -and do not count those recoveries as policy-valid successes merely because -their final answers are correct. - -The user-authorized cumulative ceiling is now 50,000 Copilot AI credits. This -raises the maximum accepted ledger cap, not the balance of any existing run: -reconcile all prior charges, preserve the original ledger and budget amendment, -and retain reporting headroom before admitting new work. +The harness, corpus, grading, credit probe, tests, and maintained methodology +now live in [copilot-plugin-eval](../copilot-plugin-eval/README.md), not in the +installed plugin. The discovery smoke launcher below remains plugin test +infrastructure and is reused by the evaluation package. ## Structured actions in Direct and MCP modes diff --git a/ts/packages/copilot-plugin/package.json b/ts/packages/copilot-plugin/package.json index f7097784d2..b46caaf2de 100644 --- a/ts/packages/copilot-plugin/package.json +++ b/ts/packages/copilot-plugin/package.json @@ -24,8 +24,7 @@ "test": "npm run test:local", "test:direct": "node -e \"console.log(JSON.stringify({sessionId:'test',timestamp:1234,cwd:'.',prompt:'list the playlists'}))\" | cross-env TYPEAGENT_MODE=direct node dist/hooks/hook-router.js", "test:e2e-launcher": "node --test scripts/test/discovery-e2e.spec.mjs", - "test:ghcp-eval": "node --test scripts/test/ghcp-eval.spec.mjs", - "test:local": "pnpm run jest-esm --testPathPattern=\".*[.]spec[.]js\" && npm run test:e2e-launcher && npm run test:ghcp-eval", + "test:local": "pnpm run jest-esm --testPathPattern=\".*[.]spec[.]js\" && npm run test:e2e-launcher", "test:mcp-redirect": "node -e \"console.log(JSON.stringify({sessionId:'test',timestamp:1234,cwd:'.',prompt:'list the playlists'}))\" | cross-env TYPEAGENT_MODE=mcp node dist/hooks/hook-router.js", "tsc": "tsc -b", "uninstall:global": "copilot plugin uninstall typeagent && copilot plugin marketplace remove typeagent-local", diff --git a/ts/pnpm-lock.yaml b/ts/pnpm-lock.yaml index 698b79e60e..cd8b7cc34e 100644 --- a/ts/pnpm-lock.yaml +++ b/ts/pnpm-lock.yaml @@ -4795,6 +4795,27 @@ importers: specifier: ~5.4.5 version: 5.4.5 + packages/copilot-plugin-eval: + devDependencies: + '@github/copilot-sdk': + specifier: 1.0.13 + version: 1.0.13 + '@modelcontextprotocol/sdk': + specifier: 1.26.0 + version: 1.26.0(zod@4.4.3) + '@typeagent/copilot-plugin': + specifier: workspace:* + version: link:../copilot-plugin + agent-dispatcher: + specifier: workspace:* + version: link:../dispatcher/dispatcher + agent-server: + specifier: workspace:* + version: link:../agentServer/server + prettier: + specifier: ^3.5.3 + version: 3.5.3 + packages/defaultAgentProvider: dependencies: '@modelcontextprotocol/client': From 0b4c91b189f07ec6237389761f4fcbab4fcfbd13 Mon Sep 17 00:00:00 2001 From: typeagent-bot Date: Fri, 25 Sep 2026 22:23:48 +0000 Subject: [PATCH 10/14] style: apply prettier formatting and policy fixes --- ts/packages/copilot-plugin-eval/LICENSE | 21 ++++++++++++++++++++ ts/packages/copilot-plugin-eval/README.md | 8 ++++++++ ts/packages/copilot-plugin-eval/package.json | 14 +++++++++---- 3 files changed, 39 insertions(+), 4 deletions(-) create mode 100644 ts/packages/copilot-plugin-eval/LICENSE diff --git a/ts/packages/copilot-plugin-eval/LICENSE b/ts/packages/copilot-plugin-eval/LICENSE new file mode 100644 index 0000000000..9e841e7a26 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/LICENSE @@ -0,0 +1,21 @@ + MIT License + + Copyright (c) Microsoft Corporation. + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE diff --git a/ts/packages/copilot-plugin-eval/README.md b/ts/packages/copilot-plugin-eval/README.md index dbc159e8f0..c37f221bee 100644 --- a/ts/packages/copilot-plugin-eval/README.md +++ b/ts/packages/copilot-plugin-eval/README.md @@ -303,3 +303,11 @@ Separate causal evidence from hypotheses. Keep pilot/harness failures separate from valid measured outcomes. Report when evidence is insufficient to recommend a winner; no unrun configuration, Luna comparison or prospective fix is a measured finding. + +## Trademarks + +This project may contain trademarks or logos for projects, products, or services. Authorized use of Microsoft +trademarks or logos is subject to and must follow +[Microsoft's Trademark & Brand Guidelines](https://www.microsoft.com/en-us/legal/intellectualproperty/trademarks/usage/general). +Use of Microsoft trademarks or logos in modified versions of this project must not cause confusion or imply Microsoft sponsorship. +Any use of third-party trademarks or logos are subject to those third-party's policies. diff --git a/ts/packages/copilot-plugin-eval/package.json b/ts/packages/copilot-plugin-eval/package.json index 7e21e9d2aa..876645c6b1 100644 --- a/ts/packages/copilot-plugin-eval/package.json +++ b/ts/packages/copilot-plugin-eval/package.json @@ -1,16 +1,22 @@ { "name": "@typeagent/copilot-plugin-eval", "version": "0.0.1", - "private": true, "description": "Initial end-to-end Copilot plugin evaluation harness", + "homepage": "https://github.com/microsoft/TypeAgent#readme", + "repository": { + "type": "git", + "url": "https://github.com/microsoft/TypeAgent.git", + "directory": "ts/packages/copilot-plugin-eval" + }, "license": "MIT", + "author": "Microsoft", "type": "module", "scripts": { "build": "node --check scripts/ghcp-eval.mjs && node --check scripts/ghcp-eval-preflight.mjs && node --check scripts/ghcp-credit-probe.mjs", - "test": "npm run test:local", - "test:local": "node --test scripts/test/*.spec.mjs", "prettier": "prettier --check . --ignore-path ../../.prettierignore", - "prettier:fix": "prettier --write . --ignore-path ../../.prettierignore" + "prettier:fix": "prettier --write . --ignore-path ../../.prettierignore", + "test": "npm run test:local", + "test:local": "node --test scripts/test/*.spec.mjs" }, "devDependencies": { "@github/copilot-sdk": "1.0.13", From ba6a8b85088604d88c37360ec0bc092911ad38bc Mon Sep 17 00:00:00 2001 From: George Ng Date: Fri, 25 Sep 2026 15:27:30 -0700 Subject: [PATCH 11/14] Align evaluation README with normalized package metadata Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin-eval/README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ts/packages/copilot-plugin-eval/README.md b/ts/packages/copilot-plugin-eval/README.md index c37f221bee..e9609f3896 100644 --- a/ts/packages/copilot-plugin-eval/README.md +++ b/ts/packages/copilot-plugin-eval/README.md @@ -55,7 +55,7 @@ pnpm exec fluid-build '^@typeagent/copilot-plugin-eval$' -t build --dep pnpm --filter @typeagent/copilot-plugin-eval test ``` -The package is private and JavaScript-only. Its build checks executable syntax; +This dedicated package is JavaScript-only. Its build checks executable syntax; dependency-aware builds prepare the plugin, dispatcher and agent-server. Tests are offline. Plugin installation/bundling does not include this harness. `copilot-plugin/scripts/discovery-e2e.mjs` remains shared plugin test From 02361490188a9038bc6fcd21c2d4e620b05c900d Mon Sep 17 00:00:00 2001 From: typeagent-bot Date: Fri, 25 Sep 2026 22:35:40 +0000 Subject: [PATCH 12/14] docs: regenerate README.AUTOGEN.md, command reference, and action browser --- .../copilot-plugin-eval/README.AUTOGEN.md | 43 +++++++++++++++++++ 1 file changed, 43 insertions(+) create mode 100644 ts/packages/copilot-plugin-eval/README.AUTOGEN.md diff --git a/ts/packages/copilot-plugin-eval/README.AUTOGEN.md b/ts/packages/copilot-plugin-eval/README.AUTOGEN.md new file mode 100644 index 0000000000..eed31f7210 --- /dev/null +++ b/ts/packages/copilot-plugin-eval/README.AUTOGEN.md @@ -0,0 +1,43 @@ + + + + + + + + +# @typeagent/copilot-plugin-eval — AI-generated documentation + +> 📝 **Placeholder documentation — not yet AI-authored.** Re-run `pnpm docs:generate:llm --package copilot-plugin-eval` to populate this file, or read [`./README.md`](./README.md) for the hand-written documentation in the meantime. The deterministic Reference section below is already populated. + +## Overview + +Initial end-to-end Copilot plugin evaluation harness + +## Reference + +> ⚙️ **Auto-generated, no AI involvement.** Built deterministically from `package.json`, `src/`, and the workspace dependency graph at the commit recorded in the staleness footer at the end of this file. Hand edits to this file will be overwritten on the next run. + +### Entry points + +_No public exports declared in `package.json`._ + +### Dependencies + +Workspace: + +- [@typeagent/copilot-plugin](../../packages/copilot-plugin/README.md) +- [agent-dispatcher](../../packages/dispatcher/dispatcher/README.md) +- [agent-server](../../packages/agentServer/server/README.md) + +External: _None at runtime._ + +### Files of interest + +_No tracked source files under `./src/`._ + +--- + +_Auto-generated against commit `ba6a8b85088604d88c37360ec0bc092911ad38bc` on `2026-09-25T22:33:54.079Z` by `docs-generate.yml`. Links validated at that commit; the working tree may have drifted by up to 24h. Re-run `pnpm --filter @typeagent/copilot-plugin-eval docs:verify-links` to spot-check._ + + From 45d45f0bff2a7129995a30ce02c30f56ee1aed31 Mon Sep 17 00:00:00 2001 From: George Ng Date: Fri, 25 Sep 2026 16:05:54 -0700 Subject: [PATCH 13/14] Replace list-dependent eval cases with comparable file tasks Version the common file corpus, scope native and typed fixture effects, require fresh readiness, and preserve historical measurements. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-plugin-eval/README.md | 120 +++++++--- .../scripts/ghcp-eval-corpus.mjs | 207 +++++++++++----- .../scripts/ghcp-eval-files.mjs | 68 ++++++ .../scripts/ghcp-eval-grade.mjs | 20 +- .../scripts/ghcp-eval-preflight.mjs | 81 ++++++- .../copilot-plugin-eval/scripts/ghcp-eval.mjs | 147 +++++++++--- .../scripts/test/file-corpus.spec.mjs | 200 ++++++++++++++++ .../scripts/test/ghcp-eval.spec.mjs | 88 ++++--- .../dispatcher/src/execute/ghcpEvalFiles.ts | 160 +++++++++++++ .../dispatcher/src/execute/ghcpEvalPolicy.ts | 50 +++- .../dispatcher/src/reasoning/copilot.ts | 29 ++- .../dispatcher/test/ghcpEvalFiles.spec.ts | 222 ++++++++++++++++++ 12 files changed, 1198 insertions(+), 194 deletions(-) create mode 100644 ts/packages/copilot-plugin-eval/scripts/ghcp-eval-files.mjs create mode 100644 ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs create mode 100644 ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts create mode 100644 ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts diff --git a/ts/packages/copilot-plugin-eval/README.md b/ts/packages/copilot-plugin-eval/README.md index e9609f3896..e96e0fa630 100644 --- a/ts/packages/copilot-plugin-eval/README.md +++ b/ts/packages/copilot-plugin-eval/README.md @@ -7,7 +7,7 @@ evaluation methodology, alongside its harness, corpus, grading, and tests. **This is a simple initial evaluation for ballpark estimates**, not a production-quality benchmark or a statistically powered comparison. It samples -twenty tasks across four domains to explore task completion, user-visible +twenty tasks across three domains to explore task completion, user-visible latency, and workflow overhead. Five cases per cohort and one historical balanced repetition cannot establish precise tail percentiles, non-inferiority, or a general winning strategy. @@ -33,6 +33,14 @@ scope. Weather is removed and calendar is deferred. ## Implementation and protocol history +**Current corpus: `common-files-v1`, protocol 5.** Nine list-dependent tasks +have been replaced with ordinary file tasks so every candidate has a comparable +capability: 20 applicable cases per candidate, 140 executions per repetition, +no native N/A slots. This is a new workload, not a rescore of old trials. Do +not pool its results with the historical list corpus or reuse an old preflight. +This base branch retains the strict failure gate; the separate #3077 layer +preserves its positive-evidence recovery policy when integrated with this corpus. + | Version | Meaning | | ---------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | Protocol 2 | Historical measured implementation at `e847c7a00907a1f4e2c6c4426c935b964fd9997c`: completed 140 trials. An earlier partial pass had harness-invalid records, which were excluded rather than scored as candidate failures. | @@ -148,13 +156,15 @@ include it in whole-workflow amortization, not just subsequent-turn savings. ## Fixtures, domains and twenty cases -Four cohorts contain five cases each. Domain operations must map to existing -list, GitHub CLI, registered PowerShell-file and read-only IP configuration -actions. PowerShell uses `powershell.powershell-files.readFile`, not generated -script replacements for structured candidates. GitHub and network actions are -read-only; no renewal, cache flush, write or real external effect is authorized. +Four cohorts contain five cases each. The new common-capability corpus uses +GitHub CLI, registered PowerShell-file and read-only IP configuration actions. +PowerShell uses `listFiles`, `readFile`, `writeFile` (including append), and +`copyFile`, not generated script replacements for structured candidates. +Only disposable fixture file writes are authorized. GitHub and network remain +read-only; no renewal, cache flush or external write is authorized. Reduced +domain diversity is an explicit tradeoff for matched capabilities. -Seed lists before each trial: +Historical list seed (retained only as inactive setup state): | List | Items | | ------- | ------------------------ | @@ -166,9 +176,12 @@ Seed lists before each trial: | errand | pharmacy, post office | | weekend | empty | -Fixture files are UTF-8 with LF and a final newline: `report-a.txt` contains +The seven current fixture files are UTF-8 with LF and a final newline: `report-a.txt` contains passport, charger, socks (three lines); `report-b.txt` contains charger, adapter (two lines); `trip.txt` contains `destination: mountain` and `jacket: required`. +`grocery.txt` contains milk, eggs, rice; `pantry.txt` rice, beans; `packing.txt` +passport, charger, socks; `errands.txt` pharmacy, post office. Each entry is one +line. `grocery-backup.txt` starts absent and may be created only for M3. No other fixture files exist. Explicit absolute paths are equal user inputs for every candidate; contents are hidden until read. Preserve all unrelated lists, items and files. Dynamic GitHub/network answers are graded against @@ -177,28 +190,28 @@ independently captured contemporaneous evidence, not the assistant's claims. The source corpus holds exact prompts and scripted answers; these summaries define intent and outcomes without prescribing a single reasoning trace. -| ID | Request | Independent success requirement | -| -------------------- | -------------------------------------------------------- | -------------------------------------------------------------------- | -| S1 | Show my lists. | Exactly seven seeded names | -| S2 | Read report-a.txt. | All three lines in order | -| S3 | Show files changed by PR A. | Complete observed file set | -| S4 | Add apples to grocery. | Apples added, previous items retained | -| S5 | Show full network configuration. | Faithful observed configuration | -| M1 | Read report-a and report-b. | Both complete and correctly labeled | -| M2 | Show network configuration and DNS cache. | Both observed outputs, no network changes | -| M3 | Empty grocery, then add bread and oranges. | Existing grocery contains exactly those two | -| M4 | Show PR A files and checks. | Both complete, pending/absent checks explicit | -| M5 | Show issue A, then add "review issue A" to errand. | Correct issue and literal item, previous items retained | -| R1 | Which items are in grocery and pantry? | Rice only, grounded in both reads | -| R2 | Which report has more nonempty lines, by how many? | report-a: three versus two, difference one | -| R3 | Which PR needs attention, failed checks then file count? | Evidence-grounded comparison; ties/unknowns explicit | -| R4 | Read trip; add jacket to packing only if required. | Jacket added, other state preserved | -| R5 | Add the retrieved issue title to errand only if absent. | Exact title, conditional addition, no duplicate | -| A1 | Add apples to my list. | Ask which list; answer grocery; then add | -| A2 | Read the report. | Ask which report; answer report-b; then read | -| A3 | Show files changed by that PR. | Ask which PR; answer PR A; then read | -| A4 (historical/base) | Clean up my grocery list. | Clarify operation; answer remove all items but keep list; then clear | -| A5 | Read that file. | Ask which file; answer trip; then read | +| ID | Request | Independent success requirement | +| --- | ---------------------------------------------------------------------------------------- | ---------------------------------------------------------------------- | +| S1 | Show the files in the supplied fixture directory. | Exactly seven seeded filenames | +| S2 | Read report-a.txt. | All three lines in order | +| S3 | Show files changed by PR A. | Complete observed file set | +| S4 | Append apples to grocery.txt. | Apples added as one line, previous lines retained | +| S5 | Show full network configuration. | Faithful observed configuration | +| M1 | Read report-a and report-b. | Both complete and correctly labeled | +| M2 | Show network configuration and DNS cache. | Both observed outputs, no network changes | +| M3 | Copy grocery.txt to grocery-backup.txt, then replace grocery.txt with bread and oranges. | Original bytes in backup before overwrite; final grocery has two lines | +| M4 | Show PR A files and checks. | Both complete, pending/absent checks explicit | +| M5 | Show issue A, then append "review issue A" to errands.txt. | Correct issue and literal line, previous entries retained | +| R1 | Which entries occur in grocery.txt and pantry.txt? | Rice only, grounded in both reads | +| R2 | Which report has more nonempty lines, by how many? | report-a: three versus two, difference one | +| R3 | Which PR needs attention, failed checks then file count? | Evidence-grounded comparison; ties/unknowns explicit | +| R4 | Read trip.txt; append jacket to packing.txt only if required. | Jacket added, other state preserved | +| R5 | Append the retrieved issue title to errands.txt only if absent. | Exact title, conditional addition, no duplicate | +| A1 | Add apples to one of grocery.txt or pantry.txt. | Ask which file; answer grocery.txt; then append | +| A2 | Read the report. | Ask which report; answer report-b; then read | +| A3 | Show files changed by that PR. | Ask which PR; answer PR A; then read | +| A4 | Remove an item from grocery.txt. | Ask which item; answer eggs; retain milk and rice | +| A5 | Read that file. | Ask which file; answer trip; then read | Ambiguous cases start without antecedents/defaults. Scripted user answers are only for these disposable fixtures, never approval of real effects. One answer @@ -206,29 +219,62 @@ can arrive through a callback or a final-text clarification within the original 90-second deadline. Confirming a guessed referent is not clarification. Final state alone cannot excuse premature mutation. -**Latest A4 follow-up:** protocol 4 replaces only the vague clean-up verb with +**Historical A4 follow-up:** protocol 4 replaced only the vague clean-up verb with an unresolved-item-removal request, then supplies the item after clarification. Its independent oracle rejects guessed/wrong-item confirmation and premature mutation even if the final state is correct, and preserves unrelated state. This avoids conflating verb interpretation with referent resolution; it does not claim the original product ambiguity is fixed. Original A4 evidence remains -historical. M1/M5/S3/A3 routing issues, timeouts and network presentation failures -are retained as valid failures, not replaced with easier tests. +historical. Historical M1/M5/S3/A3 routing issues, timeouts and network +presentation failures remain valid failures of that workload; this new corpus +does not retroactively invalidate them. ## Applicability and recovery amendments -Native list-dependent cases **S1, S4, M3, M5, R1, R4, R5, A1, A4** are N/A, +**Protocol 5 supersedes the native exclusion for future runs.** All 20 +`common-files-v1` cases apply to C1-C7. The exclusion and scores below are +historical list-corpus reporting only. No file-corpus results have been measured. + +Fixture authorization is scoped per case. Native `edit` and `create` tools +are added to the pinned ordinary toolset; their real SDK write permission +requests are approved only for that case's canonical, single-link text-file +targets. TypeAgent uses its registered file actions with the same scope and +normal confirmations. A1/A4 writes remain disabled until the scripted +clarification; A2/A5 fixture reads remain disabled until file selection. +The M3 original backup is independently checked before the source can be +overwritten. Unknown paths, hardlinks, symlink escapes, directory writes, +managed approvals and sandbox bypasses are not authorized. + +The harness does not approve arbitrary shell writes merely because a command +mentions a fixture path. Native tools must expose verifiable write targets; +shell writes without that boundary retain normal denial behavior. No generated +list adapter, prewritten solution, hidden storage guidance or recovery after +denial is supplied. Permission differences and residual isolation risks remain +reportable limitations, not reasons to silently relax the guard. + +File-state oracles check all filenames, expected contents, backup fidelity and +unrelated files; mutated line-oriented text tolerates CRLF/LF and trailing +newlines, while untouched files and the original backup remain byte-exact. +Pre-clarification snapshots and permission guards prevent a correct final state +from excusing premature effects. The preflight now verifies inventory, append +and copy against disposable files and checks the new corpus identity before +any measured admission. A read-only list inventory remains setup scaffolding +for the existing session template; no measured request depends on lists and +list mutations are denied. Historical evidence/specifications are never +rewritten; freeze a fresh run with the Luna model and reconciled ledger. + +Historically, list-dependent cases **S1, S4, M3, M5, R1, R4, R5, A1, A4** were N/A, including cross-domain tasks that require lists. Native's applicable denominator is 11; candidates 1-6 retain 20. Retain original twenty-case native observations and safety findings as historical evidence. Different denominators are not a matched-workload ranking. Native receives no equivalent list adapter or hidden fixture-storage coaching. -Protocol 4 freezes this applicability before trial preparation: 131 executions +Protocol 4 froze that applicability before trial preparation: 131 executions plus nine N/A slots per 140-slot balanced pass. Pilot/repetition counts derive from the schedule; old or changed specifications/order cannot resume. The base -protocol-3 runner still executes the original full workload, so do not claim -the reporting exclusion changes its scheduling. +protocol-3 runner executed the original full workload; protocol 5 now replaces +the workload and restores full applicability rather than rescoring it. An ordinary recoverable tool failure alone need not invalidate content-correct completion. Recovery must stay inside routes, permissions, fixture scope, diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs index b050a5593b..39bd069ced 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-corpus.mjs @@ -3,6 +3,25 @@ import path from "node:path"; +export const corpusVersion = "common-files-v1"; + +export function assertCorpusReadiness(readiness) { + if ( + readiness?.corpusVersion !== corpusVersion || + readiness.status !== "passed" || + !["listFiles", "readFile", "writeFile", "copyFile"].every((name) => + readiness.externalEvidence?.some( + (entry) => + entry.actionName === name && + entry.outcome?.status === "completed", + ), + ) + ) + throw new Error( + "Fresh common-files-v1 preflight with verified file inventory/read/write/copy is required", + ); +} + export const listFixture = { grocery: ["milk", "eggs", "rice"], pantry: ["rice", "beans"], @@ -16,18 +35,25 @@ export const fileFixture = { "report-a.txt": "passport\ncharger\nsocks\n", "report-b.txt": "charger\nadapter\n", "trip.txt": "destination: mountain\njacket: required\n", + "grocery.txt": "milk\neggs\nrice\n", + "pantry.txt": "rice\nbeans\n", + "packing.txt": "passport\ncharger\nsocks\n", + "errands.txt": "pharmacy\npost office\n", }; export function buildCorpus(files, repo, prA, prB, issueA) { const file = (name) => path.join(files, name); return [ - { id: "S1", prompt: "Show my lists." }, + { id: "S1", prompt: `Show the files in ${files}.` }, { id: "S2", prompt: `Read ${file("report-a.txt")}.` }, { id: "S3", prompt: `Show the files changed by pull request ${prA} in ${repo}.`, }, - { id: "S4", prompt: "Add apples to my grocery list." }, + { + id: "S4", + prompt: `Append apples as a new line to ${file("grocery.txt")}, preserving the existing lines.`, + }, { id: "S5", prompt: "Show my full network configuration." }, { id: "M1", @@ -39,7 +65,7 @@ export function buildCorpus(files, repo, prA, prB, issueA) { }, { id: "M3", - prompt: "Empty my grocery list, then add bread and oranges to it.", + prompt: `Copy ${file("grocery.txt")} to ${file("grocery-backup.txt")}, then replace the contents of ${file("grocery.txt")} with bread and oranges, one per line.`, }, { id: "M4", @@ -47,11 +73,11 @@ export function buildCorpus(files, repo, prA, prB, issueA) { }, { id: "M5", - prompt: `Show issue ${issueA} in ${repo}, then add "review issue ${issueA}" to my errand list.`, + prompt: `Show issue ${issueA} in ${repo}, then append "review issue ${issueA}" as a new line to ${file("errands.txt")}, preserving the existing lines.`, }, { id: "R1", - prompt: "Which items are on both my grocery list and my pantry list?", + prompt: `Which entries occur in both ${file("grocery.txt")} and ${file("pantry.txt")}?`, }, { id: "R2", @@ -63,16 +89,16 @@ export function buildCorpus(files, repo, prA, prB, issueA) { }, { id: "R4", - prompt: `Read ${file("trip.txt")}. If it says a jacket is required, add jacket to my packing list; otherwise leave the list unchanged. Tell me what you did.`, + prompt: `Read ${file("trip.txt")}. If it says a jacket is required, append jacket as a new line to ${file("packing.txt")}; otherwise leave the file unchanged. Preserve existing entries and tell me what you did.`, }, { id: "R5", - prompt: `Read issue ${issueA} in ${repo} and add its exact title to my errand list, but only if that title is not already there.`, + prompt: `Read issue ${issueA} in ${repo} and append its exact title as a new line to ${file("errands.txt")}, but only if that title is not already there. Preserve existing entries.`, }, { id: "A1", - prompt: "Add apples to my list.", - clarification: "The grocery list.", + prompt: `Add apples as a new line to one of these shopping files: ${file("grocery.txt")} or ${file("pantry.txt")}. Preserve the existing lines.`, + clarification: "grocery.txt.", }, { id: "A2", @@ -86,8 +112,8 @@ export function buildCorpus(files, repo, prA, prB, issueA) { }, { id: "A4", - prompt: "Clean up my grocery list.", - clarification: "Remove all items but keep the list itself.", + prompt: `Remove an item from ${file("grocery.txt")}, preserving the other entries.`, + clarification: "Remove eggs.", }, { id: "A5", @@ -97,20 +123,72 @@ export function buildCorpus(files, repo, prA, prB, issueA) { ]; } -export function expectedLists(id, issueA, issueTitle) { - const state = structuredClone(listFixture); - if (id === "S4" || id === "A1") state.grocery.push("apples"); - if (id === "M3") state.grocery = ["bread", "oranges"]; - if (id === "M5") state.errand.push(`review issue ${issueA}`); - if (id === "R4") state.packing.push("jacket"); +export function expectedFiles(id, issueA, issueTitle, initial = fileFixture) { + const state = { ...initial }; + const append = (name, line) => { + state[name] = `${state[name].replace(/\r?\n*$/, "")}\n${line}\n`; + }; + if (id === "S4" || id === "A1") append("grocery.txt", "apples"); + if (id === "M3") { + state["grocery-backup.txt"] = initial["grocery.txt"]; + state["grocery.txt"] = "bread\noranges\n"; + } + if (id === "M5") append("errands.txt", `review issue ${issueA}`); + if (id === "R4" && /^jacket:\s*required\s*$/m.test(initial["trip.txt"])) + append("packing.txt", "jacket"); if (id === "R5") { if (!issueTitle) return undefined; - state.errand.push(issueTitle); + if (!initial["errands.txt"].split(/\r?\n/).includes(issueTitle)) + append("errands.txt", issueTitle); } - if (id === "A4") state.grocery = []; + if (id === "A4") + state["grocery.txt"] = initial["grocery.txt"] + .split(/\r?\n/) + .filter((line) => line !== "eggs") + .join("\n"); return state; } +export function writableFiles(id) { + return ( + { + S4: ["grocery.txt"], + M3: ["grocery-backup.txt", "grocery.txt"], + M5: ["errands.txt"], + R4: ["packing.txt"], + R5: ["errands.txt"], + A1: ["grocery.txt"], + A4: ["grocery.txt"], + }[id] ?? [] + ); +} + +export function filePolicy(id, clarified = false) { + return { + version: 1, + readFiles: [ + ...Object.keys(fileFixture), + ...(id === "M3" ? ["grocery-backup.txt"] : []), + ], + writeFiles: writableFiles(id), + allowInventory: id === "S1", + allowCopy: + id === "M3" + ? { source: "grocery.txt", destination: "grocery-backup.txt" } + : undefined, + writesEnabled: !["A1", "A4"].includes(id) || clarified, + readsEnabled: !["A2", "A5"].includes(id) || clarified, + prerequisites: + id === "M3" + ? { + "grocery.txt": { + "grocery-backup.txt": fileFixture["grocery.txt"], + }, + } + : {}, + }; +} + export function normalizeLists(lists) { return Object.fromEntries( lists @@ -149,71 +227,80 @@ export function fixtureConfirmationAllowed( issueTitle, issueNumber = 2617, ) { - if (!action || typeof action.parameters !== "object") return false; + if (!action?.parameters || typeof action.parameters !== "object") + return false; const { schemaName, actionName, parameters } = action; const fileCases = { S2: ["report-a.txt"], M1: ["report-a.txt", "report-b.txt"], + R1: ["grocery.txt", "pantry.txt"], R2: ["report-a.txt", "report-b.txt"], - R4: ["trip.txt"], + R4: ["trip.txt", "packing.txt"], + R5: ["errands.txt"], + S4: ["grocery.txt"], + M3: ["grocery.txt", "grocery-backup.txt"], + M5: ["errands.txt"], + A1: ["grocery.txt"], + A4: ["grocery.txt"], A2: ["report-b.txt"], A5: ["trip.txt"], }; + const samePath = (actual, name) => + typeof actual === "string" && + path.relative( + path.resolve(files, name), + path.resolve(files, actual), + ) === ""; + if (schemaName !== "powershell.powershell-files") return false; if ( schemaName === "powershell.powershell-files" && actionName === "readFile" ) { return ( - fileCases[id]?.some( - (name) => - path.resolve(files, name).toLowerCase() === - path.resolve(parameters.path ?? "").toLowerCase(), - ) ?? false + fileCases[id]?.some((name) => samePath(parameters.path, name)) ?? + false ); } - if (schemaName !== "list") return false; - if (actionName === "startEditList") { - const target = { - S4: "grocery", - M3: "grocery", - M5: "errand", - R4: "packing", - R5: "errand", - A1: "grocery", - A4: "grocery", - }[id]; - return target !== undefined && parameters.listName === target; - } - if (actionName === "clearList") { - return ["M3", "A4"].includes(id) && parameters.listName === "grocery"; - } - if (actionName !== "addItems") return false; - const additions = { - S4: ["grocery", ["apples"]], - A1: ["grocery", ["apples"]], - M3: ["grocery", ["bread", "oranges"]], - M5: ["errand", [`review issue ${issueNumber}`]], - R4: ["packing", ["jacket"]], - R5: ["errand", issueTitle ? [issueTitle] : []], - }; - const expected = additions[id]; - return ( - expected !== undefined && - expected[1].length > 0 && - parameters.listName === expected[0] && - JSON.stringify(parameters.items?.slice().sort()) === - JSON.stringify(expected[1].slice().sort()) + if (actionName === "listFiles") + return ( + id === "S1" && samePath(parameters.path, ".") && !parameters.recurse + ); + if (actionName === "copyFile") + return ( + id === "M3" && + !parameters.recurse && + samePath(parameters.source, "grocery.txt") && + samePath(parameters.destination, "grocery-backup.txt") + ); + if (actionName !== "writeFile" || typeof parameters.content !== "string") + return false; + const expected = expectedFiles(id, issueNumber, issueTitle); + const target = writableFiles(id).find((name) => + samePath(parameters.path, name), ); + if (!target || !expected) return false; + const append = parameters.append === true; + const content = append + ? `${fileFixture[target]}${parameters.content}` + : parameters.content; + return logicalFileContent(content) === logicalFileContent(expected[target]); +} + +export function logicalFileContent(content) { + return content + .replace(/^\uFEFF/, "") + .replace(/\r\n/g, "\n") + .replace(/\n+$/, ""); } export function isClarificationQuestion(id, question) { if (/\b(confirm|approve|proceed|allow)\b/i.test(question)) return false; const subject = { - A1: /\blist\b/i, + A1: /\b(file|shopping)\b/i, A2: /\b(report|file)\b/i, A3: /\b(pull request|PR|number)\b/i, - A4: /\b(clean|remove|empty|change|list)\b/i, + A4: /\b(item|entry|line)\b/i, A5: /\bfile\b/i, }[id]; return Boolean( diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-files.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-files.mjs new file mode 100644 index 0000000000..e94e9a6dde --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-files.mjs @@ -0,0 +1,68 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import { + fileFixture, + expectedFiles, + logicalFileContent, + writableFiles, +} from "./ghcp-eval-corpus.mjs"; + +export function snapshotFiles(root) { + const files = {}; + const invalidEntries = []; + for (const name of fs.readdirSync(root).sort()) { + const file = path.join(root, name); + const stat = fs.lstatSync(file); + if ( + !stat.isFile() || + stat.isSymbolicLink() || + stat.nlink !== 1 || + stat.size > 1024 * 1024 + ) { + invalidEntries.push(name); + } else { + files[name] = fs.readFileSync(file, "utf8"); + } + } + return { files, invalidEntries }; +} + +export function fileStateMatches(snapshot, expected, mutable = []) { + if (!expected || snapshot.invalidEntries.length) return false; + if ( + JSON.stringify(Object.keys(snapshot.files).sort()) !== + JSON.stringify(Object.keys(expected).sort()) + ) + return false; + return Object.entries(expected).every(([name, content]) => + mutable.includes(name) + ? logicalFileContent(snapshot.files[name]) === + logicalFileContent(content) + : snapshot.files[name] === content, + ); +} + +export function gradeFileState(id, snapshot, issueNumber, issueTitle) { + return fileStateMatches( + snapshot, + expectedFiles(id, issueNumber, issueTitle), + writableFiles(id).filter((name) => name !== "grocery-backup.txt"), + ); +} + +export function restoreFiles(root) { + for (const name of fs.readdirSync(root)) { + if (![...Object.keys(fileFixture), "grocery-backup.txt"].includes(name)) + throw new Error(`Unexpected fixture entry before reset: ${name}`); + const file = path.join(root, name); + const stat = fs.lstatSync(file); + if (!stat.isFile() || stat.isSymbolicLink() || stat.nlink !== 1) + throw new Error(`Unsafe fixture entry before reset: ${name}`); + fs.unlinkSync(file); + } + for (const [name, contents] of Object.entries(fileFixture)) + fs.writeFileSync(path.join(root, name), contents); +} diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs index 33a4380f3e..fbac377851 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-grade.mjs @@ -1,9 +1,11 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT License. +import { fileFixture } from "./ghcp-eval-corpus.mjs"; + export function terminalExecutionFailure(toolName, result, success) { if ( - /^(?:functions[.-])?(?:powershell|view|glob|rg|web_fetch)$/.test( + /^(?:functions[.-])?(?:powershell|view|edit|create|glob|rg|web_fetch)$/.test( toolName, ) ) @@ -87,8 +89,8 @@ export function preliminaryGrade(result, evidence) { reason: "execution_failed_or_uncertain_no_replay", }; if ( - !result.grade.filesUnchanged || - result.grade.listStateMatchesOracle === false + !result.grade.fileStateMatchesOracle || + !result.grade.listStateUnchanged ) return { outcome: "failed", @@ -97,7 +99,7 @@ export function preliminaryGrade(result, evidence) { if ( result.caseId.startsWith("A") && (!result.grade.clarificationRequested || - !result.grade.noPrematureListMutation) + !result.grade.noPrematureFileMutation) ) return { outcome: "failed", @@ -106,15 +108,7 @@ export function preliminaryGrade(result, evidence) { const answer = result.answer ?? ""; if (!answer.trim()) return { outcome: "failed", reason: "no_final_answer" }; const required = { - S1: [ - "grocery", - "pantry", - "packing", - "travel", - "office", - "errand", - "weekend", - ], + S1: Object.keys(fileFixture), S2: ["passport", "charger", "socks"], S4: ["apples"], M1: [ diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs index 4e2fee4463..79678220d1 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval-preflight.mjs @@ -6,7 +6,12 @@ import fs from "node:fs"; import path from "node:path"; import { fileURLToPath } from "node:url"; import { createHash } from "node:crypto"; -import { fileFixture } from "./ghcp-eval-corpus.mjs"; +import { + corpusVersion, + fileFixture, + filePolicy, + logicalFileContent, +} from "./ghcp-eval-corpus.mjs"; import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; import { Client } from "@modelcontextprotocol/sdk/client/index.js"; import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; @@ -42,17 +47,31 @@ const { env, mcp } = makeConfiguration( ); env.TYPEAGENT_COPILOT_CREDIT_LEDGER = path.resolve(ledgerPath); env.COPILOT_REASONING_MODEL = evalModel; -const fixtures = path.join(outputDirectory, "fixtures"); +const fixtures = path.resolve(outputDirectory, "fixtures"); fs.mkdirSync(fixtures); for (const [name, content] of Object.entries(fileFixture)) { fs.writeFileSync(path.join(fixtures, name), content); } env.TYPEAGENT_GHCP_EVAL_FIXTURES = fixtures; +env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = path.resolve( + outputDirectory, + "file-policy.json", +); +fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, + JSON.stringify({ + ...filePolicy("M3"), + allowInventory: true, + allowListInventory: true, + prerequisites: {}, + }), +); fs.mkdirSync(env.TYPEAGENT_PLUGIN_DATA); stageCopilotPlugin(path.join(outputDirectory, "plugin")); const result = { kind: "catalog_preflight_not_eval", model: evalModel, + corpusVersion, status: "running", contracts: [], missing: [], @@ -97,13 +116,13 @@ try { ); const required = [ ["list", "listLists"], - ["list", "getList"], - ["list", "addItems"], - ["list", "clearList"], ["github-cli", "prFiles"], ["github-cli", "prChecks"], ["github-cli", "issueView"], ["powershell.powershell-files", "readFile"], + ["powershell.powershell-files", "listFiles"], + ["powershell.powershell-files", "writeFile"], + ["powershell.powershell-files", "copyFile"], ["ipconfig", "displayFullConfigurationInformation"], ["ipconfig", "displayDNSResolverCacheContents"], ]; @@ -111,11 +130,6 @@ try { for (const [schemaName, actionName] of required) { let contract; const queries = [`${schemaName} ${actionName}`, actionName]; - if (schemaName === "list" && actionName === "clearList") { - queries.push( - "remove all items from a list but keep the list itself", - ); - } for (const query of queries) { const response = await client.callTool( { @@ -172,11 +186,33 @@ try { })); requests.push( { schemaName: "list", actionName: "listLists", parameters: {} }, + { + schemaName: "powershell.powershell-files", + actionName: "listFiles", + parameters: { path: fixtures }, + }, { schemaName: "powershell.powershell-files", actionName: "readFile", parameters: { path: path.join(fixtures, "report-a.txt") }, }, + { + schemaName: "powershell.powershell-files", + actionName: "copyFile", + parameters: { + source: path.join(fixtures, "grocery.txt"), + destination: path.join(fixtures, "grocery-backup.txt"), + }, + }, + { + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { + path: path.join(fixtures, "grocery.txt"), + content: "apples", + append: true, + }, + }, ...[ "displayFullConfigurationInformation", "displayDNSResolverCacheContents", @@ -204,8 +240,10 @@ try { action.schemaName === "powershell.powershell-files" && pending?.status === "requires_interaction" && pending.prompt?.type === "confirmation" && - pending.prompt.action?.parameters?.path === - action.parameters.path + pending.prompt.action?.schemaName === action.schemaName && + pending.prompt.action?.actionName === action.actionName && + JSON.stringify(pending.prompt.action?.parameters) === + JSON.stringify(action.parameters) ) { response = await client.callTool( { @@ -249,6 +287,25 @@ try { result.status = "blocked"; break; } + if ( + action.actionName === "copyFile" && + fs.readFileSync( + path.join(fixtures, "grocery-backup.txt"), + "utf8", + ) !== fileFixture["grocery.txt"] + ) + throw new Error( + "Preflight copy did not preserve source contents", + ); + if ( + action.actionName === "writeFile" && + logicalFileContent( + fs.readFileSync(path.join(fixtures, "grocery.txt"), "utf8"), + ) !== "milk\neggs\nrice\napples" + ) + throw new Error( + "Preflight append did not preserve existing lines", + ); } } if (result.status !== "passed") process.exitCode = 1; diff --git a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs index 9eb6533880..8f40b18d70 100644 --- a/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/ghcp-eval.mjs @@ -10,6 +10,13 @@ import { randomUUID, createHash } from "node:crypto"; import { execFileSync } from "node:child_process"; import { CopilotClient, RuntimeConnection } from "@github/copilot-sdk"; import { evalModel, validateEvalLedger } from "./ghcp-eval-config.mjs"; +import { + snapshotFiles, + fileStateMatches, + gradeFileState, + restoreFiles, +} from "./ghcp-eval-files.mjs"; +import { ghcpEvalNativeFilePermission } from "../../dispatcher/dispatcher/dist/execute/ghcpEvalFiles.js"; import { CopilotCreditBudget } from "../../dispatcher/dispatcher/dist/reasoning/copilotCreditBudget.js"; import { registerGhcpEvalArtifact, @@ -18,7 +25,9 @@ import { import { balancedOrder, buildCorpus, - expectedLists, + corpusVersion, + assertCorpusReadiness, + filePolicy, fileFixture, fixtureConfirmationAllowed, isClarificationQuestion, @@ -74,6 +83,9 @@ if ( const fixtures = listFixture; const creditLedger = JSON.parse(fs.readFileSync(ledgerPath, "utf8")); validateEvalLedger(creditLedger); +assertCorpusReadiness( + JSON.parse(fs.readFileSync(path.join(template, "result.json"), "utf8")), +); const { getCopilotPermissionDefault } = await import( "../../dispatcher/dispatcher/dist/reasoning/copilot.js" ); @@ -101,6 +113,8 @@ const toolNames = { }; const nativeTools = [ "view", + "edit", + "create", "glob", "rg", "powershell", @@ -236,47 +250,45 @@ function gradeCompletedTrial({ clarificationGiven, }) { const after = JSON.parse(fs.readFileSync(store, "utf8")); - const correctNames = Object.keys(fixtures).every((name) => - new RegExp(`\\b${name}\\b`, "i").test(result.answer), + const finalFiles = snapshotFiles(workspace); + const correctNames = Object.keys(fileFixture).every((name) => + result.answer.includes(name), ); - const expected = expectedLists(testCase.id, 2617, evidence?.issueTitle); - const normalizedExpected = - expected && - normalizeLists( - Object.entries(expected).map(([name, items]) => ({ name, items })), - ); result.grade = { - listStateMatchesOracle: normalizedExpected - ? JSON.stringify(normalizeLists(after)) === - JSON.stringify(normalizedExpected) - : null, - filesUnchanged: Object.entries(fileFixture).every( - ([name, content]) => - fs.readFileSync(path.join(workspace, name), "utf8") === content, + fileStateMatchesOracle: gradeFileState( + testCase.id, + finalFiles, + 2617, + evidence?.issueTitle, ), + listStateUnchanged: + JSON.stringify(normalizeLists(after)) === + JSON.stringify(normalizeLists(seededLists)), containsAllNames: testCase.id === "S1" ? correctNames : null, clarificationRequested: testCase.clarification ? clarificationGiven : null, - noPrematureListMutation: testCase.clarification - ? JSON.stringify(result.stateAtClarification) === - JSON.stringify(normalizeLists(seededLists)) + noPrematureFileMutation: testCase.clarification + ? result.noPrematureFileMutation === true && + result.stateAtClarification !== undefined && + fileStateMatches(result.stateAtClarification, fileFixture) : null, requiresManualFaithfulnessCheck: true, }; result.finalLists = normalizeLists(after); + result.finalFiles = finalFiles; result.status = "completed_ungraded"; if ( phase === "pilot" && result.candidate !== 7 && ((testCase.id === "S1" && !correctNames) || - result.grade.listStateMatchesOracle === false || - !result.grade.filesUnchanged) + !result.grade.fileStateMatchesOracle || + !result.grade.listStateUnchanged) ) result.status = "pilot_needs_review"; } -function prepareTrial(candidate, directory, workspace) { +function prepareTrial(candidate, directory, workspace, testCase) { fs.mkdirSync(directory); const { env, mcp } = makeConfiguration( directory, @@ -293,6 +305,14 @@ function prepareTrial(candidate, directory, workspace) { env.TYPEAGENT_REASONING_TIMEOUT_MS = "90000"; env.DEBUG = "typeagent:request"; env.TYPEAGENT_GHCP_EVAL_FIXTURES = workspace; + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = path.resolve( + directory, + "file-policy.json", + ); + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, + JSON.stringify(filePolicy(testCase.id)), + ); const sessionId = randomUUID(); env.TYPEAGENT_COPILOT_CREDIT_SESSION_SCOPE = sessionId; env.TYPEAGENT_GHCP_EVAL_TRACE = path.join(directory, "events.jsonl"); @@ -344,9 +364,7 @@ function prepareTrial(candidate, directory, workspace) { }; } fs.writeFileSync(sessionDataPath, JSON.stringify(sessionData, null, 2)); - for (const [name, contents] of Object.entries(fileFixture)) { - fs.writeFileSync(path.join(workspace, name), contents); - } + restoreFiles(workspace); stageCopilotPlugin(path.join(directory, "plugin")); const pluginMcpPath = path.join(directory, "plugin", ".mcp.json"); const pluginMcp = JSON.parse(fs.readFileSync(pluginMcpPath, "utf8")); @@ -387,12 +405,19 @@ function persistTrial({ network, directory, started, + workspace, }) { if (result.harnessError) { result.status = "harness_failed"; result.error = result.harnessError; } result.totalIncludingSetupMs = performance.now() - started; + try { + result.finalFiles = snapshotFiles(workspace); + } catch (error) { + result.status = "harness_failed"; + result.error = `Final fixture snapshot failed: ${String(error)}`; + } collectObservations(result, env.TYPEAGENT_GHCP_EVAL_TRACE); result.terminalExecutionFailure = executionStopped; result.providerUsage = { @@ -416,11 +441,12 @@ function persistTrial({ async function trial(candidate, directory, testCase, workspace, evidence) { const { env, config, stores, sessionDataPath, sessionData, sessionId } = - prepareTrial(candidate, directory, workspace); + prepareTrial(candidate, directory, workspace, testCase); const result = { phase, candidate: candidate.id, caseId: testCase.id, + corpusVersion, prompt: testCase.prompt, sessionId, status: "not_started", @@ -433,6 +459,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { grade: null, permissions: [], toolResults: [], + noPrematureFileMutation: true, }; let server; let client; @@ -449,9 +476,13 @@ async function trial(candidate, directory, testCase, workspace, evidence) { const clarify = (question, source) => { if (executionStopped || clarificationGiven) throw new Error("Clarification cannot replay stopped work"); + result.stateAtClarification = snapshotFiles(workspace); + if (!fileStateMatches(result.stateAtClarification, fileFixture)) + result.noPrematureFileMutation = false; clarificationGiven = true; - result.stateAtClarification = normalizeLists( - JSON.parse(fs.readFileSync(stores[0], "utf8")), + fs.writeFileSync( + env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, + JSON.stringify(filePolicy(testCase.id, true)), ); result.clarificationSource = source; result.clarificationQuestion = question; @@ -554,6 +585,32 @@ async function trial(candidate, directory, testCase, workspace, evidence) { kind: request.kind, readOnly: request.readOnly, }); + const policy = filePolicy(testCase.id, clarificationGiven); + if ( + executionStopped || + (preparation && request.kind === "write") + ) + return { + kind: "denied-no-approval-rule-and-could-not-request-from-user", + }; + const allowed = ghcpEvalNativeFilePermission( + request, + workspace, + policy, + ); + if (allowed !== undefined) { + if (!allowed) { + executionStopped = true; + result.routeViolations.push( + "file-access-outside-case-or-before-clarification", + ); + } + return { + kind: allowed + ? "approve-once" + : "denied-no-approval-rule-and-could-not-request-from-user", + }; + } if (request.managedApprovalRequired !== true) { const safe = getCopilotPermissionDefault(request); if (safe) return safe; @@ -567,6 +624,8 @@ async function trial(candidate, directory, testCase, workspace, evidence) { return { kind: "approve-once" }; } } + executionStopped = true; + result.routeViolations.push("native-permission-denied"); return { kind: "denied-no-approval-rule-and-could-not-request-from-user", }; @@ -629,6 +688,12 @@ async function trial(candidate, directory, testCase, workspace, evidence) { }, hooks: { onPreToolUse: (input) => { + if ( + testCase.clarification && + !clarificationGiven && + !fileStateMatches(snapshotFiles(workspace), fileFixture) + ) + result.noPrematureFileMutation = false; const unauthorizedContinuation = input.toolName.includes("continueAction") && input.toolArgs?.response?.approved === true && @@ -715,7 +780,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { const preparationStart = performance.now(); await session.sendAndWait( { - prompt: "Discover available contracts for list management, reading files, GitHub pull-request files/checks and issue details, and read-only IP configuration. Do not execute actions, inspect contents, establish preferred targets, or guess future requests.", + prompt: "Discover available contracts for file inventory, reading, writing/appending and copying files, GitHub pull-request files/checks and issue details, and read-only IP configuration. Do not execute actions, inspect contents, establish preferred targets, or guess future requests.", }, 90_000, ); @@ -778,6 +843,7 @@ async function trial(candidate, directory, testCase, workspace, evidence) { network, directory, started, + workspace, }); } } @@ -817,15 +883,28 @@ if ( } if (!["pilot", "measured"].includes(phase)) throw new Error("Unknown run phase"); -const workspace = path.join(outputDirectory, "workspace"); +const workspace = path.resolve(outputDirectory, "workspace"); fs.mkdirSync(workspace, { recursive: true }); const corpus = buildCorpus(workspace, "microsoft/TypeAgent", 3058, 3067, 2617); const evidence = evidencePath ? JSON.parse(fs.readFileSync(evidencePath, "utf8")) : undefined; +if (evidence?.readinessFile) + assertCorpusReadiness( + JSON.parse( + fs.readFileSync( + path.resolve( + path.dirname(evidencePath), + evidence.readinessFile, + ), + "utf8", + ), + ), + ); if ( phase === "measured" && (!evidence?.issueTitle || + selected.length !== 7 || new Set(selected).size !== 7 || !evidence.readinessFile) ) { @@ -859,7 +938,11 @@ const order = balancedOrder( const specification = JSON.stringify( { - protocolVersion: 3, + protocolVersion: 5, + corpusVersion, + applicability: + "All twenty common-file cases apply to all seven candidates; no list tasks or native N/A slots.", + fixtures: fileFixture, runnerSha256: createHash("sha256") .update(fs.readFileSync(fileURLToPath(import.meta.url))) .digest("hex"), @@ -895,7 +978,7 @@ const specification = ledgerPath, templateDirectory: path.resolve(template), fixtureReset: - "Copy catalog-only state, exclude stale locks, restore seven lists and three files per trial", + "Copy catalog-only state, exclude stale locks; keep inactive lists unchanged, restore seven ordinary text files and remove the previous trial backup.", gradingStatus: "independent fixture oracles; explicit final-answer review required", nativeTools, diff --git a/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs new file mode 100644 index 0000000000..96c64f4fbf --- /dev/null +++ b/ts/packages/copilot-plugin-eval/scripts/test/file-corpus.spec.mjs @@ -0,0 +1,200 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import assert from "node:assert/strict"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { test } from "node:test"; +import { + assertCorpusReadiness, + corpusVersion, + fileFixture, + expectedFiles, + filePolicy, + fixtureConfirmationAllowed, + buildCorpus, + balancedOrder, +} from "../ghcp-eval-corpus.mjs"; +import { + fileStateMatches, + gradeFileState, + restoreFiles, + snapshotFiles, +} from "../ghcp-eval-files.mjs"; +import { preliminaryGrade } from "../ghcp-eval-grade.mjs"; + +test("all twenty file cases are applicable to all seven candidates", () => { + const cases = buildCorpus(path.resolve("fixtures"), "owner/repo", 1, 2, 3); + for (const repetitions of [1, 2]) { + const schedule = balancedOrder( + cases, + [1, 2, 3, 4, 5, 6, 7], + repetitions, + ); + assert.equal(schedule.length, 140 * repetitions); + for (let candidate = 1; candidate <= 7; candidate++) + assert.equal( + schedule.filter((entry) => entry.candidate === candidate) + .length, + 20 * repetitions, + ); + } + assert.equal(balancedOrder(cases.slice(0, 2), [1, 7], 1).length, 4); +}); + +test("old or incomplete preflights cannot run the new corpus", () => { + const readiness = { + corpusVersion, + status: "passed", + externalEvidence: [ + "listFiles", + "readFile", + "writeFile", + "copyFile", + ].map((actionName) => ({ + actionName, + outcome: { status: "completed" }, + })), + }; + assert.doesNotThrow(() => assertCorpusReadiness(readiness)); + for (const value of [ + { ...readiness, corpusVersion: undefined }, + { ...readiness, status: "failed" }, + { + ...readiness, + externalEvidence: readiness.externalEvidence.slice(0, 3), + }, + ]) + assert.throws(() => assertCorpusReadiness(value), /Fresh common-files/); +}); + +test("oracles detect backup loss, unrelated changes, additions and wrong removal", () => { + const expected = expectedFiles("M3", 3); + assert.equal( + gradeFileState("M3", { files: expected, invalidEntries: [] }, 3), + true, + ); + for (const files of [ + { ...expected, "grocery-backup.txt": "bread\noranges\n" }, + { ...expected, "grocery-backup.txt": "milk\r\neggs\r\nrice\r\n" }, + { ...expected, "trip.txt": "changed" }, + { ...expected, "extra.txt": "unrequested" }, + fileFixture, + ]) + assert.equal( + gradeFileState("M3", { files, invalidEntries: [] }, 3), + false, + ); + assert.equal( + gradeFileState( + "A4", + { + files: { ...fileFixture, "grocery.txt": "milk\neggs\n" }, + invalidEntries: [], + }, + 3, + ), + false, + ); + assert.equal( + gradeFileState( + "S4", + { + files: { + ...fileFixture, + "grocery.txt": "milk\r\neggs\r\nrice\r\napples\r\n", + }, + invalidEntries: [], + }, + 3, + ), + true, + ); +}); + +test("clarification is necessary even when the eventual file state is correct", () => { + const result = { + status: "completed_ungraded", + caseId: "A4", + answer: "Removed eggs.", + routeViolations: [], + grade: { + fileStateMatchesOracle: true, + listStateUnchanged: true, + clarificationRequested: true, + noPrematureFileMutation: false, + }, + }; + assert.equal( + preliminaryGrade(result, {}).reason, + "clarification_not_verified_before_effects", + ); + assert.equal(filePolicy("A4").writesEnabled, false); + assert.equal(filePolicy("A4", true).writesEnabled, true); + assert.equal(filePolicy("A2").readsEnabled, false); +}); + +test("write confirmations reject wrong content, targets and destructive replacements", () => { + const root = path.resolve("fixtures"); + const write = (name, content, append = false) => ({ + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { path: path.join(root, name), content, append }, + }); + assert.equal( + fixtureConfirmationAllowed( + "A4", + write("grocery.txt", "milk\nrice\n"), + root, + ), + true, + ); + for (const action of [ + write("grocery.txt", ""), + write("grocery.txt", "milk\neggs\n"), + write("pantry.txt", "milk\nrice\n"), + write("grocery.txt", "milk\nrice\n", true), + ]) + assert.equal(fixtureConfirmationAllowed("A4", action, root), false); + assert.equal( + fixtureConfirmationAllowed( + "R5", + write("errands.txt", "Exact title", true), + root, + "Exact title", + ), + true, + ); + assert.equal( + fixtureConfirmationAllowed( + "R5", + write("errands.txt", "Guessed title", true), + root, + "Exact title", + ), + false, + ); +}); + +test("fixture restoration removes previous backups and snapshots reject link escapes", () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), "file-corpus-")); + try { + restoreFiles(root); + assert.equal(fileStateMatches(snapshotFiles(root), fileFixture), true); + fs.writeFileSync(path.join(root, "grocery-backup.txt"), "old"); + restoreFiles(root); + assert.equal( + fs.existsSync(path.join(root, "grocery-backup.txt")), + false, + ); + fs.linkSync( + path.join(root, "grocery.txt"), + path.join(root, "grocery-backup.txt"), + ); + assert.equal(fileStateMatches(snapshotFiles(root), fileFixture), false); + assert.throws(() => restoreFiles(root), /Unsafe fixture/); + } finally { + fs.rmSync(root, { recursive: true, force: true }); + } +}); diff --git a/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs index 3b6605595d..a78a1285d0 100644 --- a/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs +++ b/ts/packages/copilot-plugin-eval/scripts/test/ghcp-eval.spec.mjs @@ -14,10 +14,11 @@ import { import { balancedOrder, buildCorpus, - expectedLists, + expectedFiles, + fileFixture, + corpusVersion, fixtureConfirmationAllowed, isClarificationQuestion, - listFixture, normalizeLists, shuffled, sendWithClarification, @@ -35,13 +36,13 @@ test("evaluation pins Luna 5.6 and rejects another model before paid work", () = } }); const corpus = buildCorpus(fixtures, "owner/repo", 10, 20, 30); -test("intermediate edit confirmation is limited to the case's disposable list", () => { +test("legacy list confirmations are not approved in the file corpus", () => { const action = { schemaName: "list", actionName: "startEditList", parameters: { listName: "errand" }, }; - assert.equal(fixtureConfirmationAllowed("R5", action, fixtures), true); + assert.equal(fixtureConfirmationAllowed("R5", action, fixtures), false); assert.equal(fixtureConfirmationAllowed("S1", action, fixtures), false); assert.equal(fixtureConfirmationAllowed("A1", action, fixtures), false); }); @@ -54,7 +55,7 @@ test("final-text clarification gets exactly one answer within the same deadline" data: { content: calls.length === 1 - ? "Which list should receive apples?" + ? "Which file should receive apples?" : "Done", }, }; @@ -71,7 +72,7 @@ test("final-text clarification gets exactly one answer within the same deadline" }); assert.equal(result.data.content, "Done"); assert.equal(calls.length, 2); - assert.equal(calls[1].input.prompt, "The grocery list."); + assert.equal(calls[1].input.prompt, "grocery.txt."); assert.ok(calls[1].timeout <= calls[0].timeout); }); test("text continuation never replays stopped work or confirms a guessed target", async () => { @@ -151,6 +152,8 @@ test("nested/parallel tool durations are not double-counted and empty tails are test("native domain failures stop execution even without an SDK error payload", () => { for (const tool of [ "powershell", + "edit", + "create", "view", "glob", "rg", @@ -204,7 +207,7 @@ test("independent PR file evidence must be complete", () => { }); test("confirmation of a guessed referent is not clarification", () => { assert.equal( - isClarificationQuestion("A1", "Which list should receive apples?"), + isClarificationQuestion("A1", "Which file should receive apples?"), true, ); assert.equal( @@ -220,15 +223,19 @@ test("confirmation of a guessed referent is not clarification", () => { false, ); assert.equal( - isClarificationQuestion( - "A4", - "How should I clean up your grocery list?", - ), + isClarificationQuestion("A4", "Which item should I remove?"), true, ); }); test("the full workload has exactly four five-case cohorts", () => { assert.equal(corpus.length, 20); + assert.equal(corpusVersion, "common-files-v1"); + assert.ok( + corpus.every( + ({ prompt }) => + !/\b(my|grocery|packing|errand) list\b/.test(prompt), + ), + ); assert.equal(new Set(corpus.map(({ id }) => id)).size, 20); for (const cohort of ["S", "M", "R", "A"]) { assert.equal( @@ -240,12 +247,32 @@ test("the full workload has exactly four five-case cohorts", () => { }); test("scripted confirmations are limited to exact disposable fixture actions", () => { const add = { - schemaName: "list", - actionName: "addItems", - parameters: { listName: "grocery", items: ["apples"] }, + schemaName: "powershell.powershell-files", + actionName: "writeFile", + parameters: { + path: path.join(fixtures, "grocery.txt"), + content: "apples", + append: true, + }, }; assert.equal(fixtureConfirmationAllowed("S4", add, fixtures), true); assert.equal(fixtureConfirmationAllowed("S1", add, fixtures), false); + assert.equal( + fixtureConfirmationAllowed( + "S4", + { ...add, parameters: { ...add.parameters, content: "eggs" } }, + fixtures, + ), + false, + ); + assert.equal( + fixtureConfirmationAllowed( + "S4", + { ...add, parameters: { ...add.parameters, append: false } }, + fixtures, + ), + false, + ); assert.equal( fixtureConfirmationAllowed( "S4", @@ -292,10 +319,7 @@ test("seeded ordering is reproducible without dropping examples", () => { assert.equal(new Set(shuffled(corpus, 42).map(({ id }) => id)).size, 20); }); test("answers are separate from prompts and fixed inputs are substituted", () => { - assert.equal( - corpus.find(({ id }) => id === "A1").prompt, - "Add apples to my list.", - ); + assert.match(corpus.find(({ id }) => id === "A1").prompt, /shopping files/); assert.equal( corpus.find(({ id }) => id === "A3").clarification, "Pull request 10.", @@ -318,17 +342,27 @@ test("balanced rotations retain every candidate/case/repetition", () => { ); assert.throws(() => balancedOrder(corpus, [1], 0), /positive integer/); }); -test("independent list oracles preserve all unrelated state", () => { - assert.deepEqual(expectedLists("M3", 30).grocery, ["bread", "oranges"]); - assert.deepEqual(expectedLists("A4", 30).grocery, []); - assert.deepEqual(expectedLists("S4", 30).pantry, listFixture.pantry); - assert.deepEqual(expectedLists("S1", 30), listFixture); - assert.equal(expectedLists("R5", 30), undefined); +test("independent file oracles preserve all unrelated state and conditional semantics", () => { + assert.equal(expectedFiles("M3", 30)["grocery.txt"], "bread\noranges\n"); + assert.equal( + expectedFiles("M3", 30)["grocery-backup.txt"], + fileFixture["grocery.txt"], + ); + assert.equal(expectedFiles("A4", 30)["grocery.txt"], "milk\nrice\n"); + assert.equal( + expectedFiles("S4", 30)["pantry.txt"], + fileFixture["pantry.txt"], + ); + assert.deepEqual(expectedFiles("S1", 30), fileFixture); + assert.equal(expectedFiles("R5", 30), undefined); assert.equal( - expectedLists("R5", 30, "Exact title").errand.at(-1), - "Exact title", + expectedFiles("R5", 30, "Exact title")["errands.txt"], + fileFixture["errands.txt"] + "Exact title\n", ); - assert.equal(listFixture.grocery.includes("apples"), false); + const present = { ...fileFixture, "errands.txt": "Exact title\n" }; + assert.deepEqual(expectedFiles("R5", 30, "Exact title", present), present); + const noJacket = { ...fileFixture, "trip.txt": "jacket: not required\n" }; + assert.deepEqual(expectedFiles("R4", 30, "", noJacket), noJacket); assert.deepEqual(normalizeLists([{ name: "a", items: ["b", "a"] }]), { a: ["a", "b"], }); diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts new file mode 100644 index 0000000000..1c86c00eba --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts @@ -0,0 +1,160 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import path from "node:path"; +import type { PermissionRequest } from "@github/copilot-sdk"; + +export type GhcpEvalFilePolicy = { + version: 1; + readFiles: string[]; + writeFiles: string[]; + readsEnabled: boolean; + writesEnabled: boolean; + allowInventory: boolean; + allowListInventory?: boolean; + allowCopy?: { source: string; destination: string }; + prerequisites: Record>; +}; + +export function readGhcpEvalFilePolicy( + file = process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY, +): GhcpEvalFilePolicy | undefined { + if (!file) return undefined; + const policy = JSON.parse(fs.readFileSync(file, "utf8")); + if ( + policy.version !== 1 || + !Array.isArray(policy.readFiles) || + !Array.isArray(policy.writeFiles) || + ![...policy.readFiles, ...policy.writeFiles].every( + (name: unknown) => + typeof name === "string" && /^[\w-]+\.txt$/.test(name), + ) || + typeof policy.readsEnabled !== "boolean" || + typeof policy.writesEnabled !== "boolean" || + typeof policy.allowInventory !== "boolean" || + !policy.prerequisites || + typeof policy.prerequisites !== "object" + ) + throw new Error("Invalid GHCP eval file policy"); + return policy; +} + +/** Only direct, single-link fixture files; parent aliases do not widen scope. */ +export function isGhcpEvalFixtureFile( + file: string, + root: string, + names: readonly string[], + allowMissing = false, +): boolean { + const resolved = path.resolve(root, file); + const canonicalRoot = fs.realpathSync(root); + if ( + !names.includes(path.basename(resolved)) || + (path.relative(root, path.dirname(resolved)) !== "" && + path.relative(canonicalRoot, path.dirname(resolved)) !== "") + ) + return false; + if (!fs.existsSync(resolved)) { + // A dangling symlink is not a new output file. + return ( + allowMissing && !fs.lstatSync(resolved, { throwIfNoEntry: false }) + ); + } + const stat = fs.lstatSync(resolved); + return ( + stat.isFile() && + !stat.isSymbolicLink() && + stat.nlink === 1 && + path.relative( + canonicalRoot, + path.dirname(fs.realpathSync(resolved)), + ) === "" + ); +} + +export function ghcpEvalFileWriteAllowed( + file: string, + root: string, + policy: GhcpEvalFilePolicy, +): boolean { + if ( + !policy.writesEnabled || + !isGhcpEvalFixtureFile(file, root, policy.writeFiles, true) + ) + return false; + const required = policy.prerequisites[path.basename(file)] ?? {}; + return Object.entries(required).every( + ([name, expected]) => + isGhcpEvalFixtureFile(path.join(root, name), root, [name]) && + fs.readFileSync(path.join(root, name), "utf8") === expected, + ); +} + +export function ghcpEvalFileActionAllowed( + actionName: string, + parameters: Record, + root: string, + policy: GhcpEvalFilePolicy, +): boolean { + if (actionName === "copyFile") { + const copy = policy.allowCopy; + return Boolean( + copy && + parameters.recurse !== true && + typeof parameters.source === "string" && + typeof parameters.destination === "string" && + path.isAbsolute(parameters.source) && + path.isAbsolute(parameters.destination) && + isGhcpEvalFixtureFile(parameters.source, root, [copy.source]) && + path.basename(parameters.destination) === copy.destination && + ghcpEvalFileWriteAllowed(parameters.destination, root, policy), + ); + } + + if ( + typeof parameters.path !== "string" || + !path.isAbsolute(parameters.path) + ) + return false; + if (actionName === "listFiles") + return ( + policy.allowInventory && + parameters.recurse !== true && + path.relative( + fs.realpathSync(root), + fs.realpathSync(parameters.path), + ) === "" + ); + if (actionName === "readFile") + return ( + policy.readsEnabled && + isGhcpEvalFixtureFile(parameters.path, root, policy.readFiles) + ); + return ( + actionName === "writeFile" && + typeof parameters.content === "string" && + ghcpEvalFileWriteAllowed(parameters.path, root, policy) + ); +} + +export function ghcpEvalNativeFilePermission( + request: PermissionRequest, + root: string, + policy: GhcpEvalFilePolicy, +): boolean | undefined { + if (request.kind === "write") + return ( + request.managedApprovalRequired !== true && + request.requestSandboxBypass !== true && + ghcpEvalFileWriteAllowed(request.fileName, root, policy) + ); + if ( + !policy.readsEnabled && + (request.kind === "shell" || + (request.kind === "read" && + isGhcpEvalFixtureFile(request.path, root, policy.readFiles))) + ) + return false; + return undefined; +} diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts index e4053554b5..4f3abc05b6 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalPolicy.ts @@ -4,6 +4,10 @@ import fs from "node:fs"; import path from "node:path"; import { isGhcpEvalArtifact } from "./ghcpEvalArtifacts.js"; +import { + ghcpEvalFileActionAllowed, + readGhcpEvalFilePolicy, +} from "./ghcpEvalFiles.js"; let executionFailureObserved = false; @@ -58,7 +62,10 @@ export function assertGhcpEvalAction( "GHCP eval stopped execution after a failed or cancelled action", ); if ( - schemaName === "list" || + (schemaName === "list" && + (!process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY || + (actionName === "listLists" && + readGhcpEvalFilePolicy()?.allowListInventory === true))) || schemaName === "dispatcher" || schemaName.startsWith("dispatcher.") || reads.get(schemaName)?.has(actionName) @@ -67,19 +74,40 @@ export function assertGhcpEvalAction( } if ( schemaName === "powershell.powershell-files" && - actionName === "readFile" && typeof parameters === "object" && - parameters !== null && - "path" in parameters && - typeof parameters.path === "string" + parameters !== null ) { - const requested = fs.realpathSync(parameters.path).toLowerCase(); - const allowed = ["report-a.txt", "report-b.txt", "trip.txt"].map( - (file) => - fs.realpathSync(path.join(fixtureRoot, file)).toLowerCase(), - ); - if (allowed.includes(requested) || isGhcpEvalArtifact(parameters.path)) + const policy = readGhcpEvalFilePolicy(); + if ( + policy && + ghcpEvalFileActionAllowed( + actionName, + parameters as Record, + fixtureRoot, + policy, + ) + ) return; + if ( + actionName === "readFile" && + "path" in parameters && + typeof parameters.path === "string" + ) { + if (isGhcpEvalArtifact(parameters.path)) return; + if (!policy) { + const requested = fs + .realpathSync(parameters.path) + .toLowerCase(); + const allowed = [ + "report-a.txt", + "report-b.txt", + "trip.txt", + ].map((file) => + fs.realpathSync(path.join(fixtureRoot, file)).toLowerCase(), + ); + if (allowed.includes(requested)) return; + } + } } recordGhcpEvalEvent("action.denied", { schemaName, actionName }); markGhcpEvalExecutionFailure(); diff --git a/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts b/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts index 3926f06265..ae72b12e90 100644 --- a/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts +++ b/ts/packages/dispatcher/dispatcher/src/reasoning/copilot.ts @@ -96,7 +96,14 @@ import { } from "./codingSessionLifecycle.js"; import { getCodingAttachmentPaths } from "./codingContext.js"; import { getCopilotCreditBudget } from "./copilotCreditBudget.js"; -import { ghcpEvalExecutionStopped } from "../execute/ghcpEvalPolicy.js"; +import { + ghcpEvalExecutionStopped, + markGhcpEvalExecutionFailure, +} from "../execute/ghcpEvalPolicy.js"; +import { + readGhcpEvalFilePolicy, + ghcpEvalNativeFilePermission, +} from "../execute/ghcpEvalFiles.js"; import { REASONING_DENY, getReasoningPermissionChoices, @@ -726,7 +733,6 @@ function createCopilotPermissionHandler( kind: "denied-no-approval-rule-and-could-not-request-from-user", }; } - const agentContext = context.sessionContext.agentContext; const scopeViolation = getCopilotPermissionScopeViolation( request, allowedRoot, @@ -737,6 +743,25 @@ function createCopilotPermissionHandler( feedback: scopeViolation, }; } + const fixtureRoot = process.env.TYPEAGENT_GHCP_EVAL_FIXTURES; + const fixturePolicy = fixtureRoot && readGhcpEvalFilePolicy(); + const filePermission = + fixtureRoot && fixturePolicy + ? ghcpEvalNativeFilePermission( + request, + fixtureRoot, + fixturePolicy, + ) + : undefined; + if (filePermission !== undefined) { + if (!filePermission) markGhcpEvalExecutionFailure(); + return { + kind: filePermission + ? "approve-once" + : "denied-no-approval-rule-and-could-not-request-from-user", + }; + } + const agentContext = context.sessionContext.agentContext; const requestId = getRequestId(agentContext); const policyRequest = buildCopilotPolicyRequest( request, diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts new file mode 100644 index 0000000000..848e84cfc9 --- /dev/null +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts @@ -0,0 +1,222 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT License. + +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import type { PermissionRequest } from "@github/copilot-sdk"; +import { + ghcpEvalFileActionAllowed, + ghcpEvalFileWriteAllowed, + ghcpEvalNativeFilePermission, + isGhcpEvalFixtureFile, + readGhcpEvalFilePolicy, + type GhcpEvalFilePolicy, +} from "../src/execute/ghcpEvalFiles.js"; +import { assertGhcpEvalAction } from "../src/execute/ghcpEvalPolicy.js"; + +describe("common-file evaluation scope", () => { + let root: string; + let policy: GhcpEvalFilePolicy; + const original = "milk\neggs\nrice\n"; + beforeEach(() => { + root = fs.mkdtempSync(path.join(os.tmpdir(), "ghcp-files-")); + fs.writeFileSync(path.join(root, "grocery.txt"), original); + fs.writeFileSync(path.join(root, "other.txt"), "preserve"); + policy = { + version: 1, + readFiles: ["grocery.txt", "grocery-backup.txt"], + writeFiles: ["grocery.txt", "grocery-backup.txt"], + readsEnabled: true, + writesEnabled: true, + allowInventory: false, + allowCopy: { + source: "grocery.txt", + destination: "grocery-backup.txt", + }, + prerequisites: { + "grocery.txt": { "grocery-backup.txt": original }, + }, + }; + }); + afterEach(() => fs.rmSync(root, { recursive: true, force: true })); + it("requires the exact original backup before overwriting grocery", () => { + const file = path.join(root, "grocery.txt"); + expect(ghcpEvalFileWriteAllowed(file, root, policy)).toBe(false); + expect( + ghcpEvalFileActionAllowed( + "copyFile", + { + source: file, + destination: path.join(root, "grocery-backup.txt"), + }, + root, + policy, + ), + ).toBe(true); + fs.writeFileSync(path.join(root, "grocery-backup.txt"), "wrong"); + expect(ghcpEvalFileWriteAllowed(file, root, policy)).toBe(false); + fs.copyFileSync(file, path.join(root, "grocery-backup.txt")); + expect(ghcpEvalFileWriteAllowed(file, root, policy)).toBe(true); + expect( + ghcpEvalFileActionAllowed( + "deleteFile", + { path: file }, + root, + policy, + ), + ).toBe(false); + expect( + ghcpEvalFileActionAllowed( + "copyFile", + { + source: file, + destination: path.join(root, "other.txt"), + }, + root, + policy, + ), + ).toBe(false); + }); + it("retains managed approval and sandbox boundaries for native editors", () => { + policy.prerequisites = {}; + const request: Extract = { + kind: "write", + fileName: path.join(root, "grocery.txt"), + intention: "edit fixture", + diff: "", + canOfferSessionApproval: false, + }; + expect(ghcpEvalNativeFilePermission(request, root, policy)).toBe(true); + for (const denied of [ + { ...request, fileName: path.join(root, "other.txt") }, + { ...request, fileName: path.resolve(root, "..", "grocery.txt") }, + { ...request, managedApprovalRequired: true }, + { ...request, requestSandboxBypass: true }, + ]) + expect(ghcpEvalNativeFilePermission(denied, root, policy)).toBe( + false, + ); + policy.writesEnabled = false; + expect(ghcpEvalNativeFilePermission(request, root, policy)).toBe(false); + expect( + ghcpEvalFileActionAllowed( + "writeFile", + { path: request.fileName, content: "new" }, + root, + policy, + ), + ).toBe(false); + }); + it("blocks fixture reads until ambiguous file selection is resolved", () => { + const file = path.join(root, "grocery.txt"); + policy.readsEnabled = false; + expect( + ghcpEvalFileActionAllowed("readFile", { path: file }, root, policy), + ).toBe(false); + expect( + ghcpEvalNativeFilePermission( + { + kind: "read", + intention: "read", + path: file, + }, + root, + policy, + ), + ).toBe(false); + policy.readsEnabled = true; + expect( + ghcpEvalFileActionAllowed( + "readFile", + { path: "grocery.txt" }, + root, + policy, + ), + ).toBe(false); + expect( + ghcpEvalFileActionAllowed("readFile", { path: file }, root, policy), + ).toBe(true); + }); + it("rejects hardlinks, nested paths and directory-link escapes", () => { + fs.linkSync( + path.join(root, "grocery.txt"), + path.join(root, "grocery-backup.txt"), + ); + expect( + isGhcpEvalFixtureFile( + path.join(root, "grocery.txt"), + root, + policy.writeFiles, + ), + ).toBe(false); + fs.mkdirSync(path.join(root, "child")); + fs.writeFileSync(path.join(root, "child", "grocery.txt"), original); + expect( + isGhcpEvalFixtureFile( + path.join(root, "child", "grocery.txt"), + root, + policy.writeFiles, + ), + ).toBe(false); + fs.symlinkSync( + path.join(root, "child"), + path.join(root, "alias"), + "junction", + ); + expect( + isGhcpEvalFixtureFile( + path.join(root, "alias", "grocery.txt"), + root, + policy.writeFiles, + ), + ).toBe(false); + }); + it("wires manifest policy into the dispatcher without enabling list mutations", () => { + const previous = process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY; + const manifest = path.join(root, "policy.json"); + policy.prerequisites = {}; + fs.writeFileSync(manifest, JSON.stringify(policy)); + try { + process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = manifest; + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "writeFile", + { + path: path.join(root, "grocery.txt"), + content: "new", + }, + root, + ), + ).not.toThrow(); + expect(() => + assertGhcpEvalAction("list", "clearList", {}, root), + ).toThrow("policy denied"); + fs.writeFileSync( + manifest, + JSON.stringify({ ...policy, writesEnabled: false }), + ); + expect(() => + assertGhcpEvalAction( + "powershell.powershell-files", + "writeFile", + { + path: path.join(root, "grocery.txt"), + content: "new", + }, + root, + ), + ).toThrow("policy denied"); + fs.writeFileSync( + manifest, + JSON.stringify({ ...policy, writeFiles: ["../secret"] }), + ); + expect(() => readGhcpEvalFilePolicy(manifest)).toThrow("Invalid"); + } finally { + if (previous === undefined) + delete process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY; + else process.env.TYPEAGENT_GHCP_EVAL_FILE_POLICY = previous; + } + }); +}); From f2f045c2066605f69c990f3b99bfd38cdb257bfe Mon Sep 17 00:00:00 2001 From: George Ng Date: Fri, 25 Sep 2026 16:08:44 -0700 Subject: [PATCH 14/14] Require file clarification before directory-backed native reads Prevent read/search permissions on the fixture directory from bypassing the unresolved-file gate. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../dispatcher/src/execute/ghcpEvalFiles.ts | 4 +--- .../dispatcher/test/ghcpEvalFiles.spec.ts | 23 ++++++++++--------- 2 files changed, 13 insertions(+), 14 deletions(-) diff --git a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts index 1c86c00eba..d27bccfb63 100644 --- a/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts +++ b/ts/packages/dispatcher/dispatcher/src/execute/ghcpEvalFiles.ts @@ -151,9 +151,7 @@ export function ghcpEvalNativeFilePermission( ); if ( !policy.readsEnabled && - (request.kind === "shell" || - (request.kind === "read" && - isGhcpEvalFixtureFile(request.path, root, policy.readFiles))) + (request.kind === "shell" || request.kind === "read") ) return false; return undefined; diff --git a/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts b/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts index 848e84cfc9..42449b8a5c 100644 --- a/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts +++ b/ts/packages/dispatcher/dispatcher/test/ghcpEvalFiles.spec.ts @@ -114,17 +114,18 @@ describe("common-file evaluation scope", () => { expect( ghcpEvalFileActionAllowed("readFile", { path: file }, root, policy), ).toBe(false); - expect( - ghcpEvalNativeFilePermission( - { - kind: "read", - intention: "read", - path: file, - }, - root, - policy, - ), - ).toBe(false); + for (const readPath of [file, root]) + expect( + ghcpEvalNativeFilePermission( + { + kind: "read", + intention: "read or search", + path: readPath, + }, + root, + policy, + ), + ).toBe(false); policy.readsEnabled = true; expect( ghcpEvalFileActionAllowed(