From bc3e0804f9b41058e68cc715884f88b8a66a5584 Mon Sep 17 00:00:00 2001 From: George Ng Date: Mon, 28 Sep 2026 21:17:13 -0700 Subject: [PATCH 1/3] Classify macro tools using replay-host availability Route captured Copilot-only MCP tools through the macro runner without weakening approval or deterministic replay checks. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-macros/README.md | 25 ++- ts/packages/copilot-macros/src/contracts.ts | 1 + .../copilot-macros/src/macroDefinition.ts | 31 ++-- .../copilot-macros/src/macroManager.ts | 3 +- .../copilot-macros/test/macroCatalog.spec.ts | 151 +++++++++++++++++- .../test/macroDefinition.spec.ts | 131 ++++++++++++--- .../src/mcp/mcpReplayHost.ts | 15 +- .../test/mcpReplayHost.spec.ts | 83 ++++++++++ 8 files changed, 394 insertions(+), 46 deletions(-) diff --git a/ts/packages/copilot-macros/README.md b/ts/packages/copilot-macros/README.md index 16d87b5c0d..d67afa7541 100644 --- a/ts/packages/copilot-macros/README.md +++ b/ts/packages/copilot-macros/README.md @@ -56,7 +56,8 @@ The public contracts in `src/contracts.ts` include: Each macro and step has an execution class: -- `replayable`: TypeAgent can inspect and call the named MCP tool. +- `replayable`: TypeAgent's replay host inspected and found the named MCP tool + when the draft was created. - `agentRequired`: the step needs Copilot's live tool and permission surface. Execution class is selected for the whole macro. A run never replays a prefix @@ -72,8 +73,19 @@ before handing later steps to the agent. - infer the result type and up to 50 result paths as postconditions. It does not infer arbitrary semantic parameters, loops, branches, or a -generalized program from one example. Calls with an MCP server name are -classified as replayable. Calls without one are classified as agent-required. +generalized program from one example. Induction is asynchronous: calls with an +MCP server name are inspected through the supplied replay host using the +recorded working directory. Only tools available to that host are classified +as replayable. Native calls, missing servers or tools, and calls without a +configured replay host are classified as agent-required. The captured server +and tool names are preserved for the agent runner. + +An MCP server name alone does not establish replayability. For example, +`github-mcp-server/web_search` may be available in Copilot but not in +TypeAgent. Such a call produces an agent-required draft with a review warning, +not an unapprovable replayable draft. Connection, authentication, and tool-list +errors still fail draft creation; they are not treated as absent capabilities. +Inspection does not invoke the recorded tools. ## Validation and approval @@ -86,6 +98,13 @@ records its schema fingerprint. Approval then writes the next version with state `approved`. Disabling an approved macro also writes a new version. Agent-guided adaptations are saved as separate drafts. +Classification does not silently change during approval or execution. A +replayable tool disappearing after draft creation still blocks approval, and +existing approved replayable macros retain their preflight checks. To correct +an older draft that misclassified a Copilot-only MCP tool, create a new draft +from its original trace, review it, and explicitly approve it. Existing +versions are not rewritten. + ## Deterministic replay Replay preflights the complete macro before invoking step one: diff --git a/ts/packages/copilot-macros/src/contracts.ts b/ts/packages/copilot-macros/src/contracts.ts index a604183985..4695c66c94 100644 --- a/ts/packages/copilot-macros/src/contracts.ts +++ b/ts/packages/copilot-macros/src/contracts.ts @@ -328,6 +328,7 @@ export interface ReplayToolContext { } export interface ReplayToolHost { + // Missing servers/tools return undefined; connection and inspection failures throw. inspectTool( mcpServerName: string | undefined, toolName: string, diff --git a/ts/packages/copilot-macros/src/macroDefinition.ts b/ts/packages/copilot-macros/src/macroDefinition.ts index 8aa33bc552..10bdbe613c 100644 --- a/ts/packages/copilot-macros/src/macroDefinition.ts +++ b/ts/packages/copilot-macros/src/macroDefinition.ts @@ -11,14 +11,21 @@ import type { MacroValidationIssue, MacroValidationReport, RecordedInteractionTrace, + ReplayToolHost, ValueExpression, } from "./contracts.js"; -function classifyTool( - _toolName: string, +async function classifyTool( + toolName: string, mcpServerName: string | undefined, -): MacroExecutionClass { - return mcpServerName ? "replayable" : "agentRequired"; + cwd: string, + replayHost: ReplayToolHost | undefined, +): Promise { + if (!mcpServerName || !replayHost) return "agentRequired"; + const descriptor = await replayHost.inspectTool(mcpServerName, toolName, { + cwd, + }); + return descriptor ? "replayable" : "agentRequired"; } function getValueType(value: unknown): MacroValueType { @@ -151,20 +158,26 @@ function convertArguments( : { kind: "template", value, bindings }; } -export function induceMacroFromTrace( +export async function induceMacroFromTrace( traceId: string, trace: RecordedInteractionTrace, macroId: string, name: string, description: string, createdAt: string, -): CopilotToolMacro { + replayHost?: ReplayToolHost, +): Promise { const warnings: string[] = []; const inputs: MacroInput[] = []; const steps: MacroStep[] = []; - trace.toolCalls.forEach((call, index) => { + for (const [index, call] of trace.toolCalls.entries()) { const id = `step-${index + 1}`; - const executionClass = classifyTool(call.name, call.mcpServerName); + const executionClass = await classifyTool( + call.name, + call.mcpServerName, + trace.cwd, + replayHost, + ); if (executionClass === "agentRequired") { warnings.push( `${id} uses ${call.mcpServerName ? `${call.mcpServerName}/` : ""}${call.name} and requires agent-guided execution.`, @@ -196,7 +209,7 @@ export function induceMacroFromTrace( ? {} : { postconditions: inferPostconditions(call.result) }), }); - }); + } return { schemaVersion: 1, diff --git a/ts/packages/copilot-macros/src/macroManager.ts b/ts/packages/copilot-macros/src/macroManager.ts index a2a4954e37..a558980842 100644 --- a/ts/packages/copilot-macros/src/macroManager.ts +++ b/ts/packages/copilot-macros/src/macroManager.ts @@ -227,13 +227,14 @@ export class MacroManager { if (!request.name.trim()) throw new Error("Macro name is required."); return this.mutateCatalog(async () => { const trace = await this.readTrace(request.traceId); - const macro = induceMacroFromTrace( + const macro = await induceMacroFromTrace( request.traceId, trace, randomUUID(), request.name.trim(), request.description?.trim() ?? trace.prompt, new Date().toISOString(), + this.replayHost, ); await this.writeVersion(macro); await this.upsertSummary(macro); diff --git a/ts/packages/copilot-macros/test/macroCatalog.spec.ts b/ts/packages/copilot-macros/test/macroCatalog.spec.ts index 1334f3a5c5..a2153a5c41 100644 --- a/ts/packages/copilot-macros/test/macroCatalog.spec.ts +++ b/ts/packages/copilot-macros/test/macroCatalog.spec.ts @@ -5,7 +5,11 @@ import { createHash } from "node:crypto"; import { mkdtemp, readFile } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; -import { MacroManager, type ReplayToolHost } from "@typeagent/copilot-macros"; +import { + MacroManager, + type RecordedToolCall, + type ReplayToolHost, +} from "@typeagent/copilot-macros"; async function captureTrace( manager: MacroManager, @@ -15,6 +19,7 @@ async function captureTrace( mcpServerName?: string; status?: "completed" | "failed" | "denied"; prompt?: string; + additionalCalls?: RecordedToolCall[]; } = {}, ): Promise { const sessionId = options.sessionId ?? "session-1"; @@ -48,6 +53,7 @@ async function captureTrace( result: { content: "{}" }, status: options.status ?? "completed", }, + ...(options.additionalCalls ?? []), ], }, }); @@ -107,7 +113,7 @@ describe("MacroManager draft catalog", () => { await expect(manager.validateMacro(draft)).resolves.toMatchObject({ valid: true, - executionClass: "replayable", + executionClass: "agentRequired", }); const approved = await manager.approveMacro(draft); expect(approved).toMatchObject({ version: 2, state: "approved" }); @@ -118,7 +124,7 @@ describe("MacroManager draft catalog", () => { ).resolves.toMatchObject({ version: 2, state: "approved", - executionClass: "replayable", + executionClass: "agentRequired", }); await expect( restarted.inspectMacro({ macroId: draft.macroId, version: 1 }), @@ -156,7 +162,7 @@ describe("MacroManager draft catalog", () => { await expect( manager.getMacroRequirements({ macroId: draft.macroId }), ).resolves.toMatchObject({ - executionClass: "replayable", + executionClass: "agentRequired", tools: [{ toolName: "read" }], }); await expect( @@ -187,6 +193,143 @@ describe("MacroManager draft catalog", () => { ).toHaveLength(1); }); + it.each([false, true])( + "hands Copilot-only MCP tools to the agent without replaying a prefix (mixed: %s)", + async (mixed) => { + const instanceDir = await mkdtemp( + path.join(os.tmpdir(), "catalog-"), + ); + let calls = 0; + const manager = new MacroManager(instanceDir, { + inspectTool: async (mcpServerName, toolName) => + mcpServerName === "typeagent-workspace" + ? { + mcpServerName, + toolName, + schemaFingerprint: "v1", + } + : undefined, + callTool: async () => { + calls++; + return {}; + }, + }); + const searchCall = { + toolCallId: "call-2", + name: "web_search", + mcpServerName: "github-mcp-server", + arguments: { query: "Seattle weather" }, + result: { content: "Dry" }, + status: "completed", + } satisfies RecordedToolCall; + const traceId = await captureTrace( + manager, + mixed + ? { additionalCalls: [searchCall] } + : { + toolName: searchCall.name, + mcpServerName: searchCall.mcpServerName, + }, + ); + const draft = await manager.createMacroFromTrace({ + traceId, + name: "Think about going outside", + }); + const definition = await manager.inspectMacro(draft); + expect(definition.executionClass).toBe("agentRequired"); + expect(definition.steps.map((step) => step.executionClass)).toEqual( + mixed ? ["replayable", "agentRequired"] : ["agentRequired"], + ); + expect(definition.steps.at(-1)).toMatchObject({ + toolName: "web_search", + mcpServerName: "github-mcp-server", + }); + await expect(manager.validateMacro(draft)).resolves.toMatchObject({ + valid: true, + executionClass: "agentRequired", + }); + const approved = await manager.approveMacro(draft); + await expect( + manager.runMacro({ + ...approved, + runId: "copilot-only-run", + preference: "auto", + }), + ).resolves.toMatchObject({ + status: "agentRequired", + launch: { + agent: "typeagent-macro-runner", + macro: { + ...definition, + version: 2, + state: "approved", + createdAt: expect.any(String), + }, + reason: { stepIds: [mixed ? "step-2" : "step-1"] }, + }, + }); + await expect( + manager.runMacro({ + ...approved, + runId: "forced-replay", + preference: "replay", + }), + ).rejects.toThrow("agent"); + expect(calls).toBe(0); + }, + ); + + it("does not persist a draft when replay tool inspection fails", async () => { + const instanceDir = await mkdtemp(path.join(os.tmpdir(), "catalog-")); + const manager = new MacroManager(instanceDir, { + inspectTool: async () => { + throw new Error("Connection failed"); + }, + callTool: async () => { + throw new Error("Must not execute"); + }, + }); + const traceId = await captureTrace(manager); + await expect( + manager.createMacroFromTrace({ + traceId, + name: "Read package", + }), + ).rejects.toThrow("Connection failed"); + await expect(manager.listMacros()).resolves.toEqual([]); + }); + + it("rejects tools removed after draft creation without changing execution class", async () => { + const instanceDir = await mkdtemp(path.join(os.tmpdir(), "catalog-")); + let available = true; + const manager = new MacroManager(instanceDir, { + inspectTool: async (mcpServerName, toolName) => + available + ? { + ...(mcpServerName ? { mcpServerName } : {}), + toolName, + schemaFingerprint: "v1", + } + : undefined, + callTool: async () => { + throw new Error("Must not execute"); + }, + }); + const traceId = await captureTrace(manager); + const draft = await manager.createMacroFromTrace({ + traceId, + name: "Read package", + }); + available = false; + await expect(manager.approveMacro(draft)).rejects.toThrow( + "Replay tool is unavailable", + ); + await expect(manager.inspectMacro(draft)).resolves.toMatchObject({ + state: "draft", + executionClass: "replayable", + }); + }); + it("approves agent-required drafts but rejects unsuccessful source calls", async () => { const instanceDir = await mkdtemp(path.join(os.tmpdir(), "catalog-")); const manager = new MacroManager(instanceDir); diff --git a/ts/packages/copilot-macros/test/macroDefinition.spec.ts b/ts/packages/copilot-macros/test/macroDefinition.spec.ts index 845ca01808..9874806503 100644 --- a/ts/packages/copilot-macros/test/macroDefinition.spec.ts +++ b/ts/packages/copilot-macros/test/macroDefinition.spec.ts @@ -5,8 +5,20 @@ import { induceMacroFromTrace, validateMacro, type RecordedInteractionTrace, + type ReplayToolHost, } from "@typeagent/copilot-macros"; +const replayHost: ReplayToolHost = { + inspectTool: async (mcpServerName, toolName) => ({ + ...(mcpServerName ? { mcpServerName } : {}), + toolName, + schemaFingerprint: "v1", + }), + callTool: async () => { + throw new Error("Induction must not execute tools."); + }, +}; + function trace( call: Omit< Partial, @@ -38,15 +50,16 @@ function trace( } describe("macro induction and validation", () => { - it("induces a replayable linear workspace draft", () => { + it("induces a replayable linear workspace draft", async () => { const source = trace(); - const macro = induceMacroFromTrace( + const macro = await induceMacroFromTrace( "trace-1", source, "macro-1", "Read package", "Reads package metadata", "2026-08-14T10:01:00.000Z", + replayHost, ); expect(macro).toMatchObject({ @@ -91,14 +104,20 @@ describe("macro induction and validation", () => { expect(validateMacro(macro, source).valid).toBe(true); }); - it("classifies unknown tools as agent required", () => { - const macro = induceMacroFromTrace( + it("classifies native tools as agent required without inspecting them", async () => { + const macro = await induceMacroFromTrace( "trace-1", trace({ name: "shell", mcpServerName: null }), "macro-1", "Run command", "", "2026-08-14T10:01:00.000Z", + { + ...replayHost, + inspectTool: async () => { + throw new Error("Native tools must not be inspected."); + }, + }, ); expect(macro.executionClass).toBe("agentRequired"); @@ -107,24 +126,96 @@ describe("macro induction and validation", () => { ); }); - it("classifies captured MCP tools as replayable", () => { - const macro = induceMacroFromTrace( + it.each(["example-server", "github-mcp-server"])( + "classifies available MCP tools on %s as replayable", + async (mcpServerName) => { + const macro = await induceMacroFromTrace( + "trace-1", + trace({ name: "create_item", mcpServerName }), + "macro-1", + "Create item", + "", + "2026-08-14T10:01:00.000Z", + replayHost, + ); + + expect(macro.executionClass).toBe("replayable"); + expect(macro.warnings).not.toContainEqual( + expect.stringContaining("agent-guided execution"), + ); + }, + ); + + it.each(["github-mcp-server", "unconfigured-server"])( + "requires an agent for unavailable MCP tools on %s", + async (mcpServerName) => { + const inspections: unknown[] = []; + const macro = await induceMacroFromTrace( + "trace-1", + trace({ name: "web_search", mcpServerName }), + "macro-1", + "Search", + "", + "2026-08-14T10:01:00.000Z", + { + ...replayHost, + inspectTool: async (...args) => { + inspections.push(args); + return undefined; + }, + }, + ); + + expect(inspections).toEqual([ + [mcpServerName, "web_search", { cwd: "." }], + ]); + expect(macro.executionClass).toBe("agentRequired"); + expect(macro.steps[0]).toMatchObject({ + toolName: "web_search", + mcpServerName, + executionClass: "agentRequired", + }); + expect(macro.warnings).toContainEqual( + expect.stringContaining(`${mcpServerName}/web_search`), + ); + expect(validateMacro(macro).valid).toBe(true); + }, + ); + + it("requires an agent when no replay host is configured", async () => { + const macro = await induceMacroFromTrace( "trace-1", - trace({ name: "create_item", mcpServerName: "example-server" }), + trace(), "macro-1", - "Create item", + "Read", "", "2026-08-14T10:01:00.000Z", ); + expect(macro.executionClass).toBe("agentRequired"); + expect(macro.steps[0].executionClass).toBe("agentRequired"); + }); - expect(macro.executionClass).toBe("replayable"); - expect(macro.warnings).not.toContainEqual( - expect.stringContaining("agent-guided execution"), - ); + it("propagates inspection failures instead of treating them as unavailable tools", async () => { + await expect( + induceMacroFromTrace( + "trace-1", + trace(), + "macro-1", + "Read", + "", + "2026-08-14T10:01:00.000Z", + { + ...replayHost, + inspectTool: async () => { + throw new Error("Authentication failed"); + }, + }, + ), + ).rejects.toThrow("Authentication failed"); }); - it("turns redacted arguments into required secret inputs", () => { - const macro = induceMacroFromTrace( + it("turns redacted arguments into required secret inputs", async () => { + const macro = await induceMacroFromTrace( "trace-1", trace({ arguments: { authorization: "[REDACTED]" } }), "macro-1", @@ -156,7 +247,7 @@ describe("macro induction and validation", () => { }); }); - it("binds a later argument to an earlier captured result", () => { + it("binds a later argument to an earlier captured result", async () => { const source = trace(); source.prompt = "Find package.json and inspect it"; source.toolCalls[0].result = { match: { path: "src/package.json" } }; @@ -169,7 +260,7 @@ describe("macro induction and validation", () => { status: "completed", }); - const macro = induceMacroFromTrace( + const macro = await induceMacroFromTrace( "trace-1", source, "macro-1", @@ -194,10 +285,10 @@ describe("macro induction and validation", () => { }); }); - it("canonicalizes omitted tool arguments as an empty object", () => { + it("canonicalizes omitted tool arguments as an empty object", async () => { const source = trace(); delete source.toolCalls[0].arguments; - const macro = induceMacroFromTrace( + const macro = await induceMacroFromTrace( "trace-1", source, "macro-1", @@ -213,9 +304,9 @@ describe("macro induction and validation", () => { expect(JSON.parse(JSON.stringify(macro))).toEqual(macro); }); - it("rejects a draft induced from a failed tool call", () => { + it("rejects a draft induced from a failed tool call", async () => { const source = trace({ status: "failed" }); - const macro = induceMacroFromTrace( + const macro = await induceMacroFromTrace( "trace-1", source, "macro-1", diff --git a/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts b/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts index adfaf98a54..b07767bf0d 100644 --- a/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts +++ b/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts @@ -91,6 +91,7 @@ export class McpReplayHost implements ReplayToolHost { context: ReplayToolContext = {}, ): Promise { const server = await this.getServer(mcpServerName, context); + if (!server) return undefined; const tool = this.getTool(server, toolName); if (!tool) return undefined; return { @@ -116,8 +117,8 @@ export class McpReplayHost implements ReplayToolHost { context: ReplayToolContext = {}, ): Promise { const server = await this.getServer(mcpServerName, context); - const tool = this.getTool(server, toolName); - if (!tool) { + const tool = server && this.getTool(server, toolName); + if (!server || !tool) { throw new Error( `MCP replay tool is unavailable: ${mcpServerName}/${toolName}`, ); @@ -187,11 +188,9 @@ export class McpReplayHost implements ReplayToolHost { private async getServer( serverName: string | undefined, context: ReplayToolContext, - ): Promise { + ): Promise { if (!serverName || serverName === "typeagent-macros") { - throw new Error( - `MCP server is not replayable: ${serverName ?? "native"}`, - ); + return undefined; } const configs = await this.getConfigs(context.cwd); const config = configs.find( @@ -199,9 +198,7 @@ export class McpReplayHost implements ReplayToolHost { candidate.name === serverName || candidate.id === serverName, ); if (!config) { - throw new Error( - `MCP server '${serverName}' is Copilot-only or not configured in TypeAgent.`, - ); + return undefined; } let active = this.active.get(config.id); if (!active) { diff --git a/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts b/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts index a48c271a2c..4f4c7235e3 100644 --- a/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts +++ b/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts @@ -8,6 +8,85 @@ import { McpReplayHost } from "../src/mcp/mcpReplayHost.js"; import type { NormalizedMcpServerConfig } from "../src/mcp/mcpServerConfig.js"; describe("MCP replay host", () => { + it.each([ + undefined, + "typeagent-macros", + "github-mcp-server", + "unconfigured-server", + ])( + "reports %s as unavailable for replay without connecting", + async (serverName) => { + const instanceDir = await mkdtemp( + path.join(os.tmpdir(), "mcp-replay-"), + ); + const host = new McpReplayHost(instanceDir, { + configs: [], + audit: { write: async () => {} }, + connectionFactory: async () => { + throw new Error( + "Must not connect to an unavailable server", + ); + }, + }); + await expect( + host.inspectTool(serverName, "web_search"), + ).resolves.toBeUndefined(); + await expect( + host.callTool( + serverName, + "web_search", + {}, + new AbortController().signal, + ), + ).rejects.toThrow("MCP replay tool is unavailable"); + await host.close(); + }, + ); + + it.each(["connect", "listTools"])( + "propagates %s failures instead of reporting unavailable tools", + async (failure) => { + const instanceDir = await mkdtemp( + path.join(os.tmpdir(), "mcp-replay-"), + ); + let closed = false; + const host = new McpReplayHost(instanceDir, { + configs: [ + { + id: "example", + name: "example", + transport: { kind: "stdio", command: "unused" }, + enabled: true, + trust: "trusted", + scope: "workspace", + provenance: { source: "captured-test" }, + }, + ], + audit: { write: async () => {} }, + connectionFactory: async () => { + if (failure === "connect") + throw new Error("Connection failed"); + return { + listTools: async () => { + throw new Error("Listing failed"); + }, + callTool: async () => { + throw new Error("Must not execute"); + }, + close: async () => { + closed = true; + }, + }; + }, + }); + await expect(host.inspectTool("example", "read")).rejects.toThrow( + failure === "connect" ? "Connection failed" : "Listing failed", + ); + expect(closed).toBe(failure === "listTools"); + await host.close(); + }, + ); + it("replays the captured tool without applying a second permission model", async () => { const instanceDir = await mkdtemp( path.join(os.tmpdir(), "mcp-replay-"), @@ -69,6 +148,10 @@ describe("MCP replay host", () => { await expect( host.inspectTool("example", "create_item"), ).resolves.toMatchObject({ toolName: "create_item" }); + await expect( + host.inspectTool("example", "missing_tool"), + ).resolves.toBeUndefined(); + expect(calls).toEqual([]); await expect( host.callTool( "example", From b46ad19a93dd240a6ea169ffa315aa5759e9bf39 Mon Sep 17 00:00:00 2001 From: George Ng Date: Mon, 28 Sep 2026 21:32:45 -0700 Subject: [PATCH 2/3] Address macro runner access and replay inspection review findings Expose live tools to the constrained runner, propagate discovery failures, and refresh tool catalogs for approval and preflight. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-macros/README.md | 9 + .../agents/typeagent-macro-runner.agent.md | 17 +- .../test/pluginArtifact.spec.ts | 24 +++ .../src/mcp/mcpReplayHost.ts | 80 +++++---- .../test/mcpReplayHost.spec.ts | 155 ++++++++++++++++++ 5 files changed, 246 insertions(+), 39 deletions(-) diff --git a/ts/packages/copilot-macros/README.md b/ts/packages/copilot-macros/README.md index d67afa7541..0c77d9bc26 100644 --- a/ts/packages/copilot-macros/README.md +++ b/ts/packages/copilot-macros/README.md @@ -85,6 +85,8 @@ An MCP server name alone does not establish replayability. For example, TypeAgent. Such a call produces an agent-required draft with a review warning, not an unapprovable replayable draft. Connection, authentication, and tool-list errors still fail draft creation; they are not treated as absent capabilities. +Unreadable configuration files and invalid entries for the requested server +also fail discovery rather than silently selecting the agent runner. Inspection does not invoke the recorded tools. ## Validation and approval @@ -104,6 +106,8 @@ existing approved replayable macros retain their preflight checks. To correct an older draft that misclassified a Copilot-only MCP tool, create a new draft from its original trace, review it, and explicitly approve it. Existing versions are not rewritten. +Tool inspection refreshes the connected server's advertised catalog, so draft +creation cannot leave approval or replay checking a stale tool schema. ## Deterministic replay @@ -129,6 +133,11 @@ If a macro is `agentRequired`, or the caller selects the `agent` preference, `MacroManager.runMacro()` returns an `AgentRunnerLaunchPayload`. The payload contains the approved macro, supplied inputs, handoff reason, execution budgets, and candidate provenance. The package does not launch the runner. +The runner inherits available Copilot tools, including MCP tools outside +TypeAgent's replay host, subject to live permissions. Its instructions restrict +tool use to the approved procedure, inspection, required discovery, and +successful candidate submission. An unavailable or denied tool stops the run; +the runner does not install tools or change permissions. After a successful agent-guided run, `submitMacroCandidate()` can save an adapted procedure. It verifies handoff identity, source version, execution diff --git a/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md b/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md index 97c1f26134..542455f2c6 100644 --- a/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md +++ b/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md @@ -1,14 +1,7 @@ --- name: typeagent-macro-runner description: "Runs an approved TypeAgent macro from a structured agentRequired handoff using live Copilot tools and permissions. Use when run_macro returns a typeagent-macro-runner launch payload." -tools: - - read - - search - - edit - - execute - - typeagent-workspace/* - - typeagent-macros/inspect_macro - - typeagent-macros/submit_macro_candidate +tools: ["*"] user-invocable: false --- @@ -27,7 +20,10 @@ that provide only a macro name or free-form procedure. version as the launch payload. 2. Execute the whole macro in step order using the supplied inputs and prior step results. Do not split execution between deterministic replay and this - runner. + runner. Discover the captured MCP tools from the live catalog when needed; + do not assume an MCP server name means TypeAgent can replay the tool. + If a required tool is unavailable, stop and report it rather than silently + substituting another tool. 3. Use Copilot's live tool permissions. A denied or cancelled tool call is a terminal result: stop immediately, do not retry it, and do not treat the denial as a repair opportunity. @@ -44,3 +40,6 @@ that provide only a macro name or free-form procedure. Do not call `run_macro` recursively. Do not submit a candidate after permission denial, cancellation, timeout, or an unsuccessful adaptation. +Use tools only for the approved procedure and its inspection, required tool +discovery, and successful candidate submission. Do not call other macro +lifecycle tools, change permissions, or install tools to complete the run. diff --git a/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts b/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts index f132bc037e..a41c9546ff 100644 --- a/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts +++ b/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts @@ -6,6 +6,7 @@ import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js" import { readFile } from "node:fs/promises"; import path from "node:path"; import { fileURLToPath } from "node:url"; +import { parse } from "yaml"; interface PluginMcpManifest { mcpServers: Record< @@ -19,6 +20,29 @@ interface PluginManifest { } describe("staged plugin artifact", () => { + it("lets the macro runner discover captured Copilot MCP tools under live permissions", async () => { + const pluginRoot = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + "..", + "..", + ); + const runner = await readFile( + path.join(pluginRoot, "agents", "typeagent-macro-runner.agent.md"), + "utf8", + ); + const frontmatter = runner.split("---")[1]; + const profile = parse(frontmatter); + expect(profile).toMatchObject({ + name: "typeagent-macro-runner", + tools: ["*"], + "user-invocable": false, + }); + expect(runner).toContain("Use Copilot's live tool permissions"); + expect(runner).toContain("Do not call `run_macro` recursively"); + expect(runner).toContain("Do not call other macro"); + expect(runner).toContain("If a required tool is unavailable, stop"); + }); + it("contains the extension bundle at the declared discovery path", async () => { const testDir = path.dirname(fileURLToPath(import.meta.url)); const pluginRoot = path.resolve(testDir, "..", ".."); diff --git a/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts b/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts index b07767bf0d..93bb8c6c73 100644 --- a/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts +++ b/ts/packages/defaultAgentProvider/src/mcp/mcpReplayHost.ts @@ -31,7 +31,7 @@ import { type ActiveServer = { config: NormalizedMcpServerConfig; connection: McpReplayConnection; - catalog: McpToolCatalog; + catalog: McpToolCatalog | undefined; }; export interface McpReplayConnection { @@ -56,10 +56,6 @@ export class McpReplayHost implements ReplayToolHost { private readonly credentialStore: McpCredentialStore; private readonly audit: McpAuditSink; private readonly configs: NormalizedMcpServerConfig[]; - private readonly discovered = new Map< - string, - Promise - >(); private readonly connectionFactory: ( config: NormalizedMcpServerConfig, ) => Promise; @@ -90,7 +86,7 @@ export class McpReplayHost implements ReplayToolHost { toolName: string, context: ReplayToolContext = {}, ): Promise { - const server = await this.getServer(mcpServerName, context); + const server = await this.getServer(mcpServerName, context, true); if (!server) return undefined; const tool = this.getTool(server, toolName); if (!tool) return undefined; @@ -188,11 +184,12 @@ export class McpReplayHost implements ReplayToolHost { private async getServer( serverName: string | undefined, context: ReplayToolContext, + refreshCatalog = false, ): Promise { if (!serverName || serverName === "typeagent-macros") { return undefined; } - const configs = await this.getConfigs(context.cwd); + const configs = this.getConfigs(context.cwd, serverName); const config = configs.find( (candidate) => candidate.name === serverName || candidate.id === serverName, @@ -205,32 +202,49 @@ export class McpReplayHost implements ReplayToolHost { active = this.connect(config); this.active.set(config.id, active); active.catch(() => this.active.delete(config.id)); + } else if (refreshCatalog) { + const server = await active; + server.catalog = await this.readCatalog( + config.id, + server.connection, + ); } return active; } - private async getConfigs( + private getConfigs( cwd: string | undefined, - ): Promise { + serverName: string, + ): NormalizedMcpServerConfig[] { if (!cwd) return this.configs; - const resolvedCwd = path.resolve(cwd); - let configs = this.discovered.get(resolvedCwd); - if (!configs) { - configs = this.discoverConfigs(resolvedCwd); - this.discovered.set(resolvedCwd, configs); - } - return [...(await configs), ...this.configs]; + return [ + ...this.discoverConfigs(path.resolve(cwd), serverName), + ...this.configs, + ]; } - private async discoverConfigs( + private discoverConfigs( cwd: string, - ): Promise { - return new McpConfigDiscovery() - .discover({ - workspacePath: cwd, - isFolderTrusted: () => true, - }) - .configs.filter((entry) => entry.config.name !== "typeagent-macros") + serverName: string, + ): NormalizedMcpServerConfig[] { + const discovery = new McpConfigDiscovery().discover({ + workspacePath: cwd, + isFolderTrusted: () => true, + }); + const failures = discovery.diagnostics.filter( + (diagnostic) => + (diagnostic.kind === "unreadable" || + diagnostic.kind === "invalid") && + (diagnostic.serverName === undefined || + diagnostic.serverName === serverName), + ); + if (failures.length > 0) { + throw new Error( + `MCP replay configuration discovery failed: ${failures.map((failure) => failure.message).join("; ")}`, + ); + } + return discovery.configs + .filter((entry) => entry.config.name !== "typeagent-macros") .map((entry) => entry.config); } @@ -239,11 +253,7 @@ export class McpReplayHost implements ReplayToolHost { ): Promise { const connection = await this.connectionFactory(config); try { - const catalog = buildMcpToolCatalog( - config.id, - await connection.listTools(), - "McpReplayAction", - ); + const catalog = await this.readCatalog(config.id, connection); return { config, connection, catalog }; } catch (error) { await connection.close(); @@ -255,11 +265,21 @@ export class McpReplayHost implements ReplayToolHost { server: ActiveServer, toolName: string, ): McpToolCatalogEntry | undefined { - return server.catalog.entries.get( + return server.catalog?.entries.get( getMcpToolIdentity(server.config.id, toolName), ); } + private async readCatalog( + configId: string, + connection: McpReplayConnection, + ): Promise { + const tools = await connection.listTools(); + return tools.length === 0 + ? undefined + : buildMcpToolCatalog(configId, tools, "McpReplayAction"); + } + private asArguments(value: unknown): Record { if ( value === null || diff --git a/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts b/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts index 4f4c7235e3..3ef97c6a62 100644 --- a/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts +++ b/ts/packages/defaultAgentProvider/test/mcpReplayHost.spec.ts @@ -4,10 +4,165 @@ import { mkdtemp, writeFile } from "node:fs/promises"; import os from "node:os"; import path from "node:path"; +import { createHash } from "node:crypto"; +import { MacroManager } from "@typeagent/copilot-macros"; import { McpReplayHost } from "../src/mcp/mcpReplayHost.js"; import type { NormalizedMcpServerConfig } from "../src/mcp/mcpServerConfig.js"; describe("MCP replay host", () => { + it.each([ + ["malformed", "{"], + ["invalid root", JSON.stringify({ invalid: true })], + ["invalid server", JSON.stringify({ mcpServers: { example: {} } })], + ])( + "rejects %s discovery instead of classifying a tool as absent", + async (_label, content) => { + const instanceDir = await mkdtemp( + path.join(os.tmpdir(), "mcp-replay-"), + ); + const configPath = path.join(instanceDir, ".mcp.json"); + await writeFile(configPath, content); + const host = new McpReplayHost(instanceDir, { configs: [] }); + try { + await expect( + host.inspectTool("example", "read", { cwd: instanceDir }), + ).rejects.toThrow("MCP replay configuration discovery failed"); + await writeFile(configPath, JSON.stringify({ mcpServers: {} })); + await expect( + host.inspectTool("example", "read", { cwd: instanceDir }), + ).resolves.toBeUndefined(); + } finally { + await host.close(); + } + }, + ); + + it("does not block discovery on an unrelated invalid server entry", async () => { + const instanceDir = await mkdtemp( + path.join(os.tmpdir(), "mcp-replay-"), + ); + await writeFile( + path.join(instanceDir, ".mcp.json"), + JSON.stringify({ mcpServers: { unrelated: {} } }), + ); + const host = new McpReplayHost(instanceDir, { configs: [] }); + try { + await expect( + host.inspectTool("github-mcp-server", "web_search", { + cwd: instanceDir, + }), + ).resolves.toBeUndefined(); + } finally { + await host.close(); + } + }); + + it.each(["removed", "schemaChanged"])( + "refreshes the production catalog between induction, approval and replay (%s)", + async (change) => { + const instanceDir = await mkdtemp( + path.join(os.tmpdir(), "mcp-replay-"), + ); + let present = true; + let schemaVersion = "v1"; + let calls = 0; + let inspections = 0; + const host = new McpReplayHost(instanceDir, { + configs: [ + { + id: "example", + name: "example", + transport: { kind: "stdio", command: "unused" }, + enabled: true, + trust: "trusted", + scope: "workspace", + provenance: { source: "captured-test" }, + }, + ], + audit: { write: async () => {} }, + connectionFactory: async () => ({ + listTools: async () => { + inspections++; + return present + ? [ + { + name: "read", + inputSchema: { + type: "object", + properties: {}, + description: schemaVersion, + }, + }, + ] + : []; + }, + callTool: async () => { + calls++; + return { content: [] }; + }, + close: async () => {}, + }), + }); + try { + const manager = new MacroManager(instanceDir, host); + const token = manager.armRecording({ sessionId: "refresh" }); + manager.claimRecording({ + sessionId: "refresh", + cwd: instanceDir, + promptHash: createHash("sha256") + .update("Read") + .digest("hex"), + }); + const trace = await manager.finalizeRecording({ + tokenId: token.id, + trace: { + schemaVersion: 1, + sessionId: "refresh", + cwd: instanceDir, + prompt: "Read", + response: "Done", + startedAt: "2026-09-28T00:00:00Z", + completedAt: "2026-09-28T00:00:01Z", + toolCalls: [ + { + toolCallId: "call-1", + name: "read", + mcpServerName: "example", + arguments: {}, + result: [], + status: "completed", + }, + ], + }, + }); + const draft = await manager.createMacroFromTrace({ + traceId: trace.traceId, + name: "Read", + }); + expect(inspections).toBe(1); + if (change === "removed") { + present = false; + await expect(manager.approveMacro(draft)).rejects.toThrow( + "Replay tool is unavailable", + ); + } else { + const approved = await manager.approveMacro(draft); + schemaVersion = "v2"; + await expect( + manager.runMacro({ ...approved, runId: "refresh-run" }), + ).resolves.toMatchObject({ + status: "failed", + run: { error: { code: "schemaDrift" }, steps: [] }, + }); + } + expect(inspections).toBe(change === "removed" ? 2 : 3); + expect(calls).toBe(0); + } finally { + await host.close(); + } + }, + ); + it.each([ undefined, "typeagent-macros", From 16f7a37bdf32fa84baf826857dda02f03ccac0ae Mon Sep 17 00:00:00 2001 From: George Ng <146492653+GeorgeNgMsft@users.noreply.github.com> Date: Tue, 29 Sep 2026 17:56:08 -0700 Subject: [PATCH 3/3] Fix Copilot macro runner tool identity and result guards (#3099) This follow-up to #3097 fixes two coupled problems in agent-guided macro execution: the runner could mistake MCP backend provenance for a callable namespace, and induced result guards described Copilot's event/UI wrapper rather than the response the runner actually receives. The change preserves the original tool identity, permissions, raw evidence, and immutable approved versions. - Resolve the exact captured callable before deferred discovery. Document the verified `web_search` / `github-mcp-server/web_search` bridge without allowing arbitrary aliases or provider substitution. - Capture separately redacted model-facing results from the SDK's `result.content`, preserving valid JSON primitives and ordinary text as well as the raw event result. - Use the model-facing representation for normal guards and prior-step result bindings throughout agent-required procedures, including mixed procedures. Deterministic-only induction continues using raw results. - Keep legacy guards intact, warn when model-facing evidence is absent, and require recapture and explicit approval when an older macro needs unobservable fields. - Add regression coverage for callable/provenance preservation, redaction, absent versus null/false/zero/empty results, mixed-step references, deterministic compatibility, and runner safeguards. ## Local validation before final adversarial review Actual standalone Copilot CLI worker **1.0.89** executed a benign IANA search, captured it, and ran its normally induced/validated/approved version 2 through the real `typeagent:typeagent-macro-runner`. The runner inspected the immutable version and executed exactly one live `web_search` with the original arguments. Execution events report `success: true`; an independent harness checked **all 11 normal inferred type/path guards**, including answer and citation paths, against the actual returned model-facing content. No guard pruning, synthetic result, retry, provider substitution, or candidate submission was used in this final test. Harness boundary: real headless CLI events were streamed through production `SessionCapture`; unrelated conversation-history sinks were isolated, and the headless CLI final-result marker was adapted to SDK `session.idle`. This was not an unmodified interactive extension-recording test. The fixture and its persisted macro/trace/handoff were removed after retaining evidence and verifying the approved version was unchanged. Evidence identifiers: run `36b64f4a-bc0d-4002-8b1e-950aeec44773`, runner `b1f17ff8-34a8-4554-8820-052e2d04b457`, live search `call_4pXqFIfIxtxnas16NpdfhPf9`. - Dependency-aware plugin, macro, and server builds passed. - 46 macro tests, 178 plugin unit tests, 14 replay-host tests, and the recording RPC test passed. - Formatting and all four PR gates passed against the exact parent base, including tests in lint/complexity checks. - Two independent final adversarial reviewers examined the expanded change **after** full guarded CLI validation; neither reported a substantive issue. Reviewer model identities were not reliably exposed, so model diversity is unverified. - Reviewed base: `b46ad19a93dd240a6ea169ffa315aa5759e9bf39`; reviewed content committed unchanged as `3aab21a035767af6d3573243ce719a2f6093c137`. ## Compatibility and recovery The original access refusal was observed in an actual user run, but baseline resolution was nondeterministic: another baseline fixture found the tool and then failed the separate wrapper guards. This PR does not claim a proven permission or subagent-inheritance defect. The user's existing **Consider Stay Home Day** version 2 was neither changed nor replayed. Its old wrapper-field guards may still be unverifiable. After updating the server/plugin and starting a fresh Copilot session, **record a new interaction, inspect the new draft's inputs and guards, and explicitly approve it**. Do not waive old guards or rewrite the approved version. Parameterization/reasoning features are out of scope. ## Local deployment and rollback Only five artifacts were updated: the installed server bundle, two runner profiles, and two extension bundles. Each extension received only the verified capture-module change; every other installed module was preserved byte-for-byte. The rebuilt server bundle differed only in the intended induction logic. Backups and hashes were recorded; the installed daemon was restarted and verified responsive. No routing settings, conversation bindings, Azure identities, or authentication helpers were changed. Local evidence/rollback directory: `C:\Users\georgeng\.copilot\session-state\af340a10-e91b-4f95-8997-d99fb88a000d\files` Key files: `guarded-verification.json`, `real-capture-definition.json`, `cli-tool-metadata.json`, `fixture-cleanup.json`, `result-deployment.json`, `extension-module-provenance.json`, and `runtime-before-results`. Original runner-only backups are `runner-before-0.agent.md` and `runner-before-1.agent.md`. Restore only receipt-listed files after checking for subsequent updates; stop/start the daemon when restoring its bundle. This PR targets the inherited parent branch, leaves #3097 unchanged, and is for human review; it has not been merged or self-approved. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- ts/packages/copilot-macros/src/contracts.ts | 2 + .../copilot-macros/src/macroDefinition.ts | 39 ++++-- .../test/macroDefinition.spec.ts | 120 ++++++++++++++++ ts/packages/copilot-plugin/README.md | 27 ++++ .../agents/typeagent-macro-runner.agent.md | 45 +++++- .../src/extension/trace-assembler.ts | 17 +++ .../test/extensionTrace.spec.ts | 130 ++++++++++++++++++ .../test/pluginArtifact.spec.ts | 27 ++++ 8 files changed, 392 insertions(+), 15 deletions(-) diff --git a/ts/packages/copilot-macros/src/contracts.ts b/ts/packages/copilot-macros/src/contracts.ts index 4695c66c94..3694b213f2 100644 --- a/ts/packages/copilot-macros/src/contracts.ts +++ b/ts/packages/copilot-macros/src/contracts.ts @@ -7,6 +7,8 @@ export interface RecordedToolCall { mcpServerName?: string; arguments?: unknown; result?: unknown; + // Parsed model-facing tool text, distinct from the raw event/UI result. + modelResult?: unknown; status: "completed" | "failed" | "denied"; permission?: unknown; } diff --git a/ts/packages/copilot-macros/src/macroDefinition.ts b/ts/packages/copilot-macros/src/macroDefinition.ts index 10bdbe613c..4211d45a57 100644 --- a/ts/packages/copilot-macros/src/macroDefinition.ts +++ b/ts/packages/copilot-macros/src/macroDefinition.ts @@ -170,19 +170,36 @@ export async function induceMacroFromTrace( const warnings: string[] = []; const inputs: MacroInput[] = []; const steps: MacroStep[] = []; - for (const [index, call] of trace.toolCalls.entries()) { - const id = `step-${index + 1}`; - const executionClass = await classifyTool( - call.name, - call.mcpServerName, - trace.cwd, - replayHost, + const executionClasses: MacroExecutionClass[] = []; + for (const call of trace.toolCalls) { + executionClasses.push( + await classifyTool( + call.name, + call.mcpServerName, + trace.cwd, + replayHost, + ), ); + } + const agentRequired = executionClasses.includes("agentRequired"); + const calls = trace.toolCalls.map((call) => + agentRequired && call.modelResult !== undefined + ? { ...call, result: call.modelResult } + : call, + ); + for (const [index, call] of calls.entries()) { + const id = `step-${index + 1}`; + const executionClass = executionClasses[index]; if (executionClass === "agentRequired") { warnings.push( `${id} uses ${call.mcpServerName ? `${call.mcpServerName}/` : ""}${call.name} and requires agent-guided execution.`, ); } + if (agentRequired && call.modelResult === undefined) { + warnings.push( + `${id} has no captured model-facing result. Review its result guards and recapture if the runner cannot observe the required fields.`, + ); + } if (call.status !== "completed") { warnings.push( `${id} was captured with status ${call.status} and requires review.`, @@ -199,7 +216,7 @@ export async function induceMacroFromTrace( id, trace.prompt, steps, - trace.toolCalls.slice(0, index), + calls.slice(0, index), inputs, warnings, ), @@ -218,11 +235,7 @@ export async function induceMacroFromTrace( name, description, state: "draft", - executionClass: steps.every( - (step) => step.executionClass === "replayable", - ) - ? "replayable" - : "agentRequired", + executionClass: agentRequired ? "agentRequired" : "replayable", inputs, steps, sourceTraceId: traceId, diff --git a/ts/packages/copilot-macros/test/macroDefinition.spec.ts b/ts/packages/copilot-macros/test/macroDefinition.spec.ts index 9874806503..478f4844ac 100644 --- a/ts/packages/copilot-macros/test/macroDefinition.spec.ts +++ b/ts/packages/copilot-macros/test/macroDefinition.spec.ts @@ -50,6 +50,126 @@ function trace( } describe("macro induction and validation", () => { + it("uses model-facing guards and references throughout a mixed agent procedure", async () => { + const source = trace({ + result: { + content: '{"item":{"id":"item-123"}}', + detailedContent: "UI-only result", + }, + modelResult: { item: { id: "item-123" } }, + }); + source.toolCalls.push({ + toolCallId: "call-2", + name: "native_tool", + arguments: { itemId: "item-123" }, + result: { content: '{"ok":true}', detailedContent: "UI-only" }, + modelResult: { ok: true }, + status: "completed", + }); + const before = structuredClone(source); + const macro = await induceMacroFromTrace( + "trace-1", + source, + "macro-1", + "Mixed result views", + "", + "2026-09-29T08:00:00.000Z", + replayHost, + ); + + expect(source).toEqual(before); + expect(macro.executionClass).toBe("agentRequired"); + expect(macro.steps[0].executionClass).toBe("replayable"); + expect(macro.steps[0].postconditions).toEqual([ + { kind: "resultType", valueType: "object" }, + { kind: "resultPathExists", path: ["item", "id"] }, + ]); + expect(macro.steps[1]).toMatchObject({ + executionClass: "agentRequired", + arguments: { + kind: "template", + bindings: [ + { + path: ["itemId"], + expression: { + kind: "stepResult", + stepId: "step-1", + path: ["item", "id"], + }, + }, + ], + }, + postconditions: [ + { kind: "resultType", valueType: "object" }, + { kind: "resultPathExists", path: ["ok"] }, + ], + }); + expect(validateMacro(macro, source).valid).toBe(true); + }); + + it.each([ + [null, "null"], + [false, "boolean"], + [0, "number"], + ["", "string"], + ])( + "retains the model-facing primitive %s", + async (modelResult, valueType) => { + const macro = await induceMacroFromTrace( + "trace-1", + trace({ mcpServerName: null, modelResult }), + "macro-1", + "Primitive result", + "", + "2026-09-29T08:00:00.000Z", + ); + expect(macro.steps[0].postconditions).toEqual([ + { kind: "resultType", valueType }, + ]); + expect(macro.warnings).not.toContainEqual( + expect.stringContaining("no captured model-facing result"), + ); + }, + ); + + it("keeps deterministic replay guards on raw results", async () => { + const macro = await induceMacroFromTrace( + "trace-1", + trace({ + result: { raw: "value" }, + modelResult: { visible: "value" }, + }), + "macro-1", + "Replay result", + "", + "2026-09-29T08:00:00.000Z", + replayHost, + ); + expect(macro.executionClass).toBe("replayable"); + expect(macro.steps[0].postconditions).toEqual([ + { kind: "resultType", valueType: "object" }, + { kind: "resultPathExists", path: ["raw"] }, + ]); + }); + + it("preserves legacy guards and warns when model-facing evidence was not captured", async () => { + const macro = await induceMacroFromTrace( + "trace-1", + trace({ mcpServerName: null, result: { content: "legacy" } }), + "macro-1", + "Legacy result", + "", + "2026-09-29T08:00:00.000Z", + ); + expect(macro.steps[0].postconditions).toContainEqual({ + kind: "resultPathExists", + path: ["content"], + }); + expect(macro.warnings).toContainEqual( + expect.stringContaining("no captured model-facing result"), + ); + }); + it("induces a replayable linear workspace draft", async () => { const source = trace(); const macro = await induceMacroFromTrace( diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index d1ca8cc681..b4624d1bdf 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -554,6 +554,33 @@ Macros containing Copilot-native tools use the macro runner for the whole procedure. TypeAgent never replays a prefix and hands only the remainder to the agent. +For agent-guided execution, `step.toolName` preserves Copilot's captured +callable name; `mcpServerName` preserves backend provenance, not a namespace +to prepend. For example, Copilot CLI exposes `web_search` (rendered as +`functions.web_search`) while its runtime metadata and execution events +identify the backend as `github-mcp-server/web_search`. The runner uses the +original callable and its live `{ query: string }` schema even when web search +is absent from the deferred GitHub catalog. This does not substitute providers, +grant permissions, or modify an existing approved macro. Unknown identities, +incompatible arguments, permission denials, and cancellations still stop the +run; no arbitrary aliases are inferred. + +Captures retain the raw Copilot event result for audit and deterministic replay, +and separately retain the model-facing result text (parsed when it is JSON). +When the whole procedure requires the runner, induction uses the model-facing +values for normal result-type/path postconditions and prior-step bindings, +including any replayable steps within that procedure. UI-only envelope fields +are not evidence the runner can verify. Plain or truncated text remains a +string, not an invented structured result. + +Older approved macros may require event-wrapper paths such as `content`, +`detailedContent`, and `contents` that are not shown to the runner. Such runs +must stop, not waive those guards. Record a new interaction after updating the +plugin and restarting Copilot, inspect its newly induced guards and inputs, and +explicitly approve the new draft. Existing traces without model-facing evidence +and existing immutable versions are not silently rewritten. Resolving tool +access alone does not make an old macro's result guards verifiable. + ### Rollout Controls Each macro boundary is enabled by default and can be disabled independently: diff --git a/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md b/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md index 542455f2c6..0c9a02e5ef 100644 --- a/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md +++ b/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md @@ -20,8 +20,8 @@ that provide only a macro name or free-form procedure. version as the launch payload. 2. Execute the whole macro in step order using the supplied inputs and prior step results. Do not split execution between deterministic replay and this - runner. Discover the captured MCP tools from the live catalog when needed; - do not assume an MCP server name means TypeAgent can replay the tool. + runner. Resolve tools as described below; do not assume an MCP server name + means TypeAgent can replay the tool. If a required tool is unavailable, stop and report it rather than silently substituting another tool. 3. Use Copilot's live tool permissions. A denied or cancelled tool call is a @@ -38,6 +38,47 @@ that provide only a macro name or free-form procedure. The result must remain a draft for explicit review. Never approve, promote, or mutate the approved version. +## Tool Identity + +`step.toolName` is the callable name captured from Copilot's +`tool.execution_start` event. `step.mcpServerName` records backend provenance; +it is not an instruction to prepend a server name to the callable name. +Check the already exposed tools for the exact recorded callable first. For a +deferred tool, load its definition through the host's tool discovery before +invoking it. Check the resolved arguments against the live input schema. + +In Copilot CLI, `toolName: "web_search"` with +`mcpServerName: "github-mcp-server"` identifies Copilot's exposed `web_search` +tool (shown as `functions.web_search` in the model's tool namespace). The CLI +runtime reports that callable's backend identity as +`github-mcp-server/web_search`; its input is `{ query: string }`. Calling that +exposed tool with the approved query is the original invocation, not a +provider substitution. It need not appear in the deferred GitHub tool list. +Do not reject it solely because that list omits it. + +Do not generalize this bridge to other similarly named tools or providers. +Do not strip arbitrary prefixes, invent aliases, or use a different search +provider. If the live definition conflicts with the recorded identity or +arguments, or you cannot establish the required tool's identity, stop and +report the recorded name, backend provenance, and discovery evidence. +Resolving the original callable does not adapt the procedure and does not +require a candidate. Existing approved versions remain unchanged. + +## Result Evidence + +For postconditions and `stepResult` references, use the tool response actually +shown to you: parse it as JSON when it is valid JSON; otherwise use the text +as a string. Newly captured agent-required macros use that same representation +for their inferred type/path guards and result bindings. + +Check every declared postcondition before continuing. Do not invent Copilot +event or UI wrapper fields such as `content`, `detailedContent`, or `contents` +around a displayed result. If an older approved macro requires fields you +cannot observe, stop with the specific unverifiable postconditions. Do not +drop guards, claim success, or submit a successful candidate. Report that the +user needs to record a new interaction with the updated plugin, inspect the +new draft's guards, and explicitly approve it. Never change the old version. + Do not call `run_macro` recursively. Do not submit a candidate after permission denial, cancellation, timeout, or an unsuccessful adaptation. Use tools only for the approved procedure and its inspection, required tool diff --git a/ts/packages/copilot-plugin/src/extension/trace-assembler.ts b/ts/packages/copilot-plugin/src/extension/trace-assembler.ts index ae82cfdc0a..4d6e95bfbb 100644 --- a/ts/packages/copilot-plugin/src/extension/trace-assembler.ts +++ b/ts/packages/copilot-plugin/src/extension/trace-assembler.ts @@ -46,6 +46,19 @@ function permissionWasDenied(result: unknown): boolean { return kind === "cancelled" || kind?.startsWith("denied") === true; } +function getModelResult(result: unknown): unknown { + if (!result || typeof result !== "object") return undefined; + const content = (result as Record).content; + if (typeof content !== "string") return undefined; + try { + return JSON.parse(content); + } catch (error) { + // Plain text (including truncated JSON) is the actual model-visible value. + if (error instanceof SyntaxError) return content; + throw error; + } +} + export class ExtensionTraceAssembler { private turn: ActiveTurn | undefined; @@ -163,6 +176,10 @@ export class ExtensionTraceAssembler { if (!call || turn.completedCalls.has(key)) return; call.result = redactTraceValue(event.data.result ?? event.data.error); + const modelResult = getModelResult(event.data.result); + if (modelResult !== undefined) { + call.modelResult = redactTraceValue(modelResult); + } if (call.status !== "denied") { call.status = event.data.success === false ? "failed" : "completed"; } diff --git a/ts/packages/copilot-plugin/test/extensionTrace.spec.ts b/ts/packages/copilot-plugin/test/extensionTrace.spec.ts index a256fdc73e..836e0e0162 100644 --- a/ts/packages/copilot-plugin/test/extensionTrace.spec.ts +++ b/ts/packages/copilot-plugin/test/extensionTrace.spec.ts @@ -2,6 +2,7 @@ // Licensed under the MIT License. import { createHash } from "node:crypto"; +import { induceMacroFromTrace } from "@typeagent/copilot-macros"; import { ExtensionTraceAssembler } from "../src/extension/trace-assembler.js"; function event( @@ -14,6 +15,135 @@ function event( } describe("extension trace assembler", () => { + it.each([ + [ + '{"answer":{"value":"public","apiKey":"secret"}}', + { + answer: { value: "public", apiKey: "[REDACTED]" }, + }, + ], + ["null", null], + ["false", false], + ["0", 0], + ['""', ""], + ["", ""], + ["plain text", "plain text"], + ['{"truncated":', '{"truncated":'], + ])( + "captures model-facing content %s without UI-only fields", + (content, expected) => { + const assembler = new ExtensionTraceAssembler("session-1", "."); + assembler.record( + event("user.message", "2026-09-29T08:00:00.000Z", { + content: "Read the result", + }), + ); + assembler.record( + event("tool.execution_start", "2026-09-29T08:00:01.000Z", { + toolCallId: "call-1", + toolName: "web_search", + mcpServerName: "github-mcp-server", + }), + ); + assembler.record( + event("tool.execution_complete", "2026-09-29T08:00:02.000Z", { + toolCallId: "call-1", + success: true, + result: { + content, + detailedContent: "UI detail", + contents: [], + }, + }), + ); + const call = assembler.finish()?.toolCalls[0]; + expect(call?.modelResult).toEqual(expected); + expect(call?.result).toMatchObject({ + detailedContent: "UI detail", + contents: [], + }); + }, + ); + + it("does not manufacture model-facing evidence when content is absent", () => { + const assembler = new ExtensionTraceAssembler("session-1", "."); + assembler.record( + event("user.message", "2026-09-29T08:00:00.000Z", { + content: "Read", + }), + ); + assembler.record( + event("tool.execution_start", "2026-09-29T08:00:01.000Z", { + toolCallId: "call-1", + toolName: "read", + }), + ); + assembler.record( + event("tool.execution_complete", "2026-09-29T08:00:02.000Z", { + toolCallId: "call-1", + success: true, + result: { detailedContent: '{"notVisible":true}' }, + }), + ); + expect(assembler.finish()?.toolCalls[0]).not.toHaveProperty( + "modelResult", + ); + }); + + it.each([ + ["web_search", "github-mcp-server", "web_search"], + ["sample-fetch_data", "sample", "fetch_data"], + ])( + "preserves callable %s separately from MCP provenance through induction", + async (toolName, mcpServerName, mcpToolName) => { + const assembler = new ExtensionTraceAssembler("session-1", "."); + assembler.record( + event("user.message", "2026-09-29T08:00:00.000Z", { + content: "Run the benign fixture", + }), + ); + assembler.record( + event("tool.execution_start", "2026-09-29T08:00:01.000Z", { + toolCallId: "call-1", + toolName, + mcpServerName, + mcpToolName, + arguments: { query: "IANA example domains" }, + }), + ); + assembler.record( + event("tool.execution_complete", "2026-09-29T08:00:02.000Z", { + toolCallId: "call-1", + success: true, + result: { content: "Reserved for documentation" }, + }), + ); + const trace = assembler.finish(); + expect(trace?.toolCalls[0]).toMatchObject({ + name: toolName, + mcpServerName, + }); + if (!trace) throw new Error("Expected completed capture"); + const macro = await induceMacroFromTrace( + "trace-1", + trace, + "macro-1", + "Callable identity fixture", + "", + "2026-09-29T08:00:03.000Z", + ); + expect(macro.steps[0]).toMatchObject({ + toolName, + mcpServerName, + executionClass: "agentRequired", + arguments: { + kind: "literal", + value: { query: "IANA example domains" }, + }, + }); + }, + ); + it("builds a redacted trace from live session events", () => { const assembler = new ExtensionTraceAssembler("session-1", "C:\\repo"); assembler.record( diff --git a/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts b/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts index c1d65c952d..d27017be45 100644 --- a/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts +++ b/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts @@ -48,6 +48,33 @@ describe("staged plugin artifact", () => { expect(runner).toContain("If a required tool is unavailable, stop"); }); + it("resolves the captured web-search callable without treating provenance as a namespace", async () => { + const pluginRoot = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + "..", + "..", + ); + const runner = await readFile( + path.join(pluginRoot, "agents", "typeagent-macro-runner.agent.md"), + "utf8", + ); + + expect(runner).toContain("Check the already exposed tools"); + expect(runner).toContain("exact recorded callable first"); + expect(runner).toContain("functions.web_search"); + expect(runner).toContain("github-mcp-server/web_search"); + expect(runner).toContain("its input is `{ query: string }`"); + expect(runner).toContain("load its definition"); + expect(runner).toContain("Do not generalize this bridge"); + expect(runner).toContain( + "cannot establish the required tool's identity", + ); + expect(runner).toContain("Existing approved versions remain unchanged"); + expect(runner).toContain("Check every declared postcondition"); + expect(runner).toContain("Do not invent Copilot"); + expect(runner).toContain("Never change the old version"); + }); + it("consumes mode commands in the bundled hook without backend connections", async () => { const directory = await mkdtemp( path.join(tmpdir(), "typeagent-mode-artifact-"),