diff --git a/ts/packages/copilot-macros/src/contracts.ts b/ts/packages/copilot-macros/src/contracts.ts index 4695c66c94..3694b213f2 100644 --- a/ts/packages/copilot-macros/src/contracts.ts +++ b/ts/packages/copilot-macros/src/contracts.ts @@ -7,6 +7,8 @@ export interface RecordedToolCall { mcpServerName?: string; arguments?: unknown; result?: unknown; + // Parsed model-facing tool text, distinct from the raw event/UI result. + modelResult?: unknown; status: "completed" | "failed" | "denied"; permission?: unknown; } diff --git a/ts/packages/copilot-macros/src/macroDefinition.ts b/ts/packages/copilot-macros/src/macroDefinition.ts index 10bdbe613c..4211d45a57 100644 --- a/ts/packages/copilot-macros/src/macroDefinition.ts +++ b/ts/packages/copilot-macros/src/macroDefinition.ts @@ -170,19 +170,36 @@ export async function induceMacroFromTrace( const warnings: string[] = []; const inputs: MacroInput[] = []; const steps: MacroStep[] = []; - for (const [index, call] of trace.toolCalls.entries()) { - const id = `step-${index + 1}`; - const executionClass = await classifyTool( - call.name, - call.mcpServerName, - trace.cwd, - replayHost, + const executionClasses: MacroExecutionClass[] = []; + for (const call of trace.toolCalls) { + executionClasses.push( + await classifyTool( + call.name, + call.mcpServerName, + trace.cwd, + replayHost, + ), ); + } + const agentRequired = executionClasses.includes("agentRequired"); + const calls = trace.toolCalls.map((call) => + agentRequired && call.modelResult !== undefined + ? { ...call, result: call.modelResult } + : call, + ); + for (const [index, call] of calls.entries()) { + const id = `step-${index + 1}`; + const executionClass = executionClasses[index]; if (executionClass === "agentRequired") { warnings.push( `${id} uses ${call.mcpServerName ? `${call.mcpServerName}/` : ""}${call.name} and requires agent-guided execution.`, ); } + if (agentRequired && call.modelResult === undefined) { + warnings.push( + `${id} has no captured model-facing result. Review its result guards and recapture if the runner cannot observe the required fields.`, + ); + } if (call.status !== "completed") { warnings.push( `${id} was captured with status ${call.status} and requires review.`, @@ -199,7 +216,7 @@ export async function induceMacroFromTrace( id, trace.prompt, steps, - trace.toolCalls.slice(0, index), + calls.slice(0, index), inputs, warnings, ), @@ -218,11 +235,7 @@ export async function induceMacroFromTrace( name, description, state: "draft", - executionClass: steps.every( - (step) => step.executionClass === "replayable", - ) - ? "replayable" - : "agentRequired", + executionClass: agentRequired ? "agentRequired" : "replayable", inputs, steps, sourceTraceId: traceId, diff --git a/ts/packages/copilot-macros/test/macroDefinition.spec.ts b/ts/packages/copilot-macros/test/macroDefinition.spec.ts index 9874806503..478f4844ac 100644 --- a/ts/packages/copilot-macros/test/macroDefinition.spec.ts +++ b/ts/packages/copilot-macros/test/macroDefinition.spec.ts @@ -50,6 +50,126 @@ function trace( } describe("macro induction and validation", () => { + it("uses model-facing guards and references throughout a mixed agent procedure", async () => { + const source = trace({ + result: { + content: '{"item":{"id":"item-123"}}', + detailedContent: "UI-only result", + }, + modelResult: { item: { id: "item-123" } }, + }); + source.toolCalls.push({ + toolCallId: "call-2", + name: "native_tool", + arguments: { itemId: "item-123" }, + result: { content: '{"ok":true}', detailedContent: "UI-only" }, + modelResult: { ok: true }, + status: "completed", + }); + const before = structuredClone(source); + const macro = await induceMacroFromTrace( + "trace-1", + source, + "macro-1", + "Mixed result views", + "", + "2026-09-29T08:00:00.000Z", + replayHost, + ); + + expect(source).toEqual(before); + expect(macro.executionClass).toBe("agentRequired"); + expect(macro.steps[0].executionClass).toBe("replayable"); + expect(macro.steps[0].postconditions).toEqual([ + { kind: "resultType", valueType: "object" }, + { kind: "resultPathExists", path: ["item", "id"] }, + ]); + expect(macro.steps[1]).toMatchObject({ + executionClass: "agentRequired", + arguments: { + kind: "template", + bindings: [ + { + path: ["itemId"], + expression: { + kind: "stepResult", + stepId: "step-1", + path: ["item", "id"], + }, + }, + ], + }, + postconditions: [ + { kind: "resultType", valueType: "object" }, + { kind: "resultPathExists", path: ["ok"] }, + ], + }); + expect(validateMacro(macro, source).valid).toBe(true); + }); + + it.each([ + [null, "null"], + [false, "boolean"], + [0, "number"], + ["", "string"], + ])( + "retains the model-facing primitive %s", + async (modelResult, valueType) => { + const macro = await induceMacroFromTrace( + "trace-1", + trace({ mcpServerName: null, modelResult }), + "macro-1", + "Primitive result", + "", + "2026-09-29T08:00:00.000Z", + ); + expect(macro.steps[0].postconditions).toEqual([ + { kind: "resultType", valueType }, + ]); + expect(macro.warnings).not.toContainEqual( + expect.stringContaining("no captured model-facing result"), + ); + }, + ); + + it("keeps deterministic replay guards on raw results", async () => { + const macro = await induceMacroFromTrace( + "trace-1", + trace({ + result: { raw: "value" }, + modelResult: { visible: "value" }, + }), + "macro-1", + "Replay result", + "", + "2026-09-29T08:00:00.000Z", + replayHost, + ); + expect(macro.executionClass).toBe("replayable"); + expect(macro.steps[0].postconditions).toEqual([ + { kind: "resultType", valueType: "object" }, + { kind: "resultPathExists", path: ["raw"] }, + ]); + }); + + it("preserves legacy guards and warns when model-facing evidence was not captured", async () => { + const macro = await induceMacroFromTrace( + "trace-1", + trace({ mcpServerName: null, result: { content: "legacy" } }), + "macro-1", + "Legacy result", + "", + "2026-09-29T08:00:00.000Z", + ); + expect(macro.steps[0].postconditions).toContainEqual({ + kind: "resultPathExists", + path: ["content"], + }); + expect(macro.warnings).toContainEqual( + expect.stringContaining("no captured model-facing result"), + ); + }); + it("induces a replayable linear workspace draft", async () => { const source = trace(); const macro = await induceMacroFromTrace( diff --git a/ts/packages/copilot-plugin/README.md b/ts/packages/copilot-plugin/README.md index d1ca8cc681..b4624d1bdf 100644 --- a/ts/packages/copilot-plugin/README.md +++ b/ts/packages/copilot-plugin/README.md @@ -554,6 +554,33 @@ Macros containing Copilot-native tools use the macro runner for the whole procedure. TypeAgent never replays a prefix and hands only the remainder to the agent. +For agent-guided execution, `step.toolName` preserves Copilot's captured +callable name; `mcpServerName` preserves backend provenance, not a namespace +to prepend. For example, Copilot CLI exposes `web_search` (rendered as +`functions.web_search`) while its runtime metadata and execution events +identify the backend as `github-mcp-server/web_search`. The runner uses the +original callable and its live `{ query: string }` schema even when web search +is absent from the deferred GitHub catalog. This does not substitute providers, +grant permissions, or modify an existing approved macro. Unknown identities, +incompatible arguments, permission denials, and cancellations still stop the +run; no arbitrary aliases are inferred. + +Captures retain the raw Copilot event result for audit and deterministic replay, +and separately retain the model-facing result text (parsed when it is JSON). +When the whole procedure requires the runner, induction uses the model-facing +values for normal result-type/path postconditions and prior-step bindings, +including any replayable steps within that procedure. UI-only envelope fields +are not evidence the runner can verify. Plain or truncated text remains a +string, not an invented structured result. + +Older approved macros may require event-wrapper paths such as `content`, +`detailedContent`, and `contents` that are not shown to the runner. Such runs +must stop, not waive those guards. Record a new interaction after updating the +plugin and restarting Copilot, inspect its newly induced guards and inputs, and +explicitly approve the new draft. Existing traces without model-facing evidence +and existing immutable versions are not silently rewritten. Resolving tool +access alone does not make an old macro's result guards verifiable. + ### Rollout Controls Each macro boundary is enabled by default and can be disabled independently: diff --git a/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md b/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md index 542455f2c6..0c9a02e5ef 100644 --- a/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md +++ b/ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md @@ -20,8 +20,8 @@ that provide only a macro name or free-form procedure. version as the launch payload. 2. Execute the whole macro in step order using the supplied inputs and prior step results. Do not split execution between deterministic replay and this - runner. Discover the captured MCP tools from the live catalog when needed; - do not assume an MCP server name means TypeAgent can replay the tool. + runner. Resolve tools as described below; do not assume an MCP server name + means TypeAgent can replay the tool. If a required tool is unavailable, stop and report it rather than silently substituting another tool. 3. Use Copilot's live tool permissions. A denied or cancelled tool call is a @@ -38,6 +38,47 @@ that provide only a macro name or free-form procedure. The result must remain a draft for explicit review. Never approve, promote, or mutate the approved version. +## Tool Identity + +`step.toolName` is the callable name captured from Copilot's +`tool.execution_start` event. `step.mcpServerName` records backend provenance; +it is not an instruction to prepend a server name to the callable name. +Check the already exposed tools for the exact recorded callable first. For a +deferred tool, load its definition through the host's tool discovery before +invoking it. Check the resolved arguments against the live input schema. + +In Copilot CLI, `toolName: "web_search"` with +`mcpServerName: "github-mcp-server"` identifies Copilot's exposed `web_search` +tool (shown as `functions.web_search` in the model's tool namespace). The CLI +runtime reports that callable's backend identity as +`github-mcp-server/web_search`; its input is `{ query: string }`. Calling that +exposed tool with the approved query is the original invocation, not a +provider substitution. It need not appear in the deferred GitHub tool list. +Do not reject it solely because that list omits it. + +Do not generalize this bridge to other similarly named tools or providers. +Do not strip arbitrary prefixes, invent aliases, or use a different search +provider. If the live definition conflicts with the recorded identity or +arguments, or you cannot establish the required tool's identity, stop and +report the recorded name, backend provenance, and discovery evidence. +Resolving the original callable does not adapt the procedure and does not +require a candidate. Existing approved versions remain unchanged. + +## Result Evidence + +For postconditions and `stepResult` references, use the tool response actually +shown to you: parse it as JSON when it is valid JSON; otherwise use the text +as a string. Newly captured agent-required macros use that same representation +for their inferred type/path guards and result bindings. + +Check every declared postcondition before continuing. Do not invent Copilot +event or UI wrapper fields such as `content`, `detailedContent`, or `contents` +around a displayed result. If an older approved macro requires fields you +cannot observe, stop with the specific unverifiable postconditions. Do not +drop guards, claim success, or submit a successful candidate. Report that the +user needs to record a new interaction with the updated plugin, inspect the +new draft's guards, and explicitly approve it. Never change the old version. + Do not call `run_macro` recursively. Do not submit a candidate after permission denial, cancellation, timeout, or an unsuccessful adaptation. Use tools only for the approved procedure and its inspection, required tool diff --git a/ts/packages/copilot-plugin/src/extension/trace-assembler.ts b/ts/packages/copilot-plugin/src/extension/trace-assembler.ts index ae82cfdc0a..4d6e95bfbb 100644 --- a/ts/packages/copilot-plugin/src/extension/trace-assembler.ts +++ b/ts/packages/copilot-plugin/src/extension/trace-assembler.ts @@ -46,6 +46,19 @@ function permissionWasDenied(result: unknown): boolean { return kind === "cancelled" || kind?.startsWith("denied") === true; } +function getModelResult(result: unknown): unknown { + if (!result || typeof result !== "object") return undefined; + const content = (result as Record).content; + if (typeof content !== "string") return undefined; + try { + return JSON.parse(content); + } catch (error) { + // Plain text (including truncated JSON) is the actual model-visible value. + if (error instanceof SyntaxError) return content; + throw error; + } +} + export class ExtensionTraceAssembler { private turn: ActiveTurn | undefined; @@ -163,6 +176,10 @@ export class ExtensionTraceAssembler { if (!call || turn.completedCalls.has(key)) return; call.result = redactTraceValue(event.data.result ?? event.data.error); + const modelResult = getModelResult(event.data.result); + if (modelResult !== undefined) { + call.modelResult = redactTraceValue(modelResult); + } if (call.status !== "denied") { call.status = event.data.success === false ? "failed" : "completed"; } diff --git a/ts/packages/copilot-plugin/test/extensionTrace.spec.ts b/ts/packages/copilot-plugin/test/extensionTrace.spec.ts index a256fdc73e..836e0e0162 100644 --- a/ts/packages/copilot-plugin/test/extensionTrace.spec.ts +++ b/ts/packages/copilot-plugin/test/extensionTrace.spec.ts @@ -2,6 +2,7 @@ // Licensed under the MIT License. import { createHash } from "node:crypto"; +import { induceMacroFromTrace } from "@typeagent/copilot-macros"; import { ExtensionTraceAssembler } from "../src/extension/trace-assembler.js"; function event( @@ -14,6 +15,135 @@ function event( } describe("extension trace assembler", () => { + it.each([ + [ + '{"answer":{"value":"public","apiKey":"secret"}}', + { + answer: { value: "public", apiKey: "[REDACTED]" }, + }, + ], + ["null", null], + ["false", false], + ["0", 0], + ['""', ""], + ["", ""], + ["plain text", "plain text"], + ['{"truncated":', '{"truncated":'], + ])( + "captures model-facing content %s without UI-only fields", + (content, expected) => { + const assembler = new ExtensionTraceAssembler("session-1", "."); + assembler.record( + event("user.message", "2026-09-29T08:00:00.000Z", { + content: "Read the result", + }), + ); + assembler.record( + event("tool.execution_start", "2026-09-29T08:00:01.000Z", { + toolCallId: "call-1", + toolName: "web_search", + mcpServerName: "github-mcp-server", + }), + ); + assembler.record( + event("tool.execution_complete", "2026-09-29T08:00:02.000Z", { + toolCallId: "call-1", + success: true, + result: { + content, + detailedContent: "UI detail", + contents: [], + }, + }), + ); + const call = assembler.finish()?.toolCalls[0]; + expect(call?.modelResult).toEqual(expected); + expect(call?.result).toMatchObject({ + detailedContent: "UI detail", + contents: [], + }); + }, + ); + + it("does not manufacture model-facing evidence when content is absent", () => { + const assembler = new ExtensionTraceAssembler("session-1", "."); + assembler.record( + event("user.message", "2026-09-29T08:00:00.000Z", { + content: "Read", + }), + ); + assembler.record( + event("tool.execution_start", "2026-09-29T08:00:01.000Z", { + toolCallId: "call-1", + toolName: "read", + }), + ); + assembler.record( + event("tool.execution_complete", "2026-09-29T08:00:02.000Z", { + toolCallId: "call-1", + success: true, + result: { detailedContent: '{"notVisible":true}' }, + }), + ); + expect(assembler.finish()?.toolCalls[0]).not.toHaveProperty( + "modelResult", + ); + }); + + it.each([ + ["web_search", "github-mcp-server", "web_search"], + ["sample-fetch_data", "sample", "fetch_data"], + ])( + "preserves callable %s separately from MCP provenance through induction", + async (toolName, mcpServerName, mcpToolName) => { + const assembler = new ExtensionTraceAssembler("session-1", "."); + assembler.record( + event("user.message", "2026-09-29T08:00:00.000Z", { + content: "Run the benign fixture", + }), + ); + assembler.record( + event("tool.execution_start", "2026-09-29T08:00:01.000Z", { + toolCallId: "call-1", + toolName, + mcpServerName, + mcpToolName, + arguments: { query: "IANA example domains" }, + }), + ); + assembler.record( + event("tool.execution_complete", "2026-09-29T08:00:02.000Z", { + toolCallId: "call-1", + success: true, + result: { content: "Reserved for documentation" }, + }), + ); + const trace = assembler.finish(); + expect(trace?.toolCalls[0]).toMatchObject({ + name: toolName, + mcpServerName, + }); + if (!trace) throw new Error("Expected completed capture"); + const macro = await induceMacroFromTrace( + "trace-1", + trace, + "macro-1", + "Callable identity fixture", + "", + "2026-09-29T08:00:03.000Z", + ); + expect(macro.steps[0]).toMatchObject({ + toolName, + mcpServerName, + executionClass: "agentRequired", + arguments: { + kind: "literal", + value: { query: "IANA example domains" }, + }, + }); + }, + ); + it("builds a redacted trace from live session events", () => { const assembler = new ExtensionTraceAssembler("session-1", "C:\\repo"); assembler.record( diff --git a/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts b/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts index c1d65c952d..d27017be45 100644 --- a/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts +++ b/ts/packages/copilot-plugin/test/pluginArtifact.spec.ts @@ -48,6 +48,33 @@ describe("staged plugin artifact", () => { expect(runner).toContain("If a required tool is unavailable, stop"); }); + it("resolves the captured web-search callable without treating provenance as a namespace", async () => { + const pluginRoot = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + "..", + "..", + ); + const runner = await readFile( + path.join(pluginRoot, "agents", "typeagent-macro-runner.agent.md"), + "utf8", + ); + + expect(runner).toContain("Check the already exposed tools"); + expect(runner).toContain("exact recorded callable first"); + expect(runner).toContain("functions.web_search"); + expect(runner).toContain("github-mcp-server/web_search"); + expect(runner).toContain("its input is `{ query: string }`"); + expect(runner).toContain("load its definition"); + expect(runner).toContain("Do not generalize this bridge"); + expect(runner).toContain( + "cannot establish the required tool's identity", + ); + expect(runner).toContain("Existing approved versions remain unchanged"); + expect(runner).toContain("Check every declared postcondition"); + expect(runner).toContain("Do not invent Copilot"); + expect(runner).toContain("Never change the old version"); + }); + it("consumes mode commands in the bundled hook without backend connections", async () => { const directory = await mkdtemp( path.join(tmpdir(), "typeagent-mode-artifact-"),