Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions ts/packages/copilot-macros/src/contracts.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,8 @@ export interface RecordedToolCall {
mcpServerName?: string;
arguments?: unknown;
result?: unknown;
// Parsed model-facing tool text, distinct from the raw event/UI result.
modelResult?: unknown;
status: "completed" | "failed" | "denied";
permission?: unknown;
}
Expand Down
39 changes: 26 additions & 13 deletions ts/packages/copilot-macros/src/macroDefinition.ts
Original file line number Diff line number Diff line change
Expand Up @@ -170,19 +170,36 @@ export async function induceMacroFromTrace(
const warnings: string[] = [];
const inputs: MacroInput[] = [];
const steps: MacroStep[] = [];
for (const [index, call] of trace.toolCalls.entries()) {
const id = `step-${index + 1}`;
const executionClass = await classifyTool(
call.name,
call.mcpServerName,
trace.cwd,
replayHost,
const executionClasses: MacroExecutionClass[] = [];
for (const call of trace.toolCalls) {
executionClasses.push(
await classifyTool(
call.name,
call.mcpServerName,
trace.cwd,
replayHost,
),
);
}
const agentRequired = executionClasses.includes("agentRequired");
const calls = trace.toolCalls.map((call) =>
agentRequired && call.modelResult !== undefined
? { ...call, result: call.modelResult }
: call,
);
for (const [index, call] of calls.entries()) {
const id = `step-${index + 1}`;
const executionClass = executionClasses[index];
if (executionClass === "agentRequired") {
warnings.push(
`${id} uses ${call.mcpServerName ? `${call.mcpServerName}/` : ""}${call.name} and requires agent-guided execution.`,
);
}
if (agentRequired && call.modelResult === undefined) {
warnings.push(
`${id} has no captured model-facing result. Review its result guards and recapture if the runner cannot observe the required fields.`,
);
}
if (call.status !== "completed") {
warnings.push(
`${id} was captured with status ${call.status} and requires review.`,
Expand All @@ -199,7 +216,7 @@ export async function induceMacroFromTrace(
id,
trace.prompt,
steps,
trace.toolCalls.slice(0, index),
calls.slice(0, index),
inputs,
warnings,
),
Expand All @@ -218,11 +235,7 @@ export async function induceMacroFromTrace(
name,
description,
state: "draft",
executionClass: steps.every(
(step) => step.executionClass === "replayable",
)
? "replayable"
: "agentRequired",
executionClass: agentRequired ? "agentRequired" : "replayable",
inputs,
steps,
sourceTraceId: traceId,
Expand Down
120 changes: 120 additions & 0 deletions ts/packages/copilot-macros/test/macroDefinition.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,126 @@ function trace(
}

describe("macro induction and validation", () => {
it("uses model-facing guards and references throughout a mixed agent procedure", async () => {
const source = trace({
result: {
content: '{"item":{"id":"item-123"}}',
detailedContent: "UI-only result",
},
modelResult: { item: { id: "item-123" } },
});
source.toolCalls.push({
toolCallId: "call-2",
name: "native_tool",
arguments: { itemId: "item-123" },
result: { content: '{"ok":true}', detailedContent: "UI-only" },
modelResult: { ok: true },
status: "completed",
});
const before = structuredClone(source);
const macro = await induceMacroFromTrace(
"trace-1",
source,
"macro-1",
"Mixed result views",
"",
"2026-09-29T08:00:00.000Z",
replayHost,
);

expect(source).toEqual(before);
expect(macro.executionClass).toBe("agentRequired");
expect(macro.steps[0].executionClass).toBe("replayable");
expect(macro.steps[0].postconditions).toEqual([
{ kind: "resultType", valueType: "object" },
{ kind: "resultPathExists", path: ["item", "id"] },
]);
expect(macro.steps[1]).toMatchObject({
executionClass: "agentRequired",
arguments: {
kind: "template",
bindings: [
{
path: ["itemId"],
expression: {
kind: "stepResult",
stepId: "step-1",
path: ["item", "id"],
},
},
],
},
postconditions: [
{ kind: "resultType", valueType: "object" },
{ kind: "resultPathExists", path: ["ok"] },
],
});
expect(validateMacro(macro, source).valid).toBe(true);
});

it.each([
[null, "null"],
[false, "boolean"],
[0, "number"],
["", "string"],
])(
"retains the model-facing primitive %s",
async (modelResult, valueType) => {
const macro = await induceMacroFromTrace(
"trace-1",
trace({ mcpServerName: null, modelResult }),
"macro-1",
"Primitive result",
"",
"2026-09-29T08:00:00.000Z",
);
expect(macro.steps[0].postconditions).toEqual([
{ kind: "resultType", valueType },
]);
expect(macro.warnings).not.toContainEqual(
expect.stringContaining("no captured model-facing result"),
);
},
);

it("keeps deterministic replay guards on raw results", async () => {
const macro = await induceMacroFromTrace(
"trace-1",
trace({
result: { raw: "value" },
modelResult: { visible: "value" },
}),
"macro-1",
"Replay result",
"",
"2026-09-29T08:00:00.000Z",
replayHost,
);
expect(macro.executionClass).toBe("replayable");
expect(macro.steps[0].postconditions).toEqual([
{ kind: "resultType", valueType: "object" },
{ kind: "resultPathExists", path: ["raw"] },
]);
});

it("preserves legacy guards and warns when model-facing evidence was not captured", async () => {
const macro = await induceMacroFromTrace(
"trace-1",
trace({ mcpServerName: null, result: { content: "legacy" } }),
"macro-1",
"Legacy result",
"",
"2026-09-29T08:00:00.000Z",
);
expect(macro.steps[0].postconditions).toContainEqual({
kind: "resultPathExists",
path: ["content"],
});
expect(macro.warnings).toContainEqual(
expect.stringContaining("no captured model-facing result"),
);
});

it("induces a replayable linear workspace draft", async () => {
const source = trace();
const macro = await induceMacroFromTrace(
Expand Down
27 changes: 27 additions & 0 deletions ts/packages/copilot-plugin/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -554,6 +554,33 @@ Macros containing Copilot-native tools use the macro runner for the whole
procedure. TypeAgent never replays a prefix and hands only the remainder to
the agent.

For agent-guided execution, `step.toolName` preserves Copilot's captured
callable name; `mcpServerName` preserves backend provenance, not a namespace
to prepend. For example, Copilot CLI exposes `web_search` (rendered as
`functions.web_search`) while its runtime metadata and execution events
identify the backend as `github-mcp-server/web_search`. The runner uses the
original callable and its live `{ query: string }` schema even when web search
is absent from the deferred GitHub catalog. This does not substitute providers,
grant permissions, or modify an existing approved macro. Unknown identities,
incompatible arguments, permission denials, and cancellations still stop the
run; no arbitrary aliases are inferred.

Captures retain the raw Copilot event result for audit and deterministic replay,
and separately retain the model-facing result text (parsed when it is JSON).
When the whole procedure requires the runner, induction uses the model-facing
values for normal result-type/path postconditions and prior-step bindings,
including any replayable steps within that procedure. UI-only envelope fields
are not evidence the runner can verify. Plain or truncated text remains a
string, not an invented structured result.

Older approved macros may require event-wrapper paths such as `content`,
`detailedContent`, and `contents` that are not shown to the runner. Such runs
must stop, not waive those guards. Record a new interaction after updating the
plugin and restarting Copilot, inspect its newly induced guards and inputs, and
explicitly approve the new draft. Existing traces without model-facing evidence
and existing immutable versions are not silently rewritten. Resolving tool
access alone does not make an old macro's result guards verifiable.

### Rollout Controls

Each macro boundary is enabled by default and can be disabled independently:
Expand Down
45 changes: 43 additions & 2 deletions ts/packages/copilot-plugin/agents/typeagent-macro-runner.agent.md
Original file line number Diff line number Diff line change
Expand Up @@ -20,8 +20,8 @@ that provide only a macro name or free-form procedure.
version as the launch payload.
2. Execute the whole macro in step order using the supplied inputs and prior
step results. Do not split execution between deterministic replay and this
runner. Discover the captured MCP tools from the live catalog when needed;
do not assume an MCP server name means TypeAgent can replay the tool.
runner. Resolve tools as described below; do not assume an MCP server name
means TypeAgent can replay the tool.
If a required tool is unavailable, stop and report it rather than silently
substituting another tool.
3. Use Copilot's live tool permissions. A denied or cancelled tool call is a
Expand All @@ -38,6 +38,47 @@ that provide only a macro name or free-form procedure.
The result must remain a draft for explicit review. Never approve, promote,
or mutate the approved version.

## Tool Identity

`step.toolName` is the callable name captured from Copilot's
`tool.execution_start` event. `step.mcpServerName` records backend provenance;
it is not an instruction to prepend a server name to the callable name.
Check the already exposed tools for the exact recorded callable first. For a
deferred tool, load its definition through the host's tool discovery before
invoking it. Check the resolved arguments against the live input schema.

In Copilot CLI, `toolName: "web_search"` with
`mcpServerName: "github-mcp-server"` identifies Copilot's exposed `web_search`
tool (shown as `functions.web_search` in the model's tool namespace). The CLI
runtime reports that callable's backend identity as
`github-mcp-server/web_search`; its input is `{ query: string }`. Calling that
exposed tool with the approved query is the original invocation, not a
provider substitution. It need not appear in the deferred GitHub tool list.
Do not reject it solely because that list omits it.

Do not generalize this bridge to other similarly named tools or providers.
Do not strip arbitrary prefixes, invent aliases, or use a different search
provider. If the live definition conflicts with the recorded identity or
arguments, or you cannot establish the required tool's identity, stop and
report the recorded name, backend provenance, and discovery evidence.
Resolving the original callable does not adapt the procedure and does not
require a candidate. Existing approved versions remain unchanged.

## Result Evidence

For postconditions and `stepResult` references, use the tool response actually
shown to you: parse it as JSON when it is valid JSON; otherwise use the text
as a string. Newly captured agent-required macros use that same representation
for their inferred type/path guards and result bindings.

Check every declared postcondition before continuing. Do not invent Copilot
event or UI wrapper fields such as `content`, `detailedContent`, or `contents`
around a displayed result. If an older approved macro requires fields you
cannot observe, stop with the specific unverifiable postconditions. Do not
drop guards, claim success, or submit a successful candidate. Report that the
user needs to record a new interaction with the updated plugin, inspect the
new draft's guards, and explicitly approve it. Never change the old version.

Do not call `run_macro` recursively. Do not submit a candidate after permission
denial, cancellation, timeout, or an unsuccessful adaptation.
Use tools only for the approved procedure and its inspection, required tool
Expand Down
17 changes: 17 additions & 0 deletions ts/packages/copilot-plugin/src/extension/trace-assembler.ts
Original file line number Diff line number Diff line change
Expand Up @@ -46,6 +46,19 @@ function permissionWasDenied(result: unknown): boolean {
return kind === "cancelled" || kind?.startsWith("denied") === true;
}

function getModelResult(result: unknown): unknown {
if (!result || typeof result !== "object") return undefined;
const content = (result as Record<string, unknown>).content;
if (typeof content !== "string") return undefined;
try {
return JSON.parse(content);
} catch (error) {
// Plain text (including truncated JSON) is the actual model-visible value.
if (error instanceof SyntaxError) return content;
throw error;
}
}

export class ExtensionTraceAssembler {
private turn: ActiveTurn | undefined;

Expand Down Expand Up @@ -163,6 +176,10 @@ export class ExtensionTraceAssembler {
if (!call || turn.completedCalls.has(key)) return;

call.result = redactTraceValue(event.data.result ?? event.data.error);
const modelResult = getModelResult(event.data.result);
if (modelResult !== undefined) {
call.modelResult = redactTraceValue(modelResult);
}
if (call.status !== "denied") {
call.status = event.data.success === false ? "failed" : "completed";
}
Expand Down
Loading
Loading