diff --git a/apps/landing/astro.config.mjs b/apps/landing/astro.config.mjs index 29694ac703..d68980b05a 100644 --- a/apps/landing/astro.config.mjs +++ b/apps/landing/astro.config.mjs @@ -5,6 +5,7 @@ import { defineConfig } from "astro/config" import { transformerNotationDiff, transformerNotationHighlight } from "@shikijs/transformers" import { unified } from "@astrojs/markdown-remark" import rehypeTableWrap from "./src/lib/rehype-table-wrap.mjs" +import remarkInstallTabs from "./src/lib/remark-install-tabs.mjs" import { codeTheme } from "./src/lib/code-theme.mjs" import mdx from "@astrojs/mdx" import { paraglideVitePlugin } from "@inlang/paraglide-js" @@ -99,7 +100,8 @@ export default defineConfig({ markdown: { // Stay on the remark pipeline: Sätteri doesn't run the Shiki transformers // configured below. Revisit when the transformer story lands there. - processor: unified({ rehypePlugins: [rehypeTableWrap] }), + // MDX inherits these plugins. + processor: unified({ remarkPlugins: [remarkInstallTabs], rehypePlugins: [rehypeTableWrap] }), shikiConfig: { theme: codeTheme, wrap: true, diff --git a/apps/landing/src/components/docs/DocsCategoryIcon.astro b/apps/landing/src/components/docs/DocsCategoryIcon.astro index f9a509ba2b..ddea378586 100644 --- a/apps/landing/src/components/docs/DocsCategoryIcon.astro +++ b/apps/landing/src/components/docs/DocsCategoryIcon.astro @@ -110,7 +110,7 @@ const base = { )} -{name === "Agent Sessions" && ( +{(name === "Agent Sessions" || name === "AI Agents") && ( diff --git a/apps/landing/src/components/docs/DocsSidebar.astro b/apps/landing/src/components/docs/DocsSidebar.astro index ae51b5b460..8d28c97b71 100644 --- a/apps/landing/src/components/docs/DocsSidebar.astro +++ b/apps/landing/src/components/docs/DocsSidebar.astro @@ -5,6 +5,7 @@ // Language rows carry their logo; Effect's platform pages nest under Effect. import { getCollection } from "astro:content"; import DocsCategoryIcon from "./DocsCategoryIcon.astro"; +import BrandMarkIcon from "../BrandMarkIcon.astro"; import LanguageLogo from "./LanguageLogo.astro"; import { groupRank, isDocGroup, sectionForGroup } from "../../lib/docs-nav"; import { getDocSections } from "../../lib/docs-order"; @@ -73,8 +74,10 @@ const rowClass = (active: boolean) => aria-current={active ? "page" : undefined} class:list={["flex items-center gap-2 px-2.5 py-1.5 text-[13px] leading-5 transition-colors focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-inset focus-visible:ring-primary", rowClass(active)]} > - {doc.data.sdk && ( + {doc.data.sdk ? ( + ) : ( + doc.data.icon && )} {label(doc)} diff --git a/apps/landing/src/components/docs/GuideGrid.astro b/apps/landing/src/components/docs/GuideGrid.astro index 393057ec48..32feee9721 100644 --- a/apps/landing/src/components/docs/GuideGrid.astro +++ b/apps/landing/src/components/docs/GuideGrid.astro @@ -1,17 +1,19 @@ --- -// One ecosystem's cards on /docs/instrumentation. Styled with scoped CSS on +// One ecosystem's cards on /docs/instrumentation (or a given card list, as on /docs/frontend). Styled with scoped CSS on // purpose: the page sits inside `.docs-content`, whose descendant rules for // `a`/`p` beat plain Tailwind utilities (see `local/InstallTabs.astro`). -import { GUIDE_SECTIONS } from "../../lib/instrumentation-guides"; +import { type GuideCard, GUIDE_SECTIONS } from "../../lib/instrumentation-guides"; import BrandMarkIcon from "../BrandMarkIcon.astro"; import LanguageLogo from "./LanguageLogo.astro"; interface Props { - section: string; + /** A section of GUIDE_SECTIONS, or `cards` for a list shown elsewhere. */ + section?: string; + cards?: readonly GuideCard[]; } -const { section } = Astro.props; -const data = GUIDE_SECTIONS.find((s) => s.id === section); +const { section, cards } = Astro.props; +const data = cards ? { cards } : GUIDE_SECTIONS.find((s) => s.id === section); if (!data) throw new Error(`Unknown guide section: ${section}`); --- diff --git a/apps/landing/src/components/docs/LanguageLogo.astro b/apps/landing/src/components/docs/LanguageLogo.astro index a3ef6387ad..5a255b1305 100644 --- a/apps/landing/src/components/docs/LanguageLogo.astro +++ b/apps/landing/src/components/docs/LanguageLogo.astro @@ -1,8 +1,8 @@ --- -// Maps a `LanguageId` to its brand logo. The only place the logo components +// Maps a `LogoId` to its brand logo. The only place the logo components // are wired to language ids — everything else passes the id around. import type { AstroComponentFactory } from "astro/runtime/server/index.js"; -import type { LanguageId } from "../../lib/docs-languages"; +import type { LogoId } from "../../lib/docs-languages"; import EffectLogomark from "../icons/EffectLogomark.astro"; import NodeJsLogo from "../icons/NodeJsLogo.astro"; import NextjsLogo from "../icons/NextjsLogo.astro"; @@ -13,9 +13,10 @@ import OpenJdkLogo from "../icons/OpenJdkLogo.astro"; import CsharpLogo from "../icons/CsharpLogo.astro"; import KotlinLogo from "../icons/KotlinLogo.astro"; import LaravelLogo from "../icons/LaravelLogo.astro"; +import TypeScriptLogo from "../icons/TypeScriptLogo.astro"; interface Props { - id: LanguageId; + id: LogoId; class?: string; } @@ -30,7 +31,8 @@ const LOGOS = { kotlin: KotlinLogo, csharp: CsharpLogo, laravel: LaravelLogo, -} satisfies Record; + typescript: TypeScriptLogo, +} satisfies Record; const { id, class: className = "h-4 w-4" } = Astro.props; const Logo = LOGOS[id]; diff --git a/apps/landing/src/components/docs/LanguageTabs.astro b/apps/landing/src/components/docs/LanguageTabs.astro index 9fadd55c1e..512c5cb981 100644 --- a/apps/landing/src/components/docs/LanguageTabs.astro +++ b/apps/landing/src/components/docs/LanguageTabs.astro @@ -2,9 +2,11 @@ // Language/framework switcher for docs pages. Children are `` // panels holding ordinary markdown, so code blocks keep their highlighting and // copy buttons. The pick is remembered and shared by every group on the site. -// Without JavaScript the first panel shows and the rest stay hidden. +// Without JavaScript the first panel shows and the rest stay hidden. Styles live +// in global.css and behavior in lib/docs-tabs.ts, shared with the package-manager +// switchers remark-install-tabs.mjs renders. import LanguageLogo from "./LanguageLogo.astro"; -import { LANGUAGE_IDS, type LanguageId } from "../../lib/docs-languages"; +import { LOGO_IDS, type LogoId } from "../../lib/docs-languages"; interface Tab { id: string; @@ -17,10 +19,10 @@ interface Props { } const { tabs, label = "Language" } = Astro.props; -const isLanguage = (id: string): id is LanguageId => (LANGUAGE_IDS as readonly string[]).includes(id); +const hasLogo = (id: string): id is LogoId => (LOGO_IDS as readonly string[]).includes(id); --- -
+
{tabs.map((tab, index) => (
- - diff --git a/apps/landing/src/components/icons/TypeScriptLogo.astro b/apps/landing/src/components/icons/TypeScriptLogo.astro new file mode 100644 index 0000000000..e299a1850c --- /dev/null +++ b/apps/landing/src/components/icons/TypeScriptLogo.astro @@ -0,0 +1,17 @@ +--- +interface Props { + class?: string; +} + +const { class: className = "w-3.5 h-3.5" } = Astro.props; +--- + + diff --git a/apps/landing/src/content.config.ts b/apps/landing/src/content.config.ts index 961b5ef88b..cba585c4bd 100644 --- a/apps/landing/src/content.config.ts +++ b/apps/landing/src/content.config.ts @@ -1,5 +1,6 @@ import { defineCollection, reference, z } from "astro:content" import { glob } from "astro/loaders" +import { BRAND_MARKS, type BrandMarkId } from "./lib/brand-marks" import { LANGUAGE_IDS } from "./lib/docs-languages" import { CHANGELOG_CATEGORIES, CONTRIBUTOR_IDS } from "./lib/changelog-meta" @@ -28,6 +29,9 @@ const docs = defineCollection({ // ("Node.js" for "Node.js Instrumentation"). navLabel: z.string().optional(), sdk: z.enum(LANGUAGE_IDS).optional(), + // Brand mark on the sidebar row, for guides that aren't a language (`sdk`), + // like the frontend framework guides. + icon: z.enum(Object.keys(BRAND_MARKS) as [BrandMarkId, ...BrandMarkId[]]).optional(), }), }) diff --git a/apps/landing/src/content/docs/agent-sessions/overview.md b/apps/landing/src/content/docs/agent-sessions/overview.md index 22bcabcdaf..46d8ac7d3c 100644 --- a/apps/landing/src/content/docs/agent-sessions/overview.md +++ b/apps/landing/src/content/docs/agent-sessions/overview.md @@ -1,197 +1,89 @@ --- title: "Agent Sessions" -description: "An AI agent conversation as one session: every turn, model call and tool call with its cost, timing and failures, built from OpenTelemetry GenAI traces. What Maple records, and how to connect an agent in any language or framework." +description: "Agent Sessions groups the OpenTelemetry traces of one AI agent conversation into a single view of its turns, model calls, tool calls, tokens, cost and failures." group: "Agent Sessions" order: 1 navLabel: "Overview" --- -**Agent Sessions** shows an AI agent conversation as one session: every turn, model call and tool call, with its cost, timing and failures. Maple builds sessions from the OpenTelemetry traces your agent already sends, using the OpenTelemetry GenAI semantic conventions. There is no extra SDK. +Each user message in a conversation is usually its own trace. **Agent Sessions** groups those traces into one conversation and shows it turn by turn. To send your agent's traces, pick your framework in [Trace your AI agent](/docs/agent-tracing). -
- No extra SDK - OpenTelemetry GenAI semconv - 20+ frameworks - Any language -
- -For each session you get: - -- an overview that splits the wall clock into model time, tool time and idle, and rolls up cost and tokens per model; -- the transcript, with each model call's model, tokens, cost and finish reason; -- the trace, grouped by turn; -- tool pages that rank every tool by volume, failure rate and latency across sessions. - -The same data is on the [MCP server](/docs/reference/mcp) as `list_agent_sessions`, `get_agent_session`, `get_agent_tools_overview` and `get_agent_tool_error`. - -## Connect your agent - -### Step 1: traces are flowing +## Sessions, turns and calls -Your service exports OTLP to `https://ingest.maple.dev` (`https://ingest.eu.maple.dev` for EU organizations) with an ingest key. If it does not yet, pick your language in [Instrument your application](/docs/instrumentation) and come back. Agent Sessions adds nothing to that setup; it reads the traces that already arrive. +| Level | What it is | Where it comes from | +| --- | --- | --- | +| **Session** | One conversation, from the first message to the last. | Every trace that carries the same session id, such as `gen_ai.conversation.id`. | +| **Turn** | One user message and everything the agent did to answer it. | Usually one trace, rooted at an `invoke_agent` span. | +| **Model call** | One request to an LLM: the prompt, the reply, tokens, finish reason. | A `chat` (or `generate_content`, `text_completion`) span. | +| **Tool call** | One function the model asked to run, with its arguments and result. | An `execute_tool` span. | -### Step 2: emit GenAI spans +Sub-agents show up inside a turn as their own lane, labeled with their `gen_ai.agent.name`. A background job with no user is also a session, usually one trace long. -There are two ways to get there. If your agent runs on a [framework Maple recognizes](#frameworks-maple-recognizes-automatically), turn on that framework's OpenTelemetry export. Otherwise, emit the **OpenTelemetry GenAI semantic conventions** directly. We recommend this path: every framework integration is normalized into these conventions anyway, and they work in every language. +## Read a session -The conventions live at [opentelemetry.io/docs/specs/semconv/gen-ai](https://opentelemetry.io/docs/specs/semconv/gen-ai/). The pages you will actually use: - -- [Inference spans](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-spans/): the `chat`, `text_completion` and `embeddings` spans, the request and response attributes, and the token usage attributes. -- [Agent and tool spans](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/): `invoke_agent`, `create_agent` and `execute_tool`, and the `gen_ai.tool.*` attributes. -- [Message content](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-events/): the `{role, parts}` shape of `gen_ai.input.messages`, `gen_ai.output.messages` and `gen_ai.system_instructions`. -- [Attribute registry](https://opentelemetry.io/docs/specs/semconv/registry/attributes/gen-ai/): every `gen_ai.*` key with its type and examples. - -The session id is `gen_ai.conversation.id`: put the same value on every span of a conversation and its traces become one session. Everything else Maple reads is in [the attribute table](#the-attributes-maple-reads) below. - -### Step 3: open Explore → Agent Sessions - -Run one conversation and open the list. The session appears as soon as its first trace lands, and the overview and transcript fill in as the rest of the turns arrive. If it does not look like the screenshots in [A session, step by step](#a-session-step-by-step), [When it does not look right](#when-it-does-not-look-right) covers the five usual reasons. - -### The attributes Maple reads - -You do not need all of these. The first two rows make a session; the rest make it useful. - -| Attribute | Used for | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `gen_ai.operation.name` | Marks the span as AI and says what kind: `chat`, `text_completion`, `embeddings`, `execute_tool`, `invoke_agent`, `create_agent`, `invoke_workflow`. | -| `gen_ai.conversation.id` | Groups traces into one session: the same value on every span of a conversation. | -| `gen_ai.provider.name`, `gen_ai.request.model`, `gen_ai.response.model` | Provider and model facets, per-model token and cost roll-ups. The older `gen_ai.system` is accepted too. | -| `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_creation.input_tokens`, `gen_ai.usage.reasoning.output_tokens` | The five token buckets. Maple knows which providers nest one bucket inside another and does not double count. | -| `gen_ai.usage.cost` | Cost in USD, if your instrumentation prices calls. The conventions define no cost attribute, so Maple does not price semconv calls itself; this is the OpenLLMetry key, and `llm.cost.total` from OpenInference is read too. | -| `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions` | The transcript. | -| `gen_ai.agent.name`, `gen_ai.agent.id`, `gen_ai.agent.description` | Agent facet, and sub-agent handoffs inside a session. | -| `gen_ai.tool.name`, `gen_ai.tool.call.id`, `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result`, `gen_ai.tool.definitions` | Tool calls in the transcript, and everything on the tool pages. | -| `gen_ai.response.id`, `gen_ai.response.finish_reasons`, `gen_ai.response.time_to_first_chunk` | Response metadata and time to first token. | -| `error.type` plus span status | Failed model and tool calls, and the failure groups on the tool pages. | - -Retrieval, memory, embeddings and evaluation attributes from the conventions are stored and shown on the span, but do not change how the session is built. - -## Frameworks Maple recognizes automatically - -If your agent runs on one of these, use the framework's own OpenTelemetry exporter or the instrumentation listed and point it at Maple. Maple recognizes the framework at ingest, normalizes its attribute dialect into the `gen_ai.*` fields above, and takes the session id from wherever that framework keeps it. - -| Framework | Instrumentation Maple recognizes | Session id | -| -------------------------------- | --------------------------------------------------- | --------------------------------------------------------- | -| Vercel AI SDK | The SDK's `experimental_telemetry` | One per trace | -| OpenAI Agents SDK | OpenInference `openai_agents` instrumentation | `session.id` or `gen_ai.conversation.id` | -| Claude Code and Claude Agent SDK | Built-in OpenTelemetry telemetry | `session.id` | -| LangChain and LangGraph | LangSmith OpenTelemetry export | `langsmith.metadata.thread_id` | -| LlamaIndex | LlamaIndex's OpenTelemetry observability package | One per trace | -| Pydantic AI | Built-in (`instrument=True`) or Logfire | `gen_ai.conversation.id` | -| Mastra | `@mastra/otel-exporter` | `gen_ai.conversation.id` | -| Google ADK | Built-in OpenTelemetry tracing | `gen_ai.conversation.id` or `gcp.vertex.agent.session_id` | -| Microsoft Agent Framework | Built-in OpenTelemetry tracing | `gen_ai.conversation.id` | -| Semantic Kernel | Built-in OpenTelemetry tracing | One per trace | -| Spring AI | Spring Boot observability over OTLP | `spring.ai.chat.client.conversation.id` | -| CrewAI | Built-in telemetry or OpenInference instrumentation | `session.id` | -| DSPy | OpenInference `dspy` instrumentation | `session.id` | -| smolagents | OpenInference `smolagents` instrumentation | `session.id` | -| Agno | OpenInference `agno` instrumentation | `session.id` | -| Strands Agents | Built-in OpenTelemetry tracing | `session.id` | -| Haystack | Built-in OpenTelemetry tracer | One per trace | -| LiteLLM | Built-in OpenTelemetry callback | One per trace | -| OpenRouter | Broadcast to an OTLP endpoint | `session.id` | -| Effect AI | `@effect/opentelemetry` | One per trace | -| OpenAI SDK via OpenInference | OpenInference `openai` instrumentation | `session.id` | - -### Claude Code - -Claude Code's tracing is a beta behind its own flag, and its content is redacted unless you opt in. Set these in the shell that runs `claude`, or under `env` in `~/.claude/settings.json` to cover every session including the desktop app: - -```bash -export CLAUDE_CODE_ENABLE_TELEMETRY=1 -export CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1 # traces; without it there are no spans to build a session from -export OTEL_TRACES_EXPORTER=otlp -export OTEL_LOGS_EXPORTER=otlp # events: prompts, responses, per-request cost -export OTEL_METRICS_EXPORTER=otlp # optional: cost and token counters -export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf -export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" # https://ingest.eu.maple.dev for EU orgs -export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" - -# Content. Each is off by default; turn on what you are allowed to store. From Claude Code -# v2.1.193 an unset OTEL_LOG_ASSISTANT_RESPONSES follows OTEL_LOG_USER_PROMPTS: set it to 0 -# explicitly if prompts are approved and replies are not. -export OTEL_LOG_USER_PROMPTS=1 # the prompt that opens each turn -export OTEL_LOG_TOOL_DETAILS=1 # tool arguments: Bash commands, file paths, MCP tool names -export OTEL_LOG_TOOL_CONTENT=1 # tool results for Read, Bash, Edit and Write (not MCP tools) -export OTEL_LOG_ASSISTANT_RESPONSES=1 # the assistant's replies, on the assistant_response event -``` - -What each span becomes: - -| Claude Code span | In Agent Sessions | -| ---------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `claude_code.interaction` | A turn, titled with the prompt when `OTEL_LOG_USER_PROMPTS=1`. | -| `claude_code.llm_request` | A model call: model, the four token buckets (Anthropic's input count excludes the cache buckets, and Maple counts it that way), TTFT, finish reason and failures. | -| `claude_code.tool` | A tool call, with the command or file path as its arguments and, for Read, Bash, Edit and Write, the `tool.output` content as its result. | -| `claude_code.tool.execution`, `claude_code.tool.blocked_on_user` | Shown in the trace as part of their tool call rather than as calls of their own. A failed run marks the tool call failed, with its error. | - -Cost and the assistant's reply text are only on Claude Code's log events (`api_request`, `assistant_response`), not on its spans. They are stored and searchable under Logs; the session views read spans, so cost reads as unpriced there for now. - -For the frameworks that give you one session per trace, stamp `gen_ai.conversation.id` on every span of the conversation (a span processor is the usual place) and Maple groups them into one session. - -Two dialects that are not frameworks are recognized as well: any **OpenInference** emitter (`openinference.span.kind`, `llm.*`, `input.value`) and any **OpenLLMetry / Traceloop** emitter (`traceloop.*`, `llm.*`). Their spans land as sessions without a framework name attached: one per trace, or one per conversation when the spans carry `gen_ai.conversation.id`. - -**Don't see yours?** Send us the framework and a sample trace at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP). Adding a framework is a detection rule and a dialect map on our side, not a new SDK, so it is usually quick. In the meantime, anything that emits `gen_ai.operation.name` already works through the generic path above. - -## A session, step by step - -One conversation from a support agent instrumented with the OpenTelemetry GenAI conventions: the customer asks to change a delivery address, gives one in Paris, and ends up canceling the order. This is what Maple recorded. +The example below is a support agent where the customer asks to change a delivery address and ends up canceling the order.
A session's overview page: a time breakdown bar, a findings list with a failed tool call, a tools table with calls, failures and a timeline, and a right column with cost by model and token buckets. -
The session overview. The failed tool call leads the page: update_shipping_address returned unsupported_destination in turn 2.
+
The overview. The failed tool call leads the page: update_shipping_address returned unsupported_destination in turn 2.
-The overview splits the wall clock into model time, tool time and idle, and rolls cost and tokens up per model. Here the agent was busy for 14 seconds of 2 minutes 25 seconds. The rest was the customer typing. +The **overview** splits the session's wall clock into model time, tool time and idle time, and totals cost and tokens per model. + +Below that is the verdict (completed, completed with warnings, or failed, with the span that ended it) and the checks behind it, failing ones first. A check that needs data your instrumentation doesn't record says it was skipped and what to capture.
The transcript view of a session: system instructions, user and assistant messages in sequence, each model call annotated with its model, tokens, cost and finish reason, and a tool call row with its latency and payload sizes.
The transcript. Each model call carries its model, tokens, cost and finish reason; tool calls sit where the model made them.
+The **transcript** is the conversation as the model saw it: system instructions, user and assistant messages, and tool calls with arguments and results. It needs message content on your spans, which most instrumentations leave off by default; each framework guide shows the switch. +
The trace view of a session: three turns, each with an invoke_agent span, chat spans labeled with their model and token counts, and execute_tool spans, on a time axis with the idle gaps between turns removed. One tool span is marked with its error type. -
The trace. Spans grouped by turn, 2m 10s of idle cut from the axis, the failed tool span flagged with its error.type.
+
The trace. Spans grouped by turn, with 2 minutes 10 seconds of idle time cut from the axis and the failed tool span flagged.
+The **trace** view puts every span of every turn on one time axis, with the idle time between turns removed. +
The Agent Sessions list in Maple, one row per session with services, model, duration, LLM call and tool call counts, tokens, cost, errors and start time, and a filter sidebar on the left. -
The list. One row per conversation for your retention period, filterable by framework, service, model, agent and tool; failed ones marked.
+
The list. One row per conversation, filterable by framework, service, environment, model, agent and tool.
-Tools get the same treatment across sessions: ranked by volume, failure rate and latency, with each failure grouped by error type and the arguments and results that produced it. See [Debug and monitor tools](#debug-and-monitor-tools). +The **list** has one row per session with its model, duration, call counts, tokens, cost and errors. Sort by cost to find expensive conversations, or filter to sessions that called a given tool. -## Debug and monitor tools +## Find the tools that fail -Every `execute_tool` span is ranked, charted and grouped across sessions, so the three questions behind a misbehaving agent each take one click. +The **Tools** tab covers every tool call across all sessions.
The Tools tab: a metric strip with tool calls, sessions, error rate and duration, a chart of calls over time, and a table ranking each tool by calls, p50, p90, p95, error rate, errors, sessions and last call. -
Which tool is failing? The Tools tab ranks every tool by calls, latency percentiles and error rate for the window. Failing only keeps just the ones that have failed.
+
Which tool is failing? Every tool ranked by calls, latency percentiles and error rate. Failing only keeps just the ones that have failed.
- A tool's detail page: four charts for tool calls, error rate, duration percentiles and calls per session, then an errors table with one row per error type showing trend, share, count, sessions and last seen. -
Why? A tool's page charts its calls, error rate, duration and calls per session, then groups its failures by error.type with a trend, share and the sessions hit.
+ A tool's detail page: four charts for tool calls, error rate, duration percentiles and calls per session, then an errors table with one row per error showing trend, share, count, sessions and last seen. +
Why? A tool's page charts its calls, error rate, duration and calls per session, then groups its failures with a trend and the sessions they hit.
The error group dialog for a tool: the error type, how many calls and sessions it affected, a failed-calls-per-day chart, a Where it happens panel listing the model and service, a sessions list, and a sample failed call with its JSON arguments and JSON result side by side. -
What exactly happened? An error group opens on the failed calls themselves: arguments on the left, the result on the right, and a jump into the trace.
+
What exactly happened? An error group opens on the failed calls themselves: arguments on the left, the result on the right, and a link to the trace.
-
- A tool span opened from the session overview: an ERROR banner with the error type and message, the result JSON, the span's timing and identifiers, and its AI attributes including operation, conversation id and tool name. -
Inside a session, a failed tool is a finding on the overview; opening it shows the error, the result and every attribute the span carried.
-
+A tool call counts as failed when its span has an `ERROR` status or an `error.type` attribute. Failures with the same message, ignoring ids and numbers, form one group. + +## Query sessions from your coding agent + +The [MCP server](/docs/reference/mcp) exposes the same data. `list_agent_sessions` finds sessions by cost, model, tool or failure, `get_agent_session` returns a session's verdict, checks and turns, and `get_agent_tools_overview` and `get_agent_tool_error` return the tool rankings and failure groups. -A tool span needs `gen_ai.operation.name` of `execute_tool`, `gen_ai.tool.name`, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, and on failure a span status of `ERROR` with a stable `error.type`. Groups are keyed on `error.type`, with ids and timestamps masked so one failure is one row. `get_agent_tools_overview` and `get_agent_tool_error` on the [MCP server](/docs/reference/mcp) return the same ranking and groups. +## When a session doesn't look right -## When it does not look right +- **Every message is its own session.** No span carried a session id Maple reads (`gen_ai.conversation.id` for most frameworks, `session.id` for some). The framework's guide shows where to set it. +- **The transcript is empty.** Content capture is off, or the framework writes content only to span events or logs. Content on span attributes must be a JSON string such as a `[{role, parts}]` array. +- **Cost shows as unpriced.** Maple shows cost only when spans carry `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or `llm.cost.total`; it doesn't price tokens itself. +- **Token totals look doubled.** Two instrumentations recorded the same model call, usually the framework's and a provider SDK instrumentor. Turn one off. +- **Nothing appears at all.** Check **Explore → Traces** for the service first. No traces there means the exporter isn't reaching Maple, often a short-lived script that exits before flushing. Traces there but no session means the framework's tracing isn't on. -- **Every turn is its own session.** No span carried a session id Maple recognizes. If the framework table lists a session key for your framework, check that its spans carry it; otherwise add `gen_ai.conversation.id` to every span of the conversation. -- **The transcript is empty.** Message content is not on the spans. Most official instrumentations leave it off by default, and some only ever write it to log events, which Maple does not read. If content is on the spans and still missing, check that the attribute holds a JSON array of `{role, parts}` objects rather than a plain string. -- **The framework shows as "Unidentified".** The spans carry `gen_ai.*` attributes but no fingerprint of a known framework. Sessions, transcripts and tool pages all work; only the framework facet is missing. Tell us which framework it is and we will add the rule. -- **Token totals look too high or too low.** Providers disagree on whether cached and reasoning tokens are included in the input and output counts. Maple resolves that per `gen_ai.provider.name`, so if the provider name is missing or unexpected, set it and the totals correct themselves. -- **Nothing appears at all.** Confirm ordinary traces from the service show under **Explore → Traces** first. If they do, no span in them carries `gen_ai.operation.name`; if they do not, the problem is the exporter, and the [instrumentation guide](/docs/instrumentation) for your language covers it. +A framework shown as **Unidentified** still gets sessions, transcripts and tools. If yours has no guide, [the OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry) works for any agent, and a sample trace sent to [support@maple.dev](mailto:support@maple.dev) or [Discord](https://discord.gg/BnXjKuwJqP) helps us add one. diff --git a/apps/landing/src/content/docs/agent-tracing.mdx b/apps/landing/src/content/docs/agent-tracing.mdx new file mode 100644 index 0000000000..6048cc0e21 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing.mdx @@ -0,0 +1,56 @@ +--- +title: "Trace your AI agent" +description: "Setup guides for sending each supported agent framework and LLM gateway to Maple as one Agent Session per conversation." +group: "AI Agents" +order: 0 +navLabel: "Overview" +--- + +import GuideGrid from "../../components/docs/GuideGrid.astro" +import { AGENT_GUIDE_CATEGORIES } from "../../lib/agent-tracing-guides" + +Each guide sets up one framework so every conversation shows up in Maple as one [Agent Session](/docs/agent-sessions/overview). You turn on the framework's own OpenTelemetry instrumentation and point it at Maple; there is no Maple SDK. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) skill, which detects your framework and installs the matching skill. + +```text +Set up Maple agent tracing in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Choose your framework + +{AGENT_GUIDE_CATEGORIES.map((category) => ( + <> +

{category.title}

+ + +))} + +If you use a gateway like OpenRouter or LiteLLM together with a framework, instrument only the framework, or every model call is recorded twice. The [OpenRouter guide](/docs/agent-tracing/openrouter) covers the one exception. + +## Connection details + +Every guide exports OTLP over HTTP with an ingest key from **Settings → Ingestion**: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" # https://ingest.eu.maple.dev for EU organizations +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_SERVICE_NAME="support-agent" +``` + +Maple doesn't accept gRPC, so set the protocol to `http/protobuf` if your exporter defaults to it. A traces endpoint passed in code or through `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` must end in `/v1/traces`. + +Keep agent traffic at 100% sampling, or sampled-out turns leave gaps in the session. See [Sampling and throughput](/docs/concepts/sampling-throughput). + +## Not listed? + +Anything that emits the OpenTelemetry GenAI conventions works; [Any language](/docs/agent-tracing/opentelemetry) lists what Maple reads. OpenInference and OpenLLMetry instrumentations also work, shown as **Unidentified**. Tell us what you run at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP) and we'll add a guide. diff --git a/apps/landing/src/content/docs/agent-tracing/agno.md b/apps/landing/src/content/docs/agent-tracing/agno.md new file mode 100644 index 0000000000..09d23d84fa --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/agno.md @@ -0,0 +1,153 @@ +--- +title: "Trace Agno agents and teams with OpenTelemetry" +description: "Send Agno agent, team and tool spans to Maple as one Agent Session per conversation, with transcript, tokens and failed tools." +group: "AI Agents" +order: 26 +navLabel: "Agno" +icon: "agno" +--- + +This guide sends Agno's OpenInference spans to Maple with an OTLP exporter. Agno's own `setup_tracing()` writes to your AgentOS database, not to Maple. + +Pass `session_id` on every run. Without it, a server with one shared `Agent` puts every user's conversation into the same Maple session. + +Tested with Agno 3.0.11 and `openinference-instrumentation-agno` 1.0.10 on Python 3.10 to 3.14. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-agno](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-agno) skill and follows it. + +```text +Set up Maple agent tracing for Agno in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-agno -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install and point the exporter at Maple + +```bash +pip install -U "agno>=3.0" "openinference-instrumentation-agno>=1.0.10" \ + opentelemetry-sdk opentelemetry-exporter-otlp-proto-http +``` + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +export AGNO_TELEMETRY=false +``` + +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. `AGNO_TELEMETRY=false` turns off Agno's anonymous usage pings. + +## Initialize tracing + +Create one tracer provider in `tracing.py` and import it at the top of your entry point, before you build agents or an `AgentOS`: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.agno import AgnoInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider() +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +AgnoInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +If you can't change the `instrument()` call, set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` instead. + +### Keep the AgentOS traces view + +Once `tracing.py` runs, `AgentOS(tracing=True)` and `setup_tracing(db=...)` stop filling the AgentOS traces view. To keep it, add Agno's database exporter to your provider, with the same `db` you give AgentOS: + +```py +from agno.tracing.exporter import DatabaseSpanExporter + +provider.add_span_processor(BatchSpanProcessor(DatabaseSpanExporter(db=db))) +``` + +## Pass session_id on every run + +Each `run()` or `arun()` is its own trace. Pass `session_id` to group them into one session: + +```py +from agno.agent import Agent +from agno.db.sqlite import SqliteDb +from agno.models.openrouter import OpenRouter + +# Built once at import time and shared by every request +agent = Agent( + name="support_agent", + model=OpenRouter(id="openai/gpt-4o-mini"), + db=SqliteDb(db_file="tmp/agno.db"), + tools=[get_weather, calculate], + add_history_to_context=True, +) + + +def chat(conversation_id: str, user_id: str, message: str) -> str: + response = agent.run(message, session_id=conversation_id, user_id=user_id) + return response.content + + +async def chat_stream(conversation_id: str, user_id: str, message: str): + async for event in agent.arun( + message, stream=True, session_id=conversation_id, user_id=user_id + ): + if getattr(event, "content", None): + yield event.content +``` + +Use the conversation id your app already stores, and pass it to `team.run()`, workflows and `continue_run()` as well. + +## Teams, failed tools and content + +Give every `Agent` and `Team` a `name=`. Unnamed ones show up as `Agent.run` or `Team.run` with no lane of their own. + +A tool that raises is marked failed. A tool that returns an error string counts as a success. + +To keep content out of Maple, set `OPENINFERENCE_HIDE_INPUT_MESSAGES`, `OPENINFERENCE_HIDE_OUTPUT_MESSAGES`, `OPENINFERENCE_HIDE_INPUTS` and `OPENINFERENCE_HIDE_OUTPUTS` to `true` before the instrumentor starts. Tool arguments are still exported; removing them needs a `redaction` processor in an OpenTelemetry Collector. + +## Flush before short-lived processes exit + +Scripts, notebooks, workers and serverless handlers need an explicit flush: + +```py +from tracing import provider + +try: + agent.run("Summarize today's tickets", session_id=conversation_id) +finally: + provider.force_flush() # serverless: flush at the end of every invocation + provider.shutdown() # scripts: flush and stop at the end of the process +``` + +## Check that it works + +Run a conversation of two or three turns with one `session_id`, including a tool call, and open **Agent Sessions** filtered by your service name. You should see one session with your `session_id` as its id, framework **Agno**, one turn per `run()`, a transcript, and model calls such as `OpenRouter.invoke` with tokens. A session named `trace:...` means the run had no `session_id`. Cost shows in the sessions list only when your provider returns one, as OpenRouter does. + +## Troubleshooting + +- **Nothing arrives in Maple.** `setup_tracing()` or `AgentOS(tracing=True)` ran before `tracing.py` and the spans went to the AgentOS database. Import `tracing` first. +- **Every user is in one giant session.** The shared `Agent` runs without `session_id=`. Pass it on every `run()`, `arun()` and `continue_run()`. +- **Every turn is its own session.** You pass a new id per request, often a fresh `uuid4()`. Use the stored conversation id. +- **An agent run lands inside the previous team run's trace.** In scripts and workers, run each team run in its own context with `asyncio.create_task(...)` or `contextvars.copy_context().run(...)`. +- **Every model call appears twice.** Remove the second instrumentor, usually `openinference-instrumentation-openai`, OpenLIT or Phoenix's `register(auto_instrument=True)`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) diff --git a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.mdx b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.mdx new file mode 100644 index 0000000000..d34192c833 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.mdx @@ -0,0 +1,250 @@ +--- +title: "Trace Claude Agent SDK agents and Claude Code sessions with OpenTelemetry" +description: "Turn on Claude Code's built-in OpenTelemetry so each Agent SDK conversation or Claude Code session shows up in Maple as one Agent Session." +group: "AI Agents" +order: 14 +navLabel: "Claude Agent SDK & Claude Code" +icon: "claude" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +The Claude Agent SDK and Claude Code export OpenTelemetry spans for each turn, model request and tool call once you set a few environment variables. There is nothing to install. + +Spans only exist with `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`, and in the SDK every `query()` starts a new session unless you resume the conversation's session id. + +Tested with `@anthropic-ai/claude-agent-sdk` 0.3.283 (TypeScript), `claude-agent-sdk` 0.2.160 (Python) and Claude Code 2.1.283. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-claude-agent-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-claude-agent-sdk) skill and follows it. + +```text +Set up Maple agent tracing for the Claude Agent SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-claude-agent-sdk -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Pass the telemetry variables to the CLI + +`OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf` is required, because Claude Code has no default protocol. EU organizations use `https://ingest.eu.maple.dev`. + +The snippets below drop any inherited `TRACEPARENT`, which Claude Code's Bash tool and most CI systems set. Otherwise your agent's turns nest inside that outer trace. + +Never set an exporter to `console` in an SDK app. It breaks the SDK's message stream. + + + + +```bash +npm install @anthropic-ai/claude-agent-sdk zod +``` + +In TypeScript, `options.env` replaces the child's environment, so spread `process.env` to keep `PATH` and `ANTHROPIC_API_KEY`: + +```ts +// maple-env.ts: built per query() so values loaded later (dotenv) are included +let warnedNoKey = false + +export function mapleEnv(): Record { + const env: Record = { ...process.env } + delete env.TRACEPARENT + delete env.TRACESTATE + const key = process.env.MAPLE_INGEST_KEY + if (!key) { + // A missing key turns telemetry off; the agent still runs. + if (!warnedNoKey) console.warn("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled") + warnedNoKey = true + return env + } + return { + ...env, + CLAUDE_CODE_ENABLE_TELEMETRY: "1", + CLAUDE_CODE_ENHANCED_TELEMETRY_BETA: "1", // spans; without it there are none + OTEL_TRACES_EXPORTER: "otlp", + OTEL_LOGS_EXPORTER: "otlp", // optional: cost and replies, under Logs + OTEL_METRICS_EXPORTER: "otlp", // optional: token and cost counters + OTEL_EXPORTER_OTLP_PROTOCOL: "http/protobuf", + OTEL_EXPORTER_OTLP_ENDPOINT: "https://ingest.maple.dev", + OTEL_EXPORTER_OTLP_HEADERS: `Authorization=Bearer ${key}`, + OTEL_SERVICE_NAME: "support-agent", + OTEL_RESOURCE_ATTRIBUTES: "deployment.environment.name=production", + OTEL_TRACES_EXPORT_INTERVAL: "1000", + OTEL_LOGS_EXPORT_INTERVAL: "1000", + // Content, off by default. See "Choose what content to record" below. + OTEL_LOG_USER_PROMPTS: "1", + OTEL_LOG_TOOL_DETAILS: "1", + OTEL_LOG_TOOL_CONTENT: "1", + } +} +``` + + + + +```bash +pip install claude-agent-sdk +``` + +In Python, `ClaudeAgentOptions.env` is merged over the inherited environment, so pass only the telemetry variables and remove `TRACEPARENT` from `os.environ`: + +```py +# maple_env.py +import logging +import os + +os.environ.pop("TRACEPARENT", None) +os.environ.pop("TRACESTATE", None) + +_warned_no_key = False + + +def maple_env() -> dict[str, str]: + """Telemetry env for the Claude Code CLI, built per query().""" + global _warned_no_key + key = os.environ.get("MAPLE_INGEST_KEY") + if not key: + # A missing key turns telemetry off; the agent still runs. + if not _warned_no_key: + logging.getLogger(__name__).warning("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled") + _warned_no_key = True + return {} + return { + "CLAUDE_CODE_ENABLE_TELEMETRY": "1", + "CLAUDE_CODE_ENHANCED_TELEMETRY_BETA": "1", # spans; without it there are none + "OTEL_TRACES_EXPORTER": "otlp", + "OTEL_LOGS_EXPORTER": "otlp", # optional: cost and replies, under Logs + "OTEL_METRICS_EXPORTER": "otlp", # optional: token and cost counters + "OTEL_EXPORTER_OTLP_PROTOCOL": "http/protobuf", + "OTEL_EXPORTER_OTLP_ENDPOINT": "https://ingest.maple.dev", + "OTEL_EXPORTER_OTLP_HEADERS": f"Authorization=Bearer {key}", + "OTEL_SERVICE_NAME": "support-agent", + "OTEL_RESOURCE_ATTRIBUTES": "deployment.environment.name=production", + "OTEL_TRACES_EXPORT_INTERVAL": "1000", + "OTEL_LOGS_EXPORT_INTERVAL": "1000", + # Content, off by default. See "Choose what content to record" below. + "OTEL_LOG_USER_PROMPTS": "1", + "OTEL_LOG_TOOL_DETAILS": "1", + "OTEL_LOG_TOOL_CONTENT": "1", + } +``` + + + + +Pass the env on every `query()` call, as in the next section. Or set the same variables in your Dockerfile or deployment manifest and skip `env`, as long as no `TRACEPARENT` is set there. + +An `env` block in `~/.claude/settings.json` or the project's `.claude/settings.json` overrides `options.env`. Server apps can pass `settingSources: []` (Python `setting_sources=[]`) to skip settings files. + +### Claude Code in your terminal, IDE or desktop app + +Put the variables under `env` in `~/.claude/settings.json`, then start a new `claude` session: + +```json +{ + "env": { + "CLAUDE_CODE_ENABLE_TELEMETRY": "1", + "CLAUDE_CODE_ENHANCED_TELEMETRY_BETA": "1", + "OTEL_TRACES_EXPORTER": "otlp", + "OTEL_LOGS_EXPORTER": "otlp", + "OTEL_METRICS_EXPORTER": "otlp", + "OTEL_EXPORTER_OTLP_PROTOCOL": "http/protobuf", + "OTEL_EXPORTER_OTLP_ENDPOINT": "https://ingest.maple.dev", + "OTEL_EXPORTER_OTLP_HEADERS": "Authorization=Bearer YOUR_INGEST_KEY", + "OTEL_LOG_USER_PROMPTS": "1", + "OTEL_LOG_TOOL_DETAILS": "1", + "OTEL_LOG_TOOL_CONTENT": "1" + } +} +``` + +A repository's `.claude/settings.json` can't set these variables. Use your user settings, your shell or [managed settings](https://code.claude.com/docs/en/managed-settings). Terminal sessions report the service `claude-code`. + +## Resume the session on every turn + +A chat backend that calls `query()` once per message without resuming gets one Maple session per message, and the agent forgets the previous message. + +Store a UUID with each conversation. Pass it as `sessionId` on the first turn and as `resume` on every turn after: + + + + +```ts +import { randomUUID } from "node:crypto" +import { query } from "@anthropic-ai/claude-agent-sdk" +import { mapleEnv } from "./maple-env" + +type Conversation = { claudeSessionId?: string } + +export async function reply(conversation: Conversation, text: string) { + const firstTurn = !conversation.claudeSessionId + const sessionId = conversation.claudeSessionId ?? randomUUID() + conversation.claudeSessionId = sessionId + + for await (const message of query({ + prompt: text, + options: { env: mapleEnv(), ...(firstTurn ? { sessionId } : { resume: sessionId }) }, + })) { + if (message.type === "result") return message.subtype === "success" ? message.result : undefined + } +} +``` + + + + +```py +import uuid +from claude_agent_sdk import ClaudeAgentOptions, ResultMessage, query +from maple_env import maple_env + + +async def reply(conversation: dict, text: str) -> str | None: + first_turn = "claude_session_id" not in conversation + session_id = conversation.setdefault("claude_session_id", str(uuid.uuid4())) + session = {"session_id": session_id} if first_turn else {"resume": session_id} + + async for message in query(prompt=text, options=ClaudeAgentOptions(env=maple_env(), **session)): + if isinstance(message, ResultMessage): + return message.result + return None +``` + + + + +`resume` reads the earlier turns from `~/.claude/projects/` on the same machine. If messages can land on different hosts, use the SDK's [`sessionStore`](https://code.claude.com/docs/en/agent-sdk/session-storage) option. A Python `ClaudeSDKClient`, or a TypeScript `query()` fed an async iterable, keeps one session for all its turns and needs none of this. + +## Choose what content to record + +Claude Code redacts content by default. `OTEL_LOG_USER_PROMPTS=1` records prompts, which title each turn. `OTEL_LOG_TOOL_DETAILS=1` records Bash commands, file paths and tool error messages. `OTEL_LOG_TOOL_CONTENT=1` records tool results, including any secrets in files Claude reads or in command output. Turn on only what your Maple organization is allowed to store. + +## Let each turn finish exporting + +Let every `query()` loop reach its `result` message. Breaking out early, calling `close()` or aborting kills the CLI before it exports the turn. In a script, keep the process alive about 5 seconds after the last `query()`. On serverless platforms, finish the loop before returning the response. + +## Check that it works + +Run a two-turn conversation through `reply()` with at least one tool call, then open **Agent Sessions**. You should see one session with the framework **Claude Agent SDK**, one turn per message titled with the prompt, model calls with their tokens, and the tool calls by name. + +The transcript has no assistant replies and cost shows as unpriced. Both are on log events under **Logs** when the logs exporter is on. + +## Troubleshooting + +- **Metrics and logs arrive, but no sessions.** Set `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` and `OTEL_TRACES_EXPORTER=otlp`. +- **Nothing arrives at all.** `OTEL_EXPORTER_OTLP_PROTOCOL` is unset or `grpc`. Set `http/protobuf`. `CLAUDE_CODE_OTEL_DIAG_STDERR=1` prints export errors to the SDK's `stderr` callback. +- **Every message is its own session.** Resume the conversation's session id as shown above, and don't set `forkSession`. +- **Your agent's turns appear inside another trace.** The process inherited a `TRACEPARENT`. Drop it and `TRACESTATE` from the environment. +- **The CLI ignores your values.** They are in a repository's `.claude/settings.json`, or a settings file overrides `options.env`. Use `~/.claude/settings.json`, or pass `settingSources: []`. +- **Extra turns titled ``.** Claude Code ran sub-agents in the background. Add `CLAUDE_CODE_DISABLE_BACKGROUND_TASKS: "1"` to the env. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Provider SDKs](/docs/agent-tracing/provider-sdks): tracing direct calls with the Anthropic SDK instead of the Agent SDK. +- Claude Code [Monitoring reference](https://code.claude.com/docs/en/monitoring-usage): every variable, span attribute and event. diff --git a/apps/landing/src/content/docs/agent-tracing/cloudflare-agents.md b/apps/landing/src/content/docs/agent-tracing/cloudflare-agents.md new file mode 100644 index 0000000000..307e69c1dd --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/cloudflare-agents.md @@ -0,0 +1,181 @@ +--- +title: "Trace Cloudflare Agents with OpenTelemetry" +description: "Export the AI SDK spans of a Cloudflare Agents SDK agent from its Durable Object to Maple and group each chat into one Agent Session." +group: "AI Agents" +order: 15 +navLabel: "Cloudflare Agents" +icon: "cloudflare" +--- + +Agents built with the Cloudflare Agents SDK (`AIChatAgent` or `Agent`) usually call models through the Vercel AI SDK, which emits the spans Maple reads. This guide sends those spans from the agent's Durable Object to Maple and passes the agent's instance name as the conversation id. + +The Workers runtime can't run the Node.js OpenTelemetry SDK, so you create a small tracer provider that exports over `fetch` and flush it at the end of every turn. It works on the Workers Free and Paid plans. + +You need `ai` 7.0.106 or newer and the `nodejs_compat` compatibility flag, which Agents SDK projects already have. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-cloudflare-agents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-cloudflare-agents) skill and follows it. + +```text +Set up Maple agent tracing for the Cloudflare Agents SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-cloudflare-agents -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the packages + +```bash +npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/api @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-http @opentelemetry/resources @opentelemetry/context-async-hooks +``` + +What each package does: + +- `ai` and `@ai-sdk/otel`: record each turn, model call and tool call. +- `@opentelemetry/sdk-trace-base`: collects those records and sends them in batches. +- `@opentelemetry/exporter-trace-otlp-http`: delivers them to Maple. +- `@opentelemetry/resources`: puts your service name on them. +- `@opentelemetry/api` and `@opentelemetry/context-async-hooks`: keep agents that a tool calls in the same conversation. + +## Store the ingest key + +Save the key as a secret so it isn't in your Wrangler config: + +```bash +npx wrangler secret put MAPLE_INGEST_KEY +``` + +For `wrangler dev`, add `MAPLE_INGEST_KEY=YOUR_INGEST_KEY` to `.dev.vars`. + +## Create the tracer provider + +Create a `telemetry.ts`. The agent code below imports `tracerProvider` from it, which also registers the AI SDK integration: + +```ts +// telemetry.ts +import { OpenTelemetry } from "@ai-sdk/otel" +import { context } from "@opentelemetry/api" +import { AsyncLocalStorageContextManager } from "@opentelemetry/context-async-hooks" +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-http" +import { resourceFromAttributes } from "@opentelemetry/resources" +import { BasicTracerProvider, BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { registerTelemetry } from "ai" +import { env } from "cloudflare:workers" + +context.setGlobalContextManager(new AsyncLocalStorageContextManager().enable()) + +// A missing key disables export; it never stops the Worker. +if (!env.MAPLE_INGEST_KEY) console.warn("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled") + +export const tracerProvider = new BasicTracerProvider({ + resource: resourceFromAttributes({ + "service.name": "support-agent", + "deployment.environment.name": "production", + }), + spanProcessors: env.MAPLE_INGEST_KEY + ? [ + new BatchSpanProcessor( + new OTLPTraceExporter({ + url: "https://ingest.maple.dev/v1/traces", // EU: https://ingest.eu.maple.dev/v1/traces + headers: { authorization: `Bearer ${env.MAPLE_INGEST_KEY}` }, + }), + ), + ] + : [], +}) + +registerTelemetry( + new OpenTelemetry({ + tracer: tracerProvider.getTracer("gen_ai"), + usage: true, + runtimeContext: true, + }), +) +``` + +The context manager keeps sub-agents called from a tool inside the same trace. + +## Pass the conversation id and flush each turn + +Each chat is one instance of your agent, so its instance name (`this.name`) is the conversation id. Pass it on every AI SDK call, and flush the tracer provider when the turn ends. A Durable Object can be evicted between messages, and spans still in the buffer are lost with it. + +In an `AIChatAgent`, flush in `onChatResponse`, which runs after every turn: + +```ts +// server.ts +import { AIChatAgent } from "@cloudflare/ai-chat" +import { convertToModelMessages, streamText } from "ai" +import { tracerProvider } from "./telemetry" + +export class ChatAgent extends AIChatAgent { + async onChatMessage() { + const result = streamText({ + model, + messages: await convertToModelMessages(this.messages), + tools, + runtimeContext: { conversationId: this.name }, + telemetry: { functionId: "support_agent", includeRuntimeContext: { conversationId: true } }, + }) + return result.toUIMessageStreamResponse() + } + + async onChatResponse() { + await tracerProvider.forceFlush() + } +} +``` + +In a plain `Agent`, flush when the method that called the model returns: + +```ts +import { Agent } from "agents" +import { generateText } from "ai" +import { tracerProvider } from "./telemetry" + +export class TaskAgent extends Agent { + async onRequest(request: Request) { + const { prompt } = await request.json<{ prompt: string }>() + try { + const result = await generateText({ + model, + prompt, + tools, + runtimeContext: { conversationId: this.name }, + telemetry: { functionId: "task_agent", includeRuntimeContext: { conversationId: true } }, + }) + return Response.json({ text: result.text }) + } finally { + this.ctx.waitUntil(tracerProvider.forceFlush()) + } + } +} +``` + +If the method returns a `streamText` stream instead, flush once the stream has been read, with `this.ctx.waitUntil(result.consumeStream().then(() => tracerProvider.forceFlush()))` before returning the response. + +Give each chat its own instance: on the client, pass a chat id as `name` to `useAgent({ agent: "ChatAgent", name: chatId })`. Without a `name`, every client connects to the `default` instance and all chats land in one session. `functionId` becomes the agent name in Maple, so give each agent a distinct one. + +To keep a call's prompts and replies out of Maple, set `recordInputs: false` and `recordOutputs: false` in its `telemetry`. + +## Check that it works + +Run `wrangler dev`, send two messages in one chat that trigger a tool call, then open **Agent Sessions** in Maple. You should see one session named after the agent instance with framework **Vercel AI SDK**, one turn per message, and a transcript with the prompts, replies and tool calls. A second chat should show up as a second session. + +## Troubleshooting + +- **No AI spans at all.** Nothing imports `telemetry.ts`, or the `MAPLE_INGEST_KEY` secret isn't set (`npx wrangler secret list`). +- **Turns are missing or arrive late.** The flush isn't running. Check that `onChatResponse` (or your `finally` block) calls `forceFlush()`. +- **All chats are one session.** The client connects without a `name`, so every chat uses the `default` instance. +- **A sub-agent shows up as its own session.** The context manager isn't registered. Keep the `setGlobalContextManager` line in `telemetry.ts`. +- **Every span shows up twice.** `registerTelemetry()` ran twice, for example from two entry points that both set it up. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Vercel AI SDK guide](/docs/agent-tracing/vercel-ai-sdk) +- [Cloudflare Workers instrumentation](/docs/guides/instrumentation-cloudflare-workers) for request and binding spans +- [Cloudflare Agents SDK](https://developers.cloudflare.com/agents/) diff --git a/apps/landing/src/content/docs/agent-tracing/crewai.md b/apps/landing/src/content/docs/agent-tracing/crewai.md new file mode 100644 index 0000000000..344b7533dc --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/crewai.md @@ -0,0 +1,185 @@ +--- +title: "Trace CrewAI crews and flows with OpenTelemetry" +description: "Send CrewAI crews and flows to Maple with OpenInference, one Agent Session per conversation." +group: "AI Agents" +order: 21 +navLabel: "CrewAI" +icon: "crewai" +--- + +CrewAI needs two OpenInference instrumentors: `openinference-instrumentation-crewai` for crews, agents and tools, and one for your model provider, which records prompts and tokens. You also wrap every `kickoff()` in a conversation id so a chat becomes one session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-crewai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-crewai) skill and follows it. + +```text +Set up Maple agent tracing for CrewAI in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-crewai -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the instrumentors + +```bash +pip install "crewai>=1.15" "openinference-instrumentation-crewai>=1.1.18" \ + "openinference-instrumentation-openai>=0.1.61" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Pick the model instrumentor by the model string you pass to `LLM(...)`: + +| Model string | Instrumentor | +| --- | --- | +| `openai/…`, `openrouter/…`, `deepseek/…`, `ollama/…`, `custom_openai=True`, or a bare name like `gpt-4.1-mini` | `openinference-instrumentation-openai` | +| `anthropic/…` or a bare `claude-…` | `openinference-instrumentation-anthropic` | +| `gemini/…` or a bare `gemini-…` | `openinference-instrumentation-google-genai` | +| `bedrock/…` | `openinference-instrumentation-bedrock` | +| Anything else (needs `crewai[litellm]`) | `openinference-instrumentation-litellm` | + +Install only the ones your crews use. The LiteLLM instrumentor records nothing for the first four rows. + +## Point the exporter at Maple + +```bash +export OTEL_SERVICE_NAME=support-crew +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export CREWAI_DISABLE_TELEMETRY=true +export CREWAI_TRACING_ENABLED=false +``` + +EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, it has to end in `/v1/traces`. + +The last two turn off CrewAI's own telemetry and a first-run prompt that blocks the process at exit. Don't use `OTEL_SDK_DISABLED=true` instead, because it disables your Maple traces too. + +## Initialize tracing + +Add a `tracing.py` and import it at the top of your entry point, before the first `kickoff()`: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.crewai import CrewAIInstrumentor +from openinference.instrumentation.openai import OpenAIInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class CrewAIAgentNames(SpanProcessor): + def on_start(self, span, parent_context=None): + parent = trace.get_current_span(parent_context) + attrs = getattr(parent, "attributes", None) or {} + role = attrs.get("graph.node.id") + if role and "gen_ai.agent.name" not in attrs and parent.is_recording(): + parent.set_attribute("gen_ai.agent.name", role) + + +provider = TracerProvider() +provider.add_span_processor(CrewAIAgentNames()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +config = TraceConfig(enable_genai_semconv=True) +CrewAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) +OpenAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) +``` + +Pass `config` to every instrumentor. Keep `skip_dep_check=True`, or an instrumentor can silently skip itself. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `CrewAIAgentNames()` and the exporter to it and pass it to `instrument()`. + +## Group a conversation into one session + +CrewAI has no conversation id, so wrap every `kickoff()` in OpenInference's `using_session` with the id your app stores the chat under: + +```py +import tracing # noqa: F401 (first import) + +from crewai import LLM, Agent, Crew, Task +from openinference.instrumentation import using_session + +llm = LLM(model="openai/gpt-4o-mini", temperature=0) + + +def build_crew(text: str, history: str, stream: bool = False) -> Crew: + assistant = Agent( + role="assistant", + goal="Answer the user's questions", + backstory="You are a concise, helpful assistant.", + llm=llm, + tools=[get_weather, calculate], + ) + task = Task( + description=f"{text}\n\nConversation so far:\n{history}", + expected_output="A short, direct reply to the user.", + agent=assistant, + name="reply", + ) + return Crew(name="support", agents=[assistant], tasks=[task], stream=stream) + + +def handle_message(conversation_id: str, text: str, history: str) -> str: + with using_session(conversation_id): + return build_crew(text, history).kickoff().raw +``` + +Put the user's message first in the task description, because Maple labels each turn with its first line. Give the crew a `name=`. `crew_id` and `crew_key` don't work as conversation ids. + +For conversational flows, wrap `flow.handle_turn(text, session_id=conversation_id)` in the same `using_session` and set `name = "support_flow"` on the flow class. + +Use `kickoff()` or `await crew.kickoff_async()`, never `akickoff()`, which isn't traced as one run. + +## Streaming crews + +`Crew(stream=True)` adds an empty turn to every message. Wrap the streamed turn in one span of your own: + +```py +from opentelemetry import trace + +tracer = trace.get_tracer("chat") + + +def stream_message(conversation_id: str, text: str, history: str, send) -> None: + with using_session(conversation_id), tracer.start_as_current_span( + "invoke_agent support", + attributes={ + "gen_ai.operation.name": "invoke_agent", + "gen_ai.conversation.id": conversation_id, + }, + ): + for chunk in build_crew(text, history, stream=True).kickoff(): + send(chunk.content) +``` + +`LLM(stream=True)` without `Crew(stream=True)` doesn't need this. + +## Flush in short-lived processes + +Servers and `crewai run` need nothing. In serverless handlers and notebooks, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. + +## Check that it works + +Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session labeled **CrewAI**, with one turn per `kickoff()` starting at `support.kickoff`, `ChatCompletion` model calls with tokens, tool calls like `get_weather.run`, and one lane per agent role. + +Cost shows as **unpriced** unless your models go through LiteLLM, which is expected. + +## Troubleshooting + +- **Agent and tool spans, but no model calls or tokens.** The instrumentor for your model's SDK is missing. Match it to the model string with the table above. +- **No spans at all.** Import `tracing` before the first kickoff, keep `skip_dep_check=True`, make sure `OTEL_SDK_DISABLED` isn't set, and check the logs for exporter errors. +- **One session per message.** The kickoff isn't inside `using_session(...)`, or the id changes per request. +- **Every model and tool call is its own trace.** Replace `akickoff()` with `kickoff()` or `kickoff_async()`. +- **The process hangs at exit asking about traces.** Set `CREWAI_TRACING_ENABLED=false`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/dspy.md b/apps/landing/src/content/docs/agent-tracing/dspy.md new file mode 100644 index 0000000000..18735eaf97 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/dspy.md @@ -0,0 +1,224 @@ +--- +title: "Trace DSPy programs and ReAct agents with OpenTelemetry" +description: "Send DSPy programs to Maple as one Agent Session per conversation, with transcript, model and tool calls, tokens and cost." +group: "AI Agents" +order: 27 +navLabel: "DSPy" +icon: "python" +--- + +This guide traces DSPy with OpenInference's DSPy instrumentor, adds tokens and tool names with a small DSPy callback, and groups each conversation into one session with `using_session`. + +Tested with DSPy 3.4 and `openinference-instrumentation-dspy` 0.1.45 on Python 3.10 or later. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-dspy](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-dspy) skill and follows it. + +```text +Set up Maple agent tracing for DSPy in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-dspy -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install and point the exporter at Maple + +```bash +pip install "dspy>=3.4" "openinference-instrumentation-dspy>=0.1.45" "openinference-instrumentation>=0.1.66" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ + "opentelemetry-instrumentation-threading>=0.66b0" +``` + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. + +## Initialize tracing + +Add a `tracing.py` and import it at the top of your entry point, before your DSPy modules are defined: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.dspy import DSPyInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.instrumentation.threading import ThreadingInstrumentor +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +DSPyInstrumentor().instrument(tracer_provider=provider, config=TraceConfig(enable_genai_semconv=True)) +ThreadingInstrumentor().instrument() +``` + +`ThreadingInstrumentor` keeps `dspy.Parallel` and thread pool workers in the caller's session. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add the OTLP exporter to it and pass it to `instrument()` instead of creating a second one. + +## Add the Maple callback + +The callback adds tokens, cost, tool names and agent spans. Save it as `maple_dspy.py`: + +```py +# maple_dspy.py +import json + +import dspy +from dspy.utils.callback import BaseCallback +from openinference.instrumentation import TraceConfig +from opentelemetry import trace + +_config = TraceConfig() + + +def _message(role, values): + text = "\n".join(v for v in values if isinstance(v, str)) + return json.dumps([{"role": role, "parts": [{"type": "text", "content": text}]}]) if text else None + + +class MapleCallback(BaseCallback): + def __init__(self): + self._agents = set() + self._lms = {} + + def on_module_start(self, call_id, instance, inputs): + if type(instance).__module__.startswith("dspy."): + return # Predict, ChainOfThought, ReAct: building blocks, not agents + self._agents.add(call_id) + span = trace.get_current_span() + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", type(instance).__name__) + user = _message("user", [*inputs.get("args", ()), *inputs.get("kwargs", {}).values()]) + if user and not _config.hide_inputs: + span.set_attribute("gen_ai.input.messages", user) + + def on_module_end(self, call_id, outputs, exception): + if call_id not in self._agents: + return + self._agents.discard(call_id) + if isinstance(outputs, dspy.Prediction) and not _config.hide_outputs: + reply = _message("assistant", [v for k, v in outputs.items() if k != "reasoning"]) + if reply: + trace.get_current_span().set_attribute("gen_ai.output.messages", reply) + + def on_tool_start(self, call_id, instance, inputs): + span = trace.get_current_span() + span.set_attribute("gen_ai.tool.name", instance.name) + span.set_attribute("gen_ai.tool.description", instance.desc or "") + if not _config.hide_inputs: + span.set_attribute("gen_ai.tool.call.arguments", json.dumps(inputs.get("kwargs", {}), default=str)) + + def on_lm_start(self, call_id, instance, inputs): + self._lms[call_id] = instance + + def on_lm_end(self, call_id, outputs, exception): + lm = self._lms.pop(call_id, None) + entry = next((e for e in reversed(lm.history[-16:]) if e["outputs"] is outputs), None) if lm else None + if entry is None or getattr(entry["response"], "cache_hit", False): + return # history is off, or a cache hit that cost nothing + usage = entry["usage"] or {} + attributes = { + "gen_ai.response.id": getattr(entry["response"], "id", None), + "gen_ai.response.model": entry.get("response_model"), + "gen_ai.usage.input_tokens": usage.get("prompt_tokens"), + "gen_ai.usage.output_tokens": usage.get("completion_tokens"), + "gen_ai.usage.cache_read.input_tokens": (usage.get("prompt_tokens_details") or {}).get("cached_tokens"), + "gen_ai.usage.reasoning.output_tokens": (usage.get("completion_tokens_details") or {}).get("reasoning_tokens"), + "gen_ai.usage.cost": entry.get("cost"), + } + trace.get_current_span().set_attributes({k: v for k, v in attributes.items() if v is not None}) +``` + +Register it with the rest of your DSPy configuration: + +```py +import tracing # must come first + +import dspy +from maple_dspy import MapleCallback + +dspy.configure(lm=dspy.LM("openai/gpt-4o-mini", temperature=0), callbacks=[MapleCallback()]) +``` + +Every `dspy.Module` subclass you write shows up as an agent named after its class, so give worker modules descriptive names like `WeatherWorker`. + +## Group a conversation into one session + +Each call to your module is its own trace. Wrap every call in `using_session` with the conversation id to group them into one session: + +```py +import dspy +from openinference.instrumentation import using_session + + +class ChatAssistant(dspy.Module): + def __init__(self): + super().__init__() + self.respond = dspy.ReAct( + "question: str, conversation: dspy.History -> answer: str", + tools=[get_weather, calculate], + ) + + def forward(self, question, conversation): + return self.respond(question=question, conversation=conversation) + + +assistant = ChatAssistant() + + +def handle_message(conversation_id: str, question: str, turns: list[dict]) -> str: + with using_session(conversation_id): + answer = assistant(question=question, conversation=dspy.History(messages=turns)).answer + turns.append({"question": question, "answer": answer}) + return answer +``` + +Use the id your app stores the chat under. A new UUID per request gives one session per message, and a constant puts every user in one session. + +If you stream with `dspy.streamify`, call it after `dspy.configure(callbacks=[MapleCallback()])`, and put `using_session` around the loop that reads the stream. + +To keep prompts and outputs out of your traces, set `OPENINFERENCE_HIDE_INPUTS=true` and `OPENINFERENCE_HIDE_OUTPUTS=true` before `tracing.py` runs. The callback follows both. + +## Flush before short-lived processes exit + +Serverless handlers, notebooks and processes that get killed need an explicit flush: + +```py +from tracing import provider + +try: + handle_message("conv-42", "What's the weather in Berlin?", turns=[]) +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +## Check that it works + +Set `cache=False` on the LM, then run two or three messages through `handle_message` with one conversation id, including one tool call. In **Agent Sessions** you should see one session with framework **DSPy**, one turn per call, and `LM.__call__` model calls with tokens. `finish.__call__` is `dspy.ReAct`'s built-in end-of-loop tool. + +## Troubleshooting + +- **No spans, or exports fail with 404.** Import `tracing` first. An `endpoint=` passed in code must end in `/v1/traces`; the env variable takes the base URL. +- **No tokens anywhere.** `MapleCallback` isn't registered, a later `dspy.configure(callbacks=[...])` replaced it, or the calls were cache hits. +- **One session per message.** The call runs outside `using_session(...)`, or the id changes per request. +- **Every model call counted twice.** Remove `openinference-instrumentation-litellm` or `-openai`; the callback already records model calls. +- **`TypeError: object of type 'History' has no len()`.** A module attribute is named `history`. Rename it, or pass the `dspy.History` as an input field. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/genkit.md b/apps/landing/src/content/docs/agent-tracing/genkit.md new file mode 100644 index 0000000000..8dd6a09464 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/genkit.md @@ -0,0 +1,229 @@ +--- +title: "Trace Genkit agents with OpenTelemetry" +description: "Send Genkit's OpenTelemetry spans to Maple and group each chat into one Agent Session." +group: "AI Agents" +order: 16 +navLabel: "Genkit" +icon: "googleadk" +--- + +Genkit traces every flow, model call and tool call with OpenTelemetry, but it records them under its own `genkit:*` attributes, which Agent Sessions doesn't read. You export those spans to Maple and add a small span processor that copies them to the GenAI attributes Maple reads. You also pass a conversation id in each flow, or each message becomes its own session. + +This guide covers Genkit for Node.js. You need `genkit` 1.22 or newer and Node.js 20 or newer. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-genkit](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-genkit) skill and follows it. + +```text +Set up Maple agent tracing for Genkit in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-genkit -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the packages + +```bash +npm install genkit @opentelemetry/sdk-node @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto +``` + +## Point the exporter at Maple + +```bash +export OTEL_SERVICE_NAME="support-agent" +export OTEL_RESOURCE_ATTRIBUTES="deployment.environment.name=production" +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +## Add the span processor + +Create `genkit-for-maple.ts` and copy it as is. It turns each flow into an `invoke_agent` span, each model call into a `chat` span with its messages and token counts, and each tool call into an `execute_tool` span: + +```ts +// genkit-for-maple.ts +import type { ReadableSpan, SpanProcessor } from "@opentelemetry/sdk-trace-base" + +type Part = { + text?: string + reasoning?: string + toolRequest?: { name: string; ref?: string; input?: unknown } + toolResponse?: { name: string; ref?: string; output?: unknown } +} +type Message = { role: string; content: Part[] } + +// Genkit message parts to OpenTelemetry GenAI parts. Media parts are left out. +function toPart({ text, reasoning, toolRequest, toolResponse }: Part) { + if (text !== undefined) return { type: "text", content: text } + if (reasoning !== undefined) return { type: "reasoning", content: reasoning } + if (toolRequest) { + return { type: "tool_call", id: toolRequest.ref, name: toolRequest.name, arguments: toolRequest.input } + } + if (toolResponse) return { type: "tool_call_response", id: toolResponse.ref, response: toolResponse.output } + return undefined +} + +const toParts = (content: Part[]) => content.map(toPart).filter((part) => part !== undefined) + +function toMessages(messages: Message[]) { + return messages.map((m) => ({ role: m.role === "model" ? "assistant" : m.role, parts: toParts(m.content) })) +} + +/** Adds the gen_ai.* attributes Maple reads to Genkit's flow, model and tool spans. */ +export class GenkitForMaple implements SpanProcessor { + onStart() {} + + onEnd(span: ReadableSpan) { + const attrs = span.attributes + const json = (key: string) => { + const value = attrs[key] + return typeof value === "string" ? JSON.parse(value) : undefined + } + const name = String(attrs["genkit:name"]) + + switch (attrs["genkit:metadata:subtype"]) { + case "flow": + case "agent": { + Object.assign(attrs, { + "gen_ai.operation.name": "invoke_agent", + "gen_ai.agent.name": name, + }) + // Set in your flow, or by Genkit for defineAgent() chats + const conversationId = attrs["genkit:metadata:conversationId"] ?? attrs["genkit:metadata:agent:sessionId"] + if (conversationId !== undefined) attrs["gen_ai.conversation.id"] = conversationId + break + } + case "model": { + const input = json("genkit:input") + const output = json("genkit:output") + const [provider, ...model] = name.split("/") + const messages: Message[] = input?.messages ?? [] + const system = messages.filter((m) => m.role === "system").flatMap((m) => toParts(m.content)) + Object.assign(attrs, { + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": provider, + "gen_ai.request.model": model.join("/") || name, + "gen_ai.input.messages": JSON.stringify(toMessages(messages.filter((m) => m.role !== "system"))), + }) + if (system.length > 0) attrs["gen_ai.system_instructions"] = JSON.stringify(system) + if (output?.message) { + attrs["gen_ai.output.messages"] = JSON.stringify( + toMessages([output.message]).map((m) => ({ ...m, finish_reason: output.finishReason })), + ) + } + if (output?.finishReason) attrs["gen_ai.response.finish_reasons"] = [output.finishReason] + if (output?.usage?.inputTokens !== undefined) attrs["gen_ai.usage.input_tokens"] = output.usage.inputTokens + if (output?.usage?.outputTokens !== undefined) attrs["gen_ai.usage.output_tokens"] = output.usage.outputTokens + break + } + case "tool": { + Object.assign(attrs, { + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.call.arguments": attrs["genkit:input"] ?? "{}", + }) + const result = json("genkit:output") + if (result !== undefined) { + attrs["gen_ai.tool.call.result"] = typeof result === "string" ? result : JSON.stringify(result) + } + break + } + } + } + + forceFlush() { + return Promise.resolve() + } + + shutdown() { + return Promise.resolve() + } +} +``` + +## Start OpenTelemetry + +Create an `instrumentation.ts` and import it as the first line of your entry point (`import "./instrumentation"`): + +```ts +// instrumentation.ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { NodeSDK } from "@opentelemetry/sdk-node" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { disableGenkitOTelInitialization } from "genkit/tracing" +import { GenkitForMaple } from "./genkit-for-maple" + +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) + +// Reads OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES and OTEL_EXPORTER_OTLP_* +export const sdk = new NodeSDK({ spanProcessors: [new GenkitForMaple(), spanProcessor] }) + +// `genkit start` sets GENKIT_ENV=dev: leave the Developer UI's own tracing alone there +if (process.env.GENKIT_ENV !== "dev") { + disableGenkitOTelInitialization() + sdk.start() +} +``` + +Keep `GenkitForMaple` before the exporting processor. `disableGenkitOTelInitialization()` stops Genkit from starting its own OpenTelemetry SDK, which can't export to Maple. It also turns off `enableFirebaseTelemetry()` and `enableGoogleCloudTelemetry()`, so traces stop going to Google Cloud. + +Runs under `genkit start` keep their traces in the Developer UI and send nothing to Maple. To trace them in Maple too, remove the `if`, and the Developer UI shows no traces. + +If your app already starts OpenTelemetry (Sentry, auto-instrumentation, your own `NodeTracerProvider`), add both processors to that provider instead of creating a `NodeSDK`, and keep the `disableGenkitOTelInitialization()` call. + +## Pass the conversation id in each flow + +Run each user message through a flow and call `setCustomMetadataAttribute("conversationId", ...)` at its start, with the chat or thread id your app already stores: + +```ts +import { genkit, z, type MessageData } from "genkit" +import { setCustomMetadataAttribute } from "genkit/tracing" + +const ai = genkit({ plugins: [/* your model plugin */], model: "googleai/gemini-2.5-flash" }) + +// One history per conversation. Store it in your database in a real backend. +const histories = new Map() + +export const supportChat = ai.defineFlow( + { name: "supportChat", inputSchema: z.object({ chatId: z.string(), text: z.string() }), outputSchema: z.string() }, + async ({ chatId, text }) => { + setCustomMetadataAttribute("conversationId", chatId) + + const response = await ai.generate({ messages: histories.get(chatId) ?? [], prompt: text, tools: [getWeather] }) + histories.set(chatId, response.messages) + return response.text + }, +) +``` + +The id must stay the same for the whole conversation and differ between conversations. The flow name becomes the agent name in Maple, so give each agent its own flow. + +Chats with an agent from `ai.defineAgent()` (in `genkit/beta`) already carry Genkit's session id, so they need no extra call. + +## Flush before the process exits + +`BatchSpanProcessor` exports every few seconds, so a short-lived process can exit before its last spans are sent. In a script, call `await sdk.shutdown()` in a `finally` block before exiting. In a serverless handler, call `await spanProcessor.forceFlush()` after the flow returns. In a long-running server, call `sdk.shutdown()` on `SIGTERM`. + +## Check that it works + +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session named after your conversation id, one turn per flow run, and a transcript with the prompts, replies and tool calls. Each model call shows its token counts. + +The framework shows as **Genkit**. Cost shows as unpriced because Genkit doesn't report it. + +## Troubleshooting + +- **No spans at all.** `instrumentation.ts` isn't the first import, or the app runs under `genkit start`. +- **Traces show up, but Agent Sessions is empty.** `GenkitForMaple` is missing from `spanProcessors`. +- **Every message is its own session.** The flow doesn't call `setCustomMetadataAttribute("conversationId", ...)`, or `ai.generate()` runs outside a flow. +- **The Developer UI shows no traces.** `disableGenkitOTelInitialization()` ran under `genkit start`. Keep the `GENKIT_ENV` check around it. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Genkit observability](https://genkit.dev/docs/observability/getting-started/) diff --git a/apps/landing/src/content/docs/agent-tracing/google-adk.mdx b/apps/landing/src/content/docs/agent-tracing/google-adk.mdx new file mode 100644 index 0000000000..f78dc374b7 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/google-adk.mdx @@ -0,0 +1,422 @@ +--- +title: "Trace Google ADK agents with OpenTelemetry" +description: "Send Google Agent Development Kit (ADK) traces to Maple with the transcript, tool calls and tokens, one session per ADK session." +group: "AI Agents" +order: 22 +navLabel: "Google ADK" +icon: "googleadk" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +Google's Agent Development Kit (ADK) emits OpenTelemetry spans on its own, in Python (`google-adk`) and TypeScript (`@google/adk`), so you need no instrumentation package. You register a tracer provider and add a few lines that put the transcript and tool calls in the format Maple reads. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-google-adk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-google-adk) skill and follows it. + +```text +Set up Maple agent tracing for Google ADK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-google-adk -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install ADK and the OTLP exporter + + + + +```bash +pip install "google-adk>=2.10" litellm opentelemetry-exporter-otlp-proto-http +``` + +`litellm` is only needed for non-Gemini models. Don't pin a newer OpenTelemetry version: ADK 2.10 caps `opentelemetry-sdk` at 1.42.1. + + + + +```bash +npm install @google/adk@^2.1 zod @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-proto +``` + +`zod` is for the tool in the example below. + + + + +## Configure the export + + + + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +# Put prompts, replies and tool calls on span attributes, in the format Maple reads +export OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental +export OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY +# Drop ADK's own copies of the same content, which Maple doesn't read +export ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false +``` + +EU organizations use `https://ingest.eu.maple.dev`. Use `SPAN_ONLY` exactly, because `true` leaves the transcript empty. + +These settings store every prompt and tool result in Maple. To keep structure and tokens without content, leave `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` unset and delete the two `gen_ai.tool.call.*` lines from the plugin below. + + + + +```bash +export OTEL_SERVICE_NAME="support-agent" +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +These settings store every prompt and tool result in Maple. To keep structure and tokens without content, set `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false`. + + + + +## Register a tracer provider + + + + +A `Runner` in your own app, worker or script exports nothing until you register a tracer provider. Add a `telemetry.py`: + +```py +# telemetry.py +import json + +from google.adk.plugins.base_plugin import BasePlugin +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class SkipDuplicateToolSpans(BatchSpanProcessor): + """Drops two ADK tool spans that would count a call twice: the + `execute_tool (merged)` summary of parallel calls, and the span of a call + paused for confirmation (it runs again, in its own span, once approved).""" + + def on_end(self, span): + if span.name != "execute_tool (merged)" and not span.attributes.get("adk.awaiting_confirmation"): + super().on_end(span) + + +class ToolCallAttributes(BasePlugin): + """Records each tool call's arguments and result on its `execute_tool` span, + and marks the span of a call that is waiting for confirmation.""" + + def __init__(self): + super().__init__(name="tool_call_attributes") + + async def before_tool_callback(self, *, tool, tool_args, tool_context): + trace.get_current_span().set_attribute("gen_ai.tool.call.arguments", json.dumps(tool_args, default=str)) + + async def after_tool_callback(self, *, tool, tool_args, tool_context, result): + span = trace.get_current_span() + if tool_context.actions.requested_tool_confirmations: + span.set_attribute("adk.awaiting_confirmation", True) + span.set_attribute("gen_ai.tool.call.result", json.dumps(result, default=str)) + + +# Reads OTEL_SERVICE_NAME, OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider = TracerProvider(resource=Resource.create()) +provider.add_span_processor(SkipDuplicateToolSpans(OTLPSpanExporter())) +trace.set_tracer_provider(provider) +``` + +Import `telemetry` as the first line of your entry point and register the plugin on the runner: + +```py +# main.py +import telemetry # first, so the provider exists before ADK runs + +from google.adk.agents import LlmAgent +from google.adk.models.lite_llm import LiteLlm +from google.adk.runners import Runner +from google.adk.sessions import InMemorySessionService +from google.genai import types + + +def get_weather(city: str) -> dict: + """Get the current weather for a city.""" + return {"city": city, "temperature_c": 21, "condition": "partly cloudy"} + + +agent = LlmAgent( + name="assistant", + model=LiteLlm(model="openrouter/openai/gpt-4o-mini"), + instruction="You are a concise assistant.", + tools=[get_weather], +) + +runner = Runner( + app_name="support", + agent=agent, + session_service=InMemorySessionService(), + plugins=[telemetry.ToolCallAttributes()], + auto_create_session=True, +) +``` + +If your app already sets up OpenTelemetry (Logfire, Sentry, a platform agent), add the `SkipDuplicateToolSpans(OTLPSpanExporter())` processor to that provider instead. + +`adk web` and `adk api_server` create the provider themselves. There, skip it and register the plugin on your `App`: `App(name="support", root_agent=agent, plugins=[ToolCallAttributes()])`. + + + + +ADK records each model request, reply and tool call on its spans in its own format. The span processor below rewrites them into the attributes Maple reads. Create an `instrumentation.ts` and import it as the first line of your entry point (`import "./instrumentation"`): + +```ts +// instrumentation.ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { NodeSDK, tracing } from "@opentelemetry/sdk-node" + +type Part = { + text?: string + thought?: boolean + functionCall?: { id?: string; name?: string; args?: unknown } + functionResponse?: { id?: string; response?: unknown } +} +type Content = { role?: string; parts?: Part[] } + +const parse = (value: unknown) => (typeof value === "string" ? JSON.parse(value) : {}) + +// Gemini content -> the GenAI message format Maple reads +const toParts = (parts: Part[] = []) => + parts.flatMap((part): object[] => { + if (part.functionCall) { + const { id, name, args } = part.functionCall + return [{ type: "tool_call", id, name, arguments: args }] + } + if (part.functionResponse) { + const { id, response } = part.functionResponse + return [{ type: "tool_call_response", id, response }] + } + if (typeof part.text !== "string") return [] + return [{ type: part.thought ? "reasoning" : "text", content: part.text }] + }) + +const toMessage = ({ role, parts }: Content) => ({ + role: parts?.some((part) => part.functionResponse) ? "tool" : role === "model" ? "assistant" : "user", + parts: toParts(parts), +}) + +/** Copies what ADK records on its spans into the GenAI attributes Maple reads. */ +class AdkSpanProcessor extends tracing.BatchSpanProcessor { + override onEnd(span: tracing.ReadableSpan) { + // ADK's summary of parallel tool calls, which are also traced one by one + if (span.name === "execute_tool (merged)") return + const attributes = span.attributes + + if (span.name === "call_llm") { + const request = parse(attributes["gcp.vertex.agent.llm_request"]) + const response = parse(attributes["gcp.vertex.agent.llm_response"]) + const usage = response.usageMetadata ?? {} + const system = request.config?.systemInstruction + attributes["gen_ai.operation.name"] = "chat" + attributes["gen_ai.provider.name"] = "gcp.gemini" + if (usage.thoughtsTokenCount) { + attributes["gen_ai.usage.reasoning.output_tokens"] = usage.thoughtsTokenCount + } + if (usage.cachedContentTokenCount) attributes["gen_ai.usage.cache_read.input_tokens"] = usage.cachedContentTokenCount + if (request.contents) attributes["gen_ai.input.messages"] = JSON.stringify(request.contents.map(toMessage)) + if (response.content?.parts?.length) { + const finish_reason = response.finishReason?.toLowerCase() + attributes["gen_ai.output.messages"] = JSON.stringify([{ ...toMessage(response.content), finish_reason }]) + } + if (system) { + const parts = typeof system === "string" ? [{ type: "text", content: system }] : toParts(system.parts) + attributes["gen_ai.system_instructions"] = JSON.stringify(parts) + } + } else if (span.name.startsWith("execute_tool ")) { + if (attributes["gen_ai.operation.name"] === undefined) { + // ADK leaves the span bare when the tool threw or doesn't exist + attributes["gen_ai.operation.name"] = "execute_tool" + attributes["gen_ai.tool.name"] = span.name.slice("execute_tool ".length) + attributes["error.type"] = "tool_error" + } else { + const result = attributes["gcp.vertex.agent.tool_response"] + attributes["gen_ai.tool.call.arguments"] = attributes["gcp.vertex.agent.tool_call_args"] + attributes["gen_ai.tool.call.result"] = result + if (parse(result).error) attributes["error.type"] = "tool_error" + } + } + + // ADK's own copies of the same content, which Maple doesn't read + for (const key of ["llm_request", "llm_response", "tool_call_args", "tool_response"]) { + delete attributes[`gcp.vertex.agent.${key}`] + } + super.onEnd(span) + } +} + +// Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +export const spanProcessor = new AdkSpanProcessor(new OTLPTraceExporter()) + +// Reads OTEL_SERVICE_NAME +export const sdk = new NodeSDK({ spanProcessors: [spanProcessor] }) +sdk.start() +``` + +If your app already starts OpenTelemetry (Sentry, auto-instrumentation, your own `NodeTracerProvider`), skip the `NodeSDK` lines and add `spanProcessor` to that provider's span processors instead. + +`npx adk web` and `npx adk api_server` create their own provider without this processor, so sessions traced there show no transcript. Trace your own `Runner` instead. + + + + +## Use one ADK session per conversation + + + + +Each `runner.run_async()` call is one turn. Pass your conversation id as `session_id` on every turn: + +```py +async def chat(conversation_id: str, user_id: str, text: str) -> str: + reply = "" + async for event in runner.run_async( + user_id=user_id, + session_id=conversation_id, # the same id on every turn of this conversation + new_message=types.Content(role="user", parts=[types.Part(text=text)]), + ): + if event.is_final_response() and event.content and event.content.parts: + reply = "".join(part.text or "" for part in event.content.parts) + return reply +``` + +With `auto_create_session=True`, the runner creates the session on the first turn. Don't call `create_session()` without a `session_id`, or every message becomes its own session. + +To call an agent like a tool, add it to `sub_agents` with `mode="single_turn"` instead of using `AgentTool`, which can move the turn into a separate session. + + + + +Each `runner.runAsync()` call is one turn. Pass your conversation id as `sessionId` on every turn: + +```ts +// agent.ts +import { FunctionTool, InMemorySessionService, LlmAgent, Runner } from "@google/adk" +import { z } from "zod" + +const getWeather = new FunctionTool({ + name: "get_weather", + description: "Get the current weather for a city.", + parameters: z.object({ city: z.string() }), + execute: async ({ city }) => ({ city, temperature_c: 21, condition: "partly cloudy" }), +}) + +const agent = new LlmAgent({ + name: "assistant", + model: "gemini-2.5-flash", + instruction: "You are a concise assistant.", + tools: [getWeather], +}) + +const sessionService = new InMemorySessionService() +const runner = new Runner({ appName: "support", agent, sessionService }) + +export async function chat(conversationId: string, userId: string, text: string) { + // The same id on every turn of this conversation + await sessionService.getOrCreateSession({ appName: "support", userId, sessionId: conversationId }) + + let reply = "" + for await (const event of runner.runAsync({ + userId, + sessionId: conversationId, + newMessage: { role: "user", parts: [{ text }] }, + })) { + if (event.content?.parts && !event.partial) { + reply = event.content.parts.map((part) => part.text ?? "").join("") + } + } + return reply +} +``` + +`getOrCreateSession()` creates the session under your id on the first turn. Don't call `createSession()` without a `sessionId`, or every message becomes its own session. + + + + +## Flush before a short-lived process exits + + + + +A script, notebook cell or job that exits within 5 seconds of its last turn loses that turn. Flush when the work ends: + +```py +# script.py +import telemetry # first + +import asyncio + +from main import chat + + +async def main(): + try: + await chat("support-4821", "user-17", "What's the weather in Berlin?") + finally: + telemetry.provider.force_flush() + telemetry.provider.shutdown() + + +asyncio.run(main()) +``` + +In a server, call `provider.shutdown()` from your shutdown hook. On Cloud Run or another platform that freezes the CPU between requests, call `provider.force_flush()` before each response returns. + + + + +A script or job that exits right after its last turn loses that turn. Shut the SDK down when the work ends: + +```ts +// script.ts +import { sdk } from "./instrumentation" +import { chat } from "./agent" + +try { + console.log(await chat("support-4821", "user-17", "What's the weather in Berlin?")) +} finally { + await sdk.shutdown() +} +``` + +In a server, call `await sdk.shutdown()` from your shutdown hook. In a serverless handler, or on a platform that freezes the CPU between requests, call `await spanProcessor.forceFlush()` before each response returns. + + + + +## Check that it works + +Run a conversation of two or three turns, one calling a tool, then open **Agent Sessions**. You should see one session with the framework **Google ADK**, one turn per run, tool calls with their arguments and results, and tokens on each model call. Cost shows as unpriced. + +## Troubleshooting + +- **No spans at all.** No tracer provider is registered when the runner runs. Import `telemetry.py` or `instrumentation.ts` first. +- **Tokens but an empty transcript.** In Python, `OTEL_SEMCONV_STABILITY_OPT_IN` or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` is missing, or the capture variable is `true`. In TypeScript, `AdkSpanProcessor` isn't on the provider, or `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS` is `false`. +- **Every message is its own session.** Pass the same conversation id as the session id on every turn. +- **A failing tool shows as successful.** It returned `{"status": "error", ...}`. Return an object with an `"error"` key, or raise an error. +- **Every model call appears twice.** Another instrumentor wraps the same calls (`litellm.callbacks = ["otel"]`, `openinference-instrumentation-google-adk`), or a second OpenTelemetry SDK exports the same spans. Remove it. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): how Maple builds sessions, turns and checks. +- [ADK agent activity traces](https://google.github.io/adk-docs/observability/traces/): ADK's span reference and export setup. diff --git a/apps/landing/src/content/docs/agent-tracing/haystack.md b/apps/landing/src/content/docs/agent-tracing/haystack.md new file mode 100644 index 0000000000..d3d4736c9c --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/haystack.md @@ -0,0 +1,265 @@ +--- +title: "Trace Haystack agents with OpenTelemetry" +description: "Send Haystack 3 Agent and pipeline runs to Maple as one Agent Session per conversation, with transcript, model, tokens, cost and failed tool calls." +group: "AI Agents" +order: 28 +navLabel: "Haystack" +icon: "haystack" +--- + +This guide adds `maple_haystack.py`, a Haystack tracer that records model, tokens, messages and failed tool calls for Maple, and a `conversation()` block that groups runs into one session. + +Tested with `haystack-ai` 3.2 and `opentelemetry-haystack` 1.0 on Python 3.10 or later. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-haystack](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-haystack) skill and follows it. + +```text +Set up Maple agent tracing for Haystack in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-haystack -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install and add the Maple tracer + +```bash +pip install "haystack-ai>=3.2" "opentelemetry-haystack>=1.0" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Save this file next to your app as `maple_haystack.py`: + +```py +import json +import logging +from collections.abc import Iterator +from contextlib import contextmanager +from contextvars import ContextVar +from typing import Any + +from haystack.dataclasses import ChatMessage +from haystack_integrations.tracing.opentelemetry import OpenTelemetrySpan, OpenTelemetryTracer +from opentelemetry import trace +from opentelemetry.trace import StatusCode + +logger = logging.getLogger(__name__) +_conversation_id: ContextVar[str | None] = ContextVar("maple_conversation_id", default=None) + + +@contextmanager +def conversation(conversation_id: str) -> Iterator[None]: + token = _conversation_id.set(conversation_id) + try: + yield + finally: + _conversation_id.reset(token) + + +def _messages(messages: list[ChatMessage]) -> str: + out = [] + for m in messages: + parts: list[dict[str, Any]] = [{"type": "text", "content": t} for t in m.texts] + parts += [{"type": "tool_call", "id": c.id, "name": c.tool_name, "arguments": c.arguments} for c in m.tool_calls] + parts += [{"type": "tool_call_response", "id": r.origin.id, "response": r.result} for r in m.tool_call_results] + out.append({"role": m.role.value, "parts": parts}) + return json.dumps(out, default=str) + + +class MapleSpan(OpenTelemetrySpan): + def __init__(self, span: trace.Span, operation: str | None, content: bool) -> None: + super().__init__(span) + self._operation = operation + self._content = content + + def set_content_tag(self, key: str, value: Any) -> None: + try: + if self._operation == "chat": + self._chat(key, value) + elif self._operation == "execute_tool": + self._tool(key, value) + except Exception: # a tracing bug must never fail the agent run + logger.exception("maple_haystack: could not map %s", key) + if self._content: + self.set_tag(key, value) + + def _chat(self, key: str, value: Any) -> None: + if key.endswith(".input") and self._content: + messages = value["messages"] + system = [{"type": "text", "content": m.text} for m in messages if m.is_from("system")] + if system: + self._span.set_attribute("gen_ai.system_instructions", json.dumps(system)) + self._span.set_attribute("gen_ai.input.messages", _messages([m for m in messages if not m.is_from("system")])) + elif key.endswith(".output"): + replies = value["replies"] + meta = replies[0].meta + usage = meta.get("usage") or {} + prompt_details = usage.get("prompt_tokens_details") or {} + attributes = { + "gen_ai.response.model": meta.get("model"), + "gen_ai.response.finish_reasons": meta.get("finish_reason"), + "gen_ai.usage.input_tokens": usage.get("prompt_tokens"), + "gen_ai.usage.output_tokens": usage.get("completion_tokens"), + "gen_ai.usage.cache_read.input_tokens": prompt_details.get("cached_tokens"), + "gen_ai.usage.cache_write.input_tokens": prompt_details.get("cache_write_tokens"), + "gen_ai.usage.reasoning.output_tokens": (usage.get("completion_tokens_details") or {}).get("reasoning_tokens"), + "gen_ai.usage.cost": usage.get("cost"), + } + if self._content: + attributes["gen_ai.output.messages"] = _messages(replies) + self._span.set_attributes({k: v for k, v in attributes.items() if v is not None}) + + def _tool(self, key: str, value: Any) -> None: + if key.endswith(".output") and isinstance(value, dict) and "error" in value: + self._span.set_status(StatusCode.ERROR, str(value["error"]) if self._content else "Tool invocation failed") + self._span.set_attribute("error.type", "ToolInvocationError") + if self._content: + attribute = "gen_ai.tool.call.arguments" if key.endswith(".input") else "gen_ai.tool.call.result" + self._span.set_attribute(attribute, value if isinstance(value, str) else json.dumps(value, default=str)) + + +class MapleHaystackTracer(OpenTelemetryTracer): + def __init__(self, tracer: trace.Tracer, *, content: bool = True) -> None: + super().__init__(tracer) + self._content = content + + @contextmanager + def trace(self, operation_name: str, tags: dict[str, Any] | None = None, parent_span: Any = None) -> Iterator[MapleSpan]: + tags = dict(tags or {}) + attributes: dict[str, str] = {} + operation = None + if operation_name == "haystack.agent.run": + parent = getattr(trace.get_current_span(), "attributes", None) or {} + attributes["gen_ai.operation.name"] = "invoke_agent" + attributes["gen_ai.agent.name"] = parent.get("haystack.component.name") or parent.get("gen_ai.tool.name") or "agent" + elif operation_name == "haystack.agent.step.llm" or str(tags.get("haystack.component.type", "")).endswith("ChatGenerator"): + operation = attributes["gen_ai.operation.name"] = "chat" + elif operation_name == "haystack.agent.step.tool": + operation = attributes["gen_ai.operation.name"] = "execute_tool" + attributes["gen_ai.tool.name"] = tags["haystack.tool.name"] + if conversation_id := _conversation_id.get(): + attributes["gen_ai.conversation.id"] = conversation_id + if not self._content: + tags.pop("haystack.pipeline.input_data", None) + + with self._tracer.start_as_current_span(operation_name, attributes=attributes) as raw_span: + span = MapleSpan(raw_span, operation, self._content) + span.set_tags(tags) + yield span +``` + +## Export to Maple + +Configure OpenTelemetry once at startup and hand Haystack the tracer: + +```py +# telemetry.py +from haystack import tracing +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +from maple_haystack import MapleHaystackTracer + +provider = TracerProvider( + resource=Resource.create({"service.name": "support-agent", "deployment.environment.name": "production"}) +) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +tracing.enable_tracing(MapleHaystackTracer(trace.get_tracer("haystack"))) +``` + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. + +Import `telemetry` before the first `pipeline.run()` or `agent.run()`. Keep the tracer name `"haystack"`, since Maple labels the sessions by it. If the app already has a `TracerProvider`, skip the provider lines and pass `trace.get_tracer("haystack")` from the existing one. + +## Group turns into one session + +Each run is its own trace. Wrap every run in `conversation()` with your chat's id: + +```py +from haystack import Pipeline +from haystack.components.agents import Agent +from haystack.dataclasses import ChatMessage +from haystack_integrations.components.generators.openrouter import OpenRouterChatGenerator + +from maple_haystack import conversation + +chat_generator = OpenRouterChatGenerator(model="openai/gpt-4o-mini") +agent = Agent(chat_generator=chat_generator, tools=[get_weather, calculate], system_prompt=SYSTEM_PROMPT) +pipeline = Pipeline() +pipeline.add_component("assistant", agent) + + +def handle_message(chat_id: str, history: list[ChatMessage], text: str) -> list[ChatMessage]: + with conversation(chat_id): + result = pipeline.run({"assistant": {"messages": [*history, ChatMessage.from_user(text)]}}) + # The Agent re-adds its system prompt on every run, so don't store it + return [m for m in result["assistant"]["messages"] if not m.is_from("system")] +``` + +Use the id your app already stores for the chat. A new UUID per request gives one session per message, and one id for the whole process merges every user into one session. + +## Agent names, content and tokens + +Each agent is named after its pipeline component (`assistant` above) or the `name=` of the `AgentTool` that wraps it. An Agent run directly with `agent.run()` is called `agent`. + +To keep prompts and responses out of Maple, pass `content=False`: + +```py +tracing.enable_tracing(MapleHaystackTracer(trace.get_tracer("haystack"), content=False)) +``` + +`HAYSTACK_CONTENT_TRACING_ENABLED` has no effect with this tracer. + +A streamed `OpenAIChatGenerator` reply carries no token counts unless you ask for them (OpenRouter always sends usage): + +```py +OpenAIChatGenerator(model="gpt-4o-mini", generation_kwargs={"stream_options": {"include_usage": True}}) +``` + +Cost only appears with `OpenRouterChatGenerator`. Other providers show tokens and read as **unpriced**. + +## Flush before short-lived processes exit + +Scripts, notebooks, cron jobs and serverless handlers need an explicit flush before they exit: + +```py +try: + handle_message(chat_id, history, text) +finally: + provider.force_flush() + provider.shutdown() +``` + +Long-running servers only need `provider.shutdown()` in their shutdown hook. + +## Check that it works + +Run a conversation of at least two turns, one with a tool call, and open **Agent Sessions** filtered by your service name. You should see one session per conversation id, labelled Haystack, with one turn per `pipeline.run()`, model calls with tokens, and tool calls with failed ones marked. + +## Troubleshooting + +- **Every request is its own session.** Wrap the `pipeline.run()` call itself in `conversation()`, and pass the chat's stored id. +- **Spans but no model calls, tokens or transcript.** `enable_tracing()` got the plain `OpenTelemetryTracer`, or a later call replaced `MapleHaystackTracer`. +- **No Haystack spans at all.** Haystack 3 doesn't trace until you call `tracing.enable_tracing(...)`. Call it before the first run. +- **Every model call appears twice.** Remove `openinference-instrumentation-haystack` or OpenLLMetry's `opentelemetry-instrumentation-haystack`. +- **Tokens are zero with a non-OpenAI generator.** Print `result["replies"][0].meta["usage"]` once and add its keys to `_chat()`. +- **A failed tool isn't counted.** The tool caught its own exception. Let it raise, or return `{"error": ...}`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) diff --git a/apps/landing/src/content/docs/agent-tracing/langchain.mdx b/apps/landing/src/content/docs/agent-tracing/langchain.mdx new file mode 100644 index 0000000000..7ac64d4b30 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/langchain.mdx @@ -0,0 +1,385 @@ +--- +title: "Trace LangChain and LangGraph agents with OpenTelemetry" +description: "Send LangChain and LangGraph runs to Maple with OpenInference, one Agent Session per thread." +group: "AI Agents" +order: 13 +navLabel: "LangChain & LangGraph" +icon: "langchain" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +OpenInference's LangChain instrumentation sends LangChain and LangGraph runs to Maple, in Python (`langchain`, `langgraph`) and TypeScript (`langchain`, `@langchain/langgraph`). Pass the conversation's `thread_id` on every call so a chat becomes one session. + +You need Python 3.10 or later, or Node.js 20 or later with `@langchain/core` 1.x (Bun works too). + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-langchain](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-langchain) skill and follows it. + +```text +Set up Maple agent tracing for LangChain & LangGraph in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-langchain -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the instrumentor + + + + +```bash +pip install "langchain>=1.4" "langgraph>=1.2" "langchain-openai>=1.6" \ + "openinference-instrumentation-langchain>=0.1.76" "openinference-instrumentation>=0.1.66" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Keep the explicit `openinference-instrumentation` pin. Older versions don't work with this setup. + + + + +```bash +npm install @arizeai/openinference-instrumentation-langchain @opentelemetry/sdk-node @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto +``` + +`@opentelemetry/sdk-node` has to be 0.209 or newer. + + + + +## Point the exporter at Maple + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. If you pass the endpoint to the exporter in code instead, it has to end in `/v1/traces`. + +## Initialize tracing + + + + +Add a `tracing.py` and import it at the top of your entry point, before the first `invoke()`: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.langchain import LangChainInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +# The name= you gave create_agent(), and graph nodes that act as agents +AGENT_NAMES = {"assistant"} +# Your tool node +STEP_NAMES = {"tools"} + + +class AgentSpans(SpanProcessor): + def on_start(self, span, parent_context=None): + if span.instrumentation_scope.name != "openinference.instrumentation.langchain": + return + if span.name in AGENT_NAMES: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", span.name) + elif span.name in STEP_NAMES: + span.set_attribute("gen_ai.operation.name", "invoke_workflow") + + +provider = TracerProvider() +provider.add_span_processor(AgentSpans()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +LangChainInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +Put every agent's `name=` in `AGENT_NAMES`, sub-agents included, so each gets its own lane. If your tool node isn't called `tools`, add its name to `STEP_NAMES`. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `AgentSpans()` and the exporter to it and pass it to `instrument()`. + + + + +The TypeScript instrumentation only writes OpenInference attributes, so add a span processor that copies them into the GenAI attributes Maple reads. Save this file as `genai-spans.ts`: + +```ts +// genai-spans.ts: adds the OpenTelemetry GenAI attributes Maple reads to OpenInference's LangChain spans +import type { Span, SpanProcessor } from "@opentelemetry/sdk-trace-base" + +// Hand-built StateGraph agents, by compiled name. createAgent({ name }) is detected on its own. +const AGENT_NAMES = new Set() + +const ROLES: Record = { human: "user", ai: "assistant" } + +function parse(value: unknown): any { + if (typeof value !== "string") return undefined + try { + return JSON.parse(value) + } catch { + return undefined + } +} + +function text(content: unknown): string { + if (typeof content === "string") return content + if (!Array.isArray(content)) return "" + return content.map((block) => (block?.type === "text" ? block.text : "")).join("") +} + +// LangChain messages are serialized as { lc, id: [..., "AIMessage"], kwargs }; plain inputs are { role, content } +function toGenAiMessage(message: any) { + const fields = message?.lc ? message.kwargs : (message ?? {}) + const type = message?.lc + ? String(message.id.at(-1)).replace(/Message(Chunk)?$/, "").toLowerCase() + : String(fields.role ?? fields.type) + const role = ROLES[type] ?? type + const parts: object[] = [] + if (role === "tool") { + parts.push({ type: "tool_call_response", id: fields.tool_call_id, response: text(fields.content) }) + } else if (text(fields.content)) { + parts.push({ type: "text", content: text(fields.content) }) + } + for (const call of fields.tool_calls ?? []) { + parts.push({ type: "tool_call", id: call.id, name: call.name, arguments: call.args }) + } + return { role, parts } +} + +export class GenAiSpans implements SpanProcessor { + // Tool call arguments by call id, from the model reply that requested them + private toolArgs = new Map() + + onEnding(span: Span) { + if (span.instrumentationScope.name !== "@arizeai/openinference-instrumentation-langchain") return + const attrs = span.attributes + const set = (key: string, value: unknown) => { + if (value === undefined || value === null || value === "") return + span.setAttribute(key, typeof value === "object" ? JSON.stringify(value) : (value as string | number)) + } + const metadata = parse(attrs["metadata"]) ?? {} + const input = parse(attrs["input.value"]) + const output = parse(attrs["output.value"]) + const kind = attrs["openinference.span.kind"] + + if (kind === "LLM") { + const generations = output?.generations?.[0] ?? [] + const reply = generations[0]?.message?.kwargs ?? {} + const finish = generations[0]?.generationInfo?.finish_reason ?? reply.response_metadata?.finish_reason + if (this.toolArgs.size > 1000) this.toolArgs.clear() // calls whose tool never ran + for (const call of reply.tool_calls ?? []) this.toolArgs.set(call.id, call.args) + set("gen_ai.operation.name", "chat") + set("gen_ai.provider.name", metadata.ls_provider) + set("gen_ai.request.model", attrs["llm.model_name"]) + set("gen_ai.response.model", reply.response_metadata?.model_name) + set("gen_ai.response.id", reply.id) + set("gen_ai.usage.input_tokens", attrs["llm.token_count.prompt"]) + set("gen_ai.usage.output_tokens", attrs["llm.token_count.completion"]) + if (finish) span.setAttribute("gen_ai.response.finish_reasons", [finish]) + set("gen_ai.input.messages", (input?.messages?.[0] ?? []).map(toGenAiMessage)) + set( + "gen_ai.output.messages", + generations.map((g: any) => ({ ...toGenAiMessage(g.message ?? { role: "assistant", content: g.text }), finish_reason: finish ?? "stop" })), + ) + } else if (kind === "TOOL") { + // A tool called by the model returns a serialized ToolMessage + const message = output?.output?.lc ? output.output.kwargs : undefined + const content = message ? text(message.content) : attrs["output.value"] + const result = parse(content) + set("gen_ai.operation.name", "execute_tool") + set("gen_ai.tool.name", attrs["tool.name"]) + set("gen_ai.tool.call.id", message?.tool_call_id) + set("gen_ai.tool.call.arguments", this.toolArgs.get(message?.tool_call_id)) + this.toolArgs.delete(message?.tool_call_id) + if (content !== undefined) set("gen_ai.tool.call.result", typeof result === "object" && result !== null ? result : content) + } else if (span.name === metadata.lc_agent_name || AGENT_NAMES.has(span.name)) { + const messages = output?.messages ?? [] + set("gen_ai.operation.name", "invoke_agent") + set("gen_ai.agent.name", span.name) + set("gen_ai.input.messages", (input?.messages ?? []).map(toGenAiMessage)) + set("gen_ai.output.messages", messages.slice(-1).map(toGenAiMessage)) + } else if (kind !== "RETRIEVER" && kind !== "EMBEDDING") { + // Graph nodes, prompts and other runnables: steps, not model or tool calls + set("gen_ai.operation.name", "invoke_workflow") + } + } + + onStart() {} + onEnd() {} + forceFlush() { + return Promise.resolve() + } + shutdown() { + return Promise.resolve() + } +} +``` + +Then create an `instrumentation.ts` and import it as the first line of your entry point (`import "./instrumentation"`): + +```ts +// instrumentation.ts +import { LangChainInstrumentation } from "@arizeai/openinference-instrumentation-langchain" +import * as CallbackManagerModule from "@langchain/core/callbacks/manager" +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { NodeSDK } from "@opentelemetry/sdk-node" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { GenAiSpans } from "./genai-spans" + +// Reads OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES and OTEL_EXPORTER_OTLP_* +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) +export const sdk = new NodeSDK({ + spanProcessors: [new GenAiSpans(), spanProcessor], + // Room for long chats: OpenInference writes several attributes per message + spanLimits: { attributeCountLimit: 1000 }, +}) +sdk.start() + +new LangChainInstrumentation().manuallyInstrument(CallbackManagerModule) +``` + +Keep the `manuallyInstrument()` call. Without it, nothing from LangChain is traced. + +Give every `createAgent()` a `name`, sub-agents included, so each gets its own lane. For a graph you build with `StateGraph`, compile it with a name (`.compile({ name: "planner" })`) and add that name to `AGENT_NAMES`. + +If the app already starts a `NodeSDK` or another tracer provider (auto-instrumentations, Sentry), add `new GenAiSpans()` and the exporter's processor to its span processors instead of starting a second SDK. + + + + +## Group a conversation with thread_id + + + + +Pass your app's conversation id as `thread_id` on every `invoke()`, `stream()` and `Command(resume=...)`: + +```py +import tracing # first, before the first invoke() + +from langchain.agents import create_agent +from langchain_openai import ChatOpenAI +from langgraph.checkpoint.memory import InMemorySaver + +agent = create_agent( + ChatOpenAI(model="gpt-4o-mini", stream_usage=True), + tools=[get_weather, calculate], + name="assistant", + checkpointer=InMemorySaver(), +) + + +def handle_message(conversation_id: str, text: str) -> str: + result = agent.invoke( + {"messages": [{"role": "user", "content": text}]}, + {"configurable": {"thread_id": conversation_id}}, + ) + return result["messages"][-1].content +``` + +Use the id your app stores the chat under. A new UUID per request gives you one session per message. A plain chain (`prompt | model`, no graph) ignores `configurable`, so pass `{"metadata": {"thread_id": conversation_id}}` instead. + +Keep `stream_usage=True` if `ChatOpenAI` uses a custom `base_url`, vLLM or a gateway. Without it, streamed replies have no tokens. + + + + +Pass your app's conversation id as `thread_id` on every `invoke()`, `stream()` and `new Command({ resume })`: + +```ts +import "./instrumentation" // first, before the first invoke() + +import { ChatOpenAI } from "@langchain/openai" +import { MemorySaver } from "@langchain/langgraph" +import { createAgent } from "langchain" + +const agent = createAgent({ + model: new ChatOpenAI({ model: "gpt-4o-mini" }), + tools: [getWeather, calculate], + name: "assistant", + checkpointer: new MemorySaver(), +}) + +export async function handleMessage(conversationId: string, text: string) { + const result = await agent.invoke( + { messages: [{ role: "user", content: text }] }, + { configurable: { thread_id: conversationId } }, + ) + return result.messages.at(-1)?.content +} +``` + +Use the id your app stores the chat under. A new UUID per request gives you one session per message. A plain chain (`prompt.pipe(model)`, no graph) ignores `configurable`, so pass `{ metadata: { thread_id: conversationId } }` instead. + + + + +## Flush in short-lived processes + + + + +Long-running servers need nothing. In serverless handlers, notebooks and task workers, flush after each run: + +```py +from tracing import provider + +try: + handle_message("conv-42", "What's the weather in Berlin?") +finally: + provider.force_flush() +``` + +On LangGraph Server (`langgraph dev` or self-hosted), import `tracing` at the top of the graph module that `langgraph.json` points to, and set the `OTEL_*` variables in the server's environment. Server threads group into sessions without extra code. + + + + +Long-running servers need nothing. At the end of a script, call `await sdk.shutdown()`. In serverless handlers and queue workers, flush after each run: + +```ts +import { spanProcessor } from "./instrumentation" + +try { + await handleMessage("conv-42", "What's the weather in Berlin?") +} finally { + await spanProcessor.forceFlush() +} +``` + + + + +## Check that it works + +Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session with one turn per `invoke()`, a readable transcript, model calls with tokens, and tool calls named after your tools. + +Cost shows as **unpriced**, which is expected. + +## Troubleshooting + +- **No spans at all.** Load the tracing file before the first `invoke()`. In Python, pass `tracer_provider=provider`; in TypeScript, keep the `manuallyInstrument()` call. Then check the logs for exporter errors. +- **One session per message.** The `thread_id` is missing or changes per request (plain chains need it in `metadata`). +- **Streamed replies have no tokens.** In Python, set `stream_usage=True` on `ChatOpenAI`. In TypeScript, remove `streamUsage: false`. +- **Every model call appears twice.** Remove `LANGSMITH_OTEL_ENABLED` and any provider instrumentor such as `openinference-instrumentation-openai`. Plain `LANGSMITH_TRACING=true` is fine. +- **No agent lane, extra tool calls, or a tool shown as an agent.** In Python, fill `AGENT_NAMES` and `STEP_NAMES`, and don't put "agent" in tool names. In TypeScript, give every `createAgent()` a `name`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [OpenRouter](/docs/agent-tracing/openrouter): if your models go through OpenRouter. +- [LangChain docs](https://docs.langchain.com) diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md new file mode 100644 index 0000000000..c5dc7204d8 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -0,0 +1,285 @@ +--- +title: "Trace LiteLLM agents and the LiteLLM Proxy with OpenTelemetry" +description: "Send LiteLLM's model-call spans to Maple from the Python SDK or the LiteLLM Proxy, and add the agent and tool spans that group each conversation into one Agent Session." +group: "AI Agents" +order: 40 +navLabel: "LiteLLM" +icon: "litellm" +--- + +LiteLLM traces each model call. Your code adds the agent and tool spans and passes a session id on every call so Maple groups the turns into one session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-litellm](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-litellm) skill and follows it. + +```text +Set up Maple agent tracing for LiteLLM in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-litellm -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Trace in your app or at the proxy + +If your code calls `litellm.acompletion()`, follow the next sections. If your app calls a LiteLLM Proxy you run, see [Trace at the LiteLLM Proxy](#trace-at-the-litellm-proxy). Trace model calls in one place only, or every call shows up twice. + +## Install LiteLLM and the exporter + +```bash +pip install "litellm==1.103.0" "opentelemetry-sdk==1.43.0" "opentelemetry-exporter-otlp-proto-http==1.43.0" +``` + +Keep OpenTelemetry at 1.43. Version 1.44 and later break LiteLLM 1.103's logger. + +Point the exporter at Maple: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. + +## Register LiteLLM's v2 logger + +Use the v2 logger (`OpenTelemetryV2`). The default v1 logger makes every call its own session. Pass it your `TracerProvider`: + +```py +# tracing.py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +import litellm +from litellm.integrations.otel.logger import OpenTelemetryV2 +from litellm.integrations.otel.model.config import OpenTelemetryV2Config + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +litellm.callbacks = [ + OpenTelemetryV2( + config=OpenTelemetryV2Config(capture_message_content="span_only"), + tracer_provider=provider, + ) +] + +tracer = trace.get_tracer("support-agent") +``` + +Import `tracing` at the top of your entry point. If your app already has a `TracerProvider`, pass that one. Don't also add `"otel"` to `litellm.callbacks`, which registers a second logger. + +`"span_only"` records prompts and replies for the transcript. Use `"no_content"` to keep them out of Maple. + +## Wrap the agent loop in agent and tool spans + +Wrap each agent run in an `invoke_agent` span and each tool call in an `execute_tool` span. The v2 logger only traces `acompletion()`, not `completion()`: + +```py +# agent.py +import asyncio +import inspect +import json +from contextlib import contextmanager +from dataclasses import dataclass, field + +import litellm +from opentelemetry.trace import StatusCode + +from tracing import tracer + + +@dataclass +class Agent: + name: str + model: str + instructions: str + tools: dict = field(default_factory=dict) # tool name -> function (sync or async) + schemas: list = field(default_factory=list) # OpenAI-style tool definitions + + +@contextmanager +def agent_span(name: str): + with tracer.start_as_current_span(f"invoke_agent {name}") as span: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", name) + yield span + + +async def run_tool(agent: Agent, call) -> str: + name = call.function.name + with tracer.start_as_current_span(f"execute_tool {name}") as span: + span.set_attribute("gen_ai.operation.name", "execute_tool") + span.set_attribute("gen_ai.tool.name", name) + span.set_attribute("gen_ai.tool.call.id", call.id) + span.set_attribute("gen_ai.tool.call.arguments", call.function.arguments) + try: + result = agent.tools[name](**json.loads(call.function.arguments or "{}")) + if inspect.isawaitable(result): + result = await result + except Exception as exc: + span.set_status(StatusCode.ERROR, str(exc)) + span.set_attribute("error.type", type(exc).__name__) + result = {"error": str(exc)} + output = json.dumps(result) + span.set_attribute("gen_ai.tool.call.result", output) + return output + + +async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: + with agent_span(agent.name): + while True: + response = await litellm.acompletion( + model=agent.model, + messages=[{"role": "system", "content": agent.instructions}, *messages], + tools=agent.schemas or None, + litellm_session_id=conversation_id, + ) + message = response.choices[0].message + messages.append(message.model_dump(exclude_none=True)) + if not message.tool_calls: + return message.content or "" + results = await asyncio.gather(*(run_tool(agent, call) for call in message.tool_calls)) + for call, output in zip(message.tool_calls, results): + messages.append({"role": "tool", "tool_call_id": call.id, "content": output}) +``` + +For sub-agents, call `run_agent` for the worker from inside a tool function, with its own `name` and the same conversation id. + +## Group every turn into one session + +`litellm_session_id=` sets the session. Pass the chat or thread id your app already has, stable for the whole conversation: + +```py +from agent import Agent, run_agent + +assistant = Agent("assistant", "openrouter/openai/gpt-4o-mini", "You are a concise assistant.") +history: dict[str, list] = {} + + +async def handle_message(chat_id: str, text: str) -> str: + messages = history.setdefault(chat_id, []) + messages.append({"role": "user", "content": text}) + return await run_agent(assistant, chat_id, messages) +``` + +Don't set `gen_ai.conversation.id` on your own `invoke_agent` span, or the session is labeled **Unidentified** instead of **LiteLLM**. + +For streaming, pass `stream_options={"include_usage": True}` and consume the stream inside the agent span, or the streamed call has no token counts. + +## Trace at the LiteLLM Proxy + +To trace at a proxy you run, enable the logger in its `config.yaml`: + +```yaml +model_list: + - model_name: gpt-4o-mini + litellm_params: + model: openrouter/openai/gpt-4o-mini + api_key: os.environ/OPENROUTER_API_KEY + +litellm_settings: + callbacks: ["otel"] +``` + +Set these in the proxy's environment: + +```bash +LITELLM_OTEL_V2=true +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +OTEL_SERVICE_NAME=litellm-proxy +OTEL_ENVIRONMENT_NAME=production +OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only +``` + +The `ghcr.io/berriai/litellm` image works as is. A pip-installed proxy needs these packages: + +```bash +pip install "litellm[proxy]==1.103.0" "opentelemetry-sdk==1.43.0" \ + "opentelemetry-exporter-otlp-proto-http==1.43.0" "opentelemetry-instrumentation-fastapi==0.64b0" +``` + +Then start it with `litellm --config config.yaml`. + +In your app, keep `agent_span` and `run_tool`, drop the LiteLLM logger from `tracing.py`, and send `traceparent` and `x-litellm-session-id` with every request: + +```py +import os + +from openai import AsyncOpenAI +from opentelemetry import propagate + +client = AsyncOpenAI(base_url="http://localhost:4000", api_key=os.environ["LITELLM_API_KEY"]) + + +async def call_model(conversation_id: str, messages: list, tools: list | None): + headers = {"x-litellm-session-id": conversation_id} + propagate.inject(headers) + return await client.chat.completions.create( + model="gpt-4o-mini", messages=messages, tools=tools, extra_headers=headers + ) +``` + +Don't also instrument the OpenAI client in the app, or calls and tokens double. On this path the session page's LLM call count shows 2x the real number. Tokens, cost and the transcript are correct. + +## Flush before a short-lived process exits + +A script, Lambda or notebook cell that ends right after its last call loses that call's span. Drain LiteLLM's queue and flush: + +```py +import asyncio + +from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER + +from tracing import provider + + +async def flush_tracing() -> None: + await asyncio.sleep(0) # LiteLLM queues its log event on the next loop tick + await GLOBAL_LOGGING_WORKER.flush() + provider.force_flush() + + +async def main() -> None: + try: + print(await handle_message("chat-42", "What's the weather in Berlin?")) + finally: + await flush_tracing() + + +asyncio.run(main()) +provider.shutdown() +``` + +A long-running server needs this only in its shutdown hook. + +## Check that it works + +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session with the id you passed and framework **LiteLLM**, one turn per message, and a transcript with the prompts, replies and tool calls. + +## Troubleshooting + +- **`ModuleNotFoundError: No module named 'opentelemetry._events'`, or no LiteLLM spans and `Error initializing custom logger` in the log.** OpenTelemetry 1.44+ on LiteLLM 1.103. Pin OpenTelemetry to 1.43.0. +- **Your spans arrive but no `chat` spans.** The code calls sync `litellm.completion()`. Switch to `acompletion()`. +- **Every call is its own session, or spans are named `litellm_request`.** You're on the v1 logger, or no session id was sent. Use the `OpenTelemetryV2` setup and pass `litellm_session_id=` (SDK) or `x-litellm-session-id` (proxy) on every call. +- **Every model call appears twice.** Two loggers, or the proxy and an in-app OpenAI instrumentor both trace the call. Keep one. +- **The last call of a script is missing.** Await `flush_tracing()` before the event loop ends. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [LiteLLM: OpenTelemetry v2](https://docs.litellm.ai/docs/observability/opentelemetry_v2) diff --git a/apps/landing/src/content/docs/agent-tracing/llamaindex.md b/apps/landing/src/content/docs/agent-tracing/llamaindex.md new file mode 100644 index 0000000000..8250b4f39f --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/llamaindex.md @@ -0,0 +1,180 @@ +--- +title: "Trace LlamaIndex agents with OpenTelemetry" +description: "Send LlamaIndex agent and workflow runs to Maple with OpenInference, one Agent Session per conversation." +group: "AI Agents" +order: 23 +navLabel: "LlamaIndex" +icon: "llamaindex" +--- + +OpenInference's `openinference-instrumentation-llama-index` sends LlamaIndex agents and workflows to Maple. You add a small span processor and wrap every `agent.run()` in a conversation id so a chat becomes one session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-llamaindex](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-llamaindex) skill and follows it. + +```text +Set up Maple agent tracing for LlamaIndex in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-llamaindex -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the instrumentor + +```bash +pip install "llama-index-core>=0.14.25" "openinference-instrumentation-llama-index>=4.5.2" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Add your model package (`llama-index-llms-openai`, `llama-index-llms-openrouter`, ...) as usual. + +If the app uses LlamaIndex's own `llama-index-observability-otel`, remove it. Maple can't read its transcripts, and running both doubles every span. + +## Point the exporter at Maple + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, it has to end in `/v1/traces`. + +## Initialize tracing + +Add a `tracing.py` and import it at the top of your entry point, before the first `agent.run()`: + +```py +# tracing.py +from llama_index.core.instrumentation.dispatcher import active_instrument_tags +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.llama_index import LlamaIndexInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +LLM_METHODS = (".chat", ".achat", ".stream_chat", ".astream_chat", + ".complete", ".acomplete", ".stream_complete", ".astream_complete") + + +class LlamaIndexForMaple(SpanProcessor): + def __init__(self, exporter_processor: SpanProcessor): + self._next = exporter_processor + self._open_llm_spans = {} + + def on_start(self, span, parent_context=None): + agent_name = active_instrument_tags.get().get("gen_ai.agent.name") + if agent_name: + span.set_attribute("gen_ai.agent.name", agent_name) + if span.name.endswith((".call_tool", ".aggregate_tool_results")): + span.set_attribute("gen_ai.operation.name", "invoke_workflow") + if span.name.endswith(LLM_METHODS): + self._open_llm_spans[span.context.span_id] = span + self._next.on_start(span, parent_context) + + def on_end(self, span): + self._open_llm_spans.pop(span.context.span_id, None) + if span.name.endswith("._prepare_chat_with_tools"): + return + if (span.status.description or "").startswith("WaitingForEvent"): + return + outer = self._open_llm_spans.get(span.parent.span_id) if span.parent else None + if outer is not None and outer.name == span.name: + outer.set_attributes(span.attributes) + return + self._next.on_end(span) + + def shutdown(self): + self._next.shutdown() + + def force_flush(self, timeout_millis=30000): + return self._next.force_flush(timeout_millis) + + +provider = TracerProvider() +provider.add_span_processor(LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))) +trace.set_tracer_provider(provider) + +LlamaIndexInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +Always add the exporter through `LlamaIndexForMaple`, never directly. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))` to it and pass it to `instrument()`. + +## Group a conversation into one session + +Wrap each `agent.run()` call in `using_session` with the conversation id your app stores the chat under, and tag it with the agent's name: + +```py +from llama_index.core.agent.workflow import AgentStream, FunctionAgent +from llama_index.core.instrumentation.dispatcher import instrument_tags +from llama_index.core.workflow import Context +from openinference.instrumentation import using_session + +agent = FunctionAgent(name="assistant", llm=llm, tools=[get_weather, calculate], + system_prompt="You are a helpful assistant.") +contexts: dict[str, Context] = {} + + +async def handle_message(conversation_id: str, text: str): + if conversation_id not in contexts: + contexts[conversation_id] = Context(agent) + ctx = contexts[conversation_id] + with using_session(conversation_id), instrument_tags({"gen_ai.agent.name": agent.name}): + handler = agent.run(user_msg=text, ctx=ctx) + async for event in handler.stream_events(): + if isinstance(event, AgentStream): + yield event.delta + await handler +``` + +Only the `agent.run()` call needs to be inside the `with`. Consume the stream outside it. + +Keep one `Context` per conversation. A new UUID per request gives you one session per message. + +For multi-agent workflows, run the whole workflow inside `using_session(conversation_id)` and wrap each sub-agent's `run()` in its own `instrument_tags({"gen_ai.agent.name": agent.name})` to give each agent its own lane. `AgentWorkflow` handoffs show as a single agent. + +## Get tokens on streamed calls + +`FunctionAgent` streams its model calls, and OpenAI only reports tokens on a stream when asked. Pass `stream_options` on OpenAI and OpenAI-compatible models: + +```py +from llama_index.llms.openai import OpenAI + +llm = OpenAI(model="gpt-4o-mini", additional_kwargs={"stream_options": {"include_usage": True}}) +``` + +With `OpenAILike` or `OpenRouter`, also pass `is_function_calling_model=True`, or the agent never calls tools. + +## Flush in short-lived processes + +Long-running servers need nothing. In serverless handlers, notebooks and task workers, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. + +## Check that it works + +Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session labeled **LlamaIndex**, with one turn per `agent.run()`, a transcript, one model call per request with tokens, and `FunctionTool.acall` tool calls. + +Cost shows as **unpriced** and streamed model calls last about 1 ms. Both are expected. + +## Troubleshooting + +- **No spans at all.** Import `tracing` before the first `agent.run()`, check the logs for `DependencyConflict` (upgrade llama-index-core) and exporter errors. +- **One session per message.** `agent.run()` isn't inside `using_session(...)`, or the id changes per request. +- **Each model or tool call counted two or three times.** The exporter was added directly. Add it through `LlamaIndexForMaple`. +- **No tokens on streamed calls.** Add `stream_options={"include_usage": True}` through `additional_kwargs`. +- **No lanes or agent names.** Wrap each agent's `run()` in `instrument_tags({"gen_ai.agent.name": agent.name})`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [OpenRouter](/docs/agent-tracing/openrouter): cost per call if your models go through OpenRouter. diff --git a/apps/landing/src/content/docs/agent-tracing/mastra.md b/apps/landing/src/content/docs/agent-tracing/mastra.md new file mode 100644 index 0000000000..09d0bd7ab9 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/mastra.md @@ -0,0 +1,172 @@ +--- +title: "Trace Mastra agents and workflows with OpenTelemetry" +description: "Export Mastra's built-in spans to Maple with @mastra/otel-exporter and group each conversation into one Agent Session." +group: "AI Agents" +order: 11 +navLabel: "Mastra" +icon: "mastra" +--- + +Mastra traces agent runs, model calls and tool calls itself, and `@mastra/otel-exporter` sends those spans to Maple. You don't need an OpenTelemetry SDK or an instrumentation package. + +The session id is Mastra's memory thread id, so every call of a conversation must pass the same `memory: { thread }`. + +You need `@mastra/core` 1.x and Node.js 22.13 or newer. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-mastra](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-mastra) skill and follows it. + +```text +Set up Maple agent tracing for Mastra in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-mastra -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the observability packages + +```bash +npm install @mastra/core@latest @mastra/observability@latest @mastra/otel-exporter@latest +``` + +Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week, or the exporter can pick the wrong span as the model call. + +## Add the Maple span processor + +Without this processor the transcript has no user messages, sub-agents land in separate sessions, and raw provider responses (including cookies) are exported. + +```ts +// src/mastra/maple-span-processor.ts +import { SpanType, type SpanOutputProcessor } from "@mastra/core/observability" + +export const mapleSpanProcessor: SpanOutputProcessor = { + name: "maple-span-processor", + process(span) { + if (!span) return span + // One conversation id per trace: sub-agents get their own thread ids otherwise. + let root = span + while (root.parent) root = root.parent + const threadId = root.metadata?.threadId + if (threadId) span.metadata = { ...span.metadata, threadId } + // The model call span is created without its prompt: take the step's messages. + if (span.type === SpanType.MODEL_INFERENCE && span.input === undefined && span.parent?.input !== undefined) { + span.input = { messages: span.parent.input } + } + // Step spans carry the raw provider response (headers, cookies, full body) as metadata. + if (span.type === SpanType.MODEL_STEP && span.metadata) { + const { headers: _headers, body: _body, ...metadata } = span.metadata + span.metadata = metadata + } + return span + }, + async shutdown() {}, +} +``` + +## Configure the exporter + +Add observability to your `Mastra` instance: + +```ts +// src/mastra/index.ts +import { Mastra } from "@mastra/core/mastra" +import { SpanType } from "@mastra/core/observability" +import { Observability } from "@mastra/observability" +import { OtelExporter } from "@mastra/otel-exporter" +import { supportAgent } from "./agents/support" +import { mapleSpanProcessor } from "./maple-span-processor" + +// A missing key disables export; it never stops the app. +const mapleKey = process.env.MAPLE_INGEST_KEY +if (!mapleKey) console.warn("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled") + +const mapleExporter = mapleKey + ? new OtelExporter({ + provider: { + custom: { + endpoint: "https://ingest.maple.dev", + protocol: "http/protobuf", + headers: { Authorization: `Bearer ${mapleKey}` }, + }, + }, + resourceAttributes: { "deployment.environment.name": "production" }, + }) + : undefined + +export const mastra = new Mastra({ + agents: { supportAgent }, + observability: new Observability({ + configs: { + maple: { + serviceName: "support-agent", + exporters: mapleExporter ? [mapleExporter] : [], + // One span per streamed chunk adds nothing Maple uses + excludeSpanTypes: [SpanType.MODEL_CHUNK], + spanOutputProcessors: [mapleSpanProcessor], + }, + }, + }), +}) +``` + +Set `MAPLE_INGEST_KEY` to your ingest key from **Settings → Ingestion**. For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +Set the endpoint, protocol and key in code, since this exporter ignores the `OTEL_EXPORTER_OTLP_*` variables. `observability` must be an `Observability` instance; a plain object silently traces nothing. + +Only agents and workflows registered on this `Mastra` instance are traced. Get them with `mastra.getAgent()` or `mastra.getWorkflow()`. + +## Pass the thread id on every call + +Pass the same thread on every call of a conversation: + +```ts +const agent = mastra.getAgent("supportAgent") + +export async function handleMessage(chatId: string, userId: string, text: string) { + const result = await agent.generate(text, { + memory: { thread: chatId, resource: userId }, + }) + return result.text +} +``` + +Use the chat id your app already has. It must stay the same for the whole conversation and differ between conversations. `agent.stream()` takes the same option; read the stream to the end, since the spans are exported when it finishes. + +Workflow runs, and agents called without `memory`, have no thread. Put the id in the root span's metadata instead: + +```ts +const run = await mastra.getWorkflow("briefingWorkflow").createRun() +const result = await run.start({ + inputData: { request }, + tracingOptions: { metadata: { threadId: conversationId } }, +}) +``` + +Give every `Agent` a distinct `name`, or sub-agents share one lane. When a workflow step calls an agent, pass it the step's `tracingContext` (`agent.generate(prompt, { tracingContext })`) so the agent joins the workflow's trace. + +## Flush in scripts and serverless functions + +A short-lived process can exit before its spans are sent. In a script, call `await mastra.shutdown()` in a `finally` block before exiting. In a serverless handler, call `await mastra.observability.flush()` at the end of each request, after any streamed response has finished. + +## Check that it works + +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session named after your thread id with framework **Mastra**, one turn per `generate()` or `stream()` call, and a transcript with the prompts, replies and tool calls. + +Cost shows as unpriced because Mastra doesn't report it. If nothing arrives, set `logLevel: "debug"` on `OtelExporter` to log each export as `Export completed` or `Export FAILED` with the reason. + +## Troubleshooting + +- **Nothing arrives and there is no error.** `observability` is a plain object instead of `new Observability(...)`, or the agent isn't registered on the `Mastra` instance. +- **`http/protobuf exporter is not installed` at startup.** The install skipped optional dependencies. Install `@opentelemetry/exporter-trace-otlp-proto`. +- **Every message is its own session.** The call has no `memory: { thread, resource }`, or the thread id changes per request. For workflows, use `tracingOptions.metadata.threadId`. +- **The transcript has no user messages, or sub-agents land in a session named `-`.** Add `mapleSpanProcessor` to `spanOutputProcessors`. +- **A failed tool shows as successful.** The tool returned an error value. Throw an `Error` instead. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Mastra: OpenTelemetry exporter](https://mastra.ai/docs/observability/tracing/exporters/otel) diff --git a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.mdx b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.mdx new file mode 100644 index 0000000000..0d42b632ed --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.mdx @@ -0,0 +1,299 @@ +--- +title: "Trace Microsoft Agent Framework and Semantic Kernel agents with OpenTelemetry" +description: "Send Microsoft Agent Framework and Semantic Kernel traces from Python or .NET to Maple as one Agent Session per conversation." +group: "AI Agents" +order: 30 +navLabel: "Microsoft Agent Framework" +icon: "dotnet" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +Microsoft Agent Framework (MAF) emits OpenTelemetry spans in Python and .NET. You point them at Maple and add a short span processor that sets the conversation id, or every turn shows up as its own session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-microsoft-agent-framework](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-microsoft-agent-framework) skill and follows it. + +```text +Set up Maple agent tracing for Microsoft Agent Framework in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-microsoft-agent-framework -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Export traces + + + + +Install MAF with the OTLP/HTTP exporter: + +```bash +pip install "agent-framework-core>=1.19.0" "agent-framework-openai>=1.14.4" opentelemetry-exporter-otlp-proto-http +``` + +Call `configure_otel_providers()` once at startup, before you create agents. Without `enable_sensitive_data=True` the transcript is empty. + +```py +# telemetry.py +import logging +import os + +from agent_framework.observability import configure_otel_providers +from opentelemetry import trace + +from maple_tracing import ConversationIdProcessor + +key = os.environ.get("MAPLE_INGEST_KEY") +if key: + configure_otel_providers( + service_name="support-agent", + resource_attributes={"deployment.environment.name": "production"}, + otlp_endpoint="https://ingest.maple.dev", # EU: https://ingest.eu.maple.dev + otlp_protocol="http/protobuf", + otlp_headers={"Authorization": f"Bearer {key}"}, + enable_sensitive_data=True, # prompts, replies, tool arguments and results + enable_message_events=False, # skip the duplicate copy of the content in OTLP logs + ) + trace.get_tracer_provider().add_span_processor(ConversationIdProcessor()) +else: + # A missing key disables export; it never stops the app. + logging.getLogger(__name__).warning("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled") +``` + +Keep `otlp_protocol="http/protobuf"`. MAF defaults to gRPC, which Maple doesn't accept. + +If your app already has a `TracerProvider`, don't call `configure_otel_providers()`. Add Maple's exporter and `ConversationIdProcessor` to your provider, then call `enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)` from `agent_framework.observability`. + + + + +```bash +dotnet add package Microsoft.Agents.AI --version 1.22.0 +dotnet add package Microsoft.Agents.AI.OpenAI --version 1.22.0 +dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol --version 1.19.1 +``` + +```csharp +using System.ClientModel; +using Microsoft.Agents.AI; +using Microsoft.Extensions.AI; +using OpenAI; +using OpenTelemetry; +using OpenTelemetry.Exporter; +using OpenTelemetry.Resources; +using OpenTelemetry.Trace; + +using var tracerProvider = Sdk.CreateTracerProviderBuilder() + .ConfigureResource(r => r.AddService("support-agent")) + .AddSource("*Microsoft.Agents.AI*") // agent, chat and workflow spans + .AddSource("*Microsoft.Extensions.AI") // chat clients you instrument yourself + .AddProcessor(new ConversationIdProcessor()) + .AddOtlpExporter(o => + { + o.Endpoint = new Uri("https://ingest.maple.dev/v1/traces"); // EU: ingest.eu.maple.dev + o.Protocol = OtlpExportProtocol.HttpProtobuf; + o.Headers = "Authorization=Bearer YOUR_INGEST_KEY"; + }) + .Build(); + +var openAi = new OpenAIClient( + new ApiKeyCredential(Environment.GetEnvironmentVariable("OPENAI_API_KEY")!)); + +AIAgent agent = openAi.GetChatClient("gpt-4o-mini").AsIChatClient() + .AsAIAgent( + instructions: "You are a helpful assistant.", + name: "support_agent", + tools: [AIFunctionFactory.Create(GetWeather, name: "get_weather")]) + .AsBuilder() + .UseOpenTelemetry(configure: a => a.EnableSensitiveData = true) + .Build(); +``` + +Keep the leading `*` in `AddSource` (the source names start with `Experimental.`), `/v1/traces` in the endpoint, and the `HttpProtobuf` line. Pass each tool a `name`, or local functions show up under compiler-generated names. + + + + +## Group each conversation into one session + +Maple groups turns into a session by `gen_ai.conversation.id`. + + + + +This processor sets it on every span started inside a `conversation()` block: + +```py +# maple_tracing.py +from contextlib import contextmanager +from contextvars import ContextVar + +from opentelemetry.sdk.trace import SpanProcessor + +_conversation_id: ContextVar[str | None] = ContextVar("conversation_id", default=None) + + +class ConversationIdProcessor(SpanProcessor): + """Puts gen_ai.conversation.id on every span started inside `conversation()`.""" + + def on_start(self, span, parent_context=None): + if (conversation_id := _conversation_id.get()) is not None: + span.set_attribute("gen_ai.conversation.id", conversation_id) + + +@contextmanager +def conversation(conversation_id: str): + token = _conversation_id.set(conversation_id) + try: + yield + finally: + _conversation_id.reset(token) +``` + +Wrap each request in it. Use `session.session_id` as the id, or create the session with your own chat id: `agent.create_session(session_id=chat_id)`. + +```py +from agent_framework import Agent, AgentSession + +from maple_tracing import conversation + + +async def handle_message(agent: Agent, session: AgentSession, text: str) -> str: + with conversation(session.session_id): + response = await agent.run(text, session=session) + return response.text +``` + +When streaming, keep the whole `async for` loop inside the block. Don't pass the `conversation_id` chat option instead; it turns off MAF's in-memory history. + + + + +In .NET, use an `Activity` processor with an `AsyncLocal` and set it before `RunAsync`: + +```csharp +using System.Diagnostics; +using OpenTelemetry; + +sealed class ConversationIdProcessor : BaseProcessor +{ + public static readonly AsyncLocal Current = new(); + + public override void OnStart(Activity activity) + { + if (Current.Value is { } id) activity.SetTag("gen_ai.conversation.id", id); + } +} +``` + +```csharp +AgentSession session = await agent.CreateSessionAsync(); +ConversationIdProcessor.Current.Value = chatId; // once per request, before RunAsync +var response = await agent.RunAsync(userMessage, session); +``` + + + + +## Flush before a script exits + +Scripts, CLIs and notebooks that exit without flushing lose their last turns. + + + + +Shut the providers down in a `finally`: + +```py +from opentelemetry import _logs, metrics, trace + + +def shutdown_telemetry() -> None: + for provider in (trace.get_tracer_provider(), metrics.get_meter_provider(), _logs.get_logger_provider()): + provider.shutdown() + + +try: + asyncio.run(main()) +finally: + shutdown_telemetry() +``` + +In a serverless handler, call `trace.get_tracer_provider().force_flush()` before returning. + + + + +In .NET, `using var tracerProvider` flushes when `Main` ends. + + + + +## Semantic Kernel + +Semantic Kernel (SK) reads its telemetry switches at import time, so set them before the first `import semantic_kernel`. Configure your own provider with the same `ConversationIdProcessor`: + +```bash +pip install "semantic-kernel>=1.44.1" opentelemetry-sdk opentelemetry-exporter-otlp-proto-http +``` + +```py +# telemetry.py: import this before anything that imports semantic_kernel +import logging +import os + +os.environ["SEMANTICKERNEL_EXPERIMENTAL_GENAI_ENABLE_OTEL_DIAGNOSTICS"] = "true" +os.environ["SEMANTICKERNEL_EXPERIMENTAL_GENAI_ENABLE_OTEL_DIAGNOSTICS_SENSITIVE"] = "true" + +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +from maple_tracing import ConversationIdProcessor + +provider = TracerProvider(resource=Resource.create({"service.name": "support-agent"})) +provider.add_span_processor(ConversationIdProcessor()) +key = os.environ.get("MAPLE_INGEST_KEY") +if key: + provider.add_span_processor( + BatchSpanProcessor( + OTLPSpanExporter( + endpoint="https://ingest.maple.dev/v1/traces", # EU: https://ingest.eu.maple.dev/v1/traces + headers={"Authorization": f"Bearer {key}"}, + ) + ) + ) +else: + # A missing key disables export; it never stops the app. + logging.getLogger(__name__).warning("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled") +trace.set_tracer_provider(provider) +``` + +Wrap each turn in `with conversation(thread.id):`. Pass messages positionally, as in `await agent.get_response(text, thread=thread)`; the `messages=` keyword records an empty input. Only `ChatCompletionAgent` calls produce a transcript; calling the kernel directly doesn't. + +## Check that it works + +Run a conversation of two or three turns where one turn calls a tool. Within about a minute, **Agent Sessions** shows one session for it, labeled **Microsoft Agent Framework** or **Semantic Kernel**, with one turn per `agent.run()`, the transcript, and tool calls with their arguments and results. Cost shows as unpriced; MAF doesn't emit cost. + +## Troubleshooting + +- **Nothing arrives, or `ImportError: opentelemetry-exporter-otlp-proto-grpc is required`.** The protocol defaulted to gRPC. Set `otlp_protocol="http/protobuf"` or `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf`. +- **Every turn is its own session.** Register `ConversationIdProcessor` and wrap each call, including the whole streaming loop, in `conversation()`. +- **Spans but an empty transcript.** Sensitive data is off, or `OTEL_SEMCONV_STABILITY_OPT_IN` is set without `gen_ai_latest_experimental` (use `http,gen_ai_latest_experimental`). +- **.NET: no spans at all.** Use `AddSource("*Microsoft.Agents.AI*")` with the leading `*`. +- **Semantic Kernel: no `chat` or `invoke_agent` spans.** The `SEMANTICKERNEL_EXPERIMENTAL_GENAI_*` variables were set after `semantic_kernel` was imported. + +## Related + +- [Agent Sessions](/docs/agent-sessions/overview): reading a session in Maple. +- [Python instrumentation](/docs/guides/instrumentation-python) and [.NET instrumentation](/docs/guides/instrumentation-csharp): tracing the rest of the service. +- [Agent Framework observability](https://learn.microsoft.com/en-us/agent-framework/agents/observability): Microsoft's reference for the settings above. +- [Semantic Kernel telemetry](https://learn.microsoft.com/en-us/semantic-kernel/concepts/enterprise-readiness/observability/): the SK diagnostics switches. diff --git a/apps/landing/src/content/docs/agent-tracing/openai-agents.mdx b/apps/landing/src/content/docs/agent-tracing/openai-agents.mdx new file mode 100644 index 0000000000..9601efbd46 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/openai-agents.mdx @@ -0,0 +1,272 @@ +--- +title: "Trace OpenAI Agents SDK runs with OpenTelemetry" +description: "Send OpenAI Agents SDK runs to Maple as one Agent Session per conversation, with the transcript, tool calls and tokens." +group: "AI Agents" +order: 12 +navLabel: "OpenAI Agents SDK" +icon: "openai" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +OpenInference's OpenAI Agents bridge exports OpenAI Agents SDK runs to Maple, in Python (`openai-agents`) and TypeScript (`@openai/agents`). Pass the conversation id with every run, or every message shows up as its own session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-openai-agents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) skill and follows it. + +```text +Set up Maple agent tracing for the OpenAI Agents SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-openai-agents -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the bridge + + + + +```bash +pip install "openai-agents>=0.22" "openinference-instrumentation-openai-agents>=2.5" "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + + + + +```bash +npm install @openai/agents @arizeai/openinference-instrumentation-openai-agents @arizeai/openinference-core @opentelemetry/api @opentelemetry/sdk-node +``` + + + + +## Point the exporter at Maple + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. If you pass the endpoint to an exporter in code instead, give the full URL ending in `/v1/traces`. + +## Register the bridge + + + + +Add a `tracing.py` and import it at the top of your entry point, before the first `Runner.run`: + +```py +# tracing.py +import re + +from agents import set_trace_processors +from agents.tracing import TracingProcessor +from agents.tracing.span_data import GenerationSpanData, HandoffSpanData +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.openai_agents import OpenAIAgentsInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +def _chat_message(response: dict) -> dict: + """The assistant message inside a Responses-shaped dict, as a Chat Completions message.""" + text, calls = "", [] + for item in response.get("output") or []: + if item.get("type") == "message": + text += "".join(c.get("text", "") for c in item.get("content") or [] if c.get("type") == "output_text") + elif item.get("type") == "function_call": + calls.append({"id": item["call_id"], "type": "function", + "function": {"name": item["name"], "arguments": item["arguments"]}}) + return {"role": "assistant", "content": text or None, "tool_calls": calls or None} + + +class MapleSpanFixes(TracingProcessor): + """Fills two gaps in what OpenInference exports. Must run before the OpenInference processor.""" + + def on_span_end(self, span): + data = span.span_data + current = trace.get_current_span() # the matching OpenTelemetry span, still open here + if isinstance(data, HandoffSpanData) and data.to_agent: + # Handoff spans carry no tool name. + current.set_attribute("gen_ai.tool.name", re.sub(r"[^a-zA-Z0-9_]", "_", f"transfer_to_{data.to_agent}").lower()) + elif isinstance(data, GenerationSpanData) and data.output and data.output[0].get("object") == "response": + # Streamed Chat Completions calls record a Responses object OpenInference can't read. + data.output = [_chat_message(data.output[0])] + + def on_trace_start(self, t): pass + def on_trace_end(self, t): pass + def on_span_start(self, span): pass + def shutdown(self): pass + def force_flush(self): pass + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +# Replaces the SDK's default processor (which uploads to OpenAI) with MapleSpanFixes, +# then appends the OpenInference processor after it. +set_trace_processors([MapleSpanFixes()]) +OpenAIAgentsInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), + exclusive_processor=False, +) +``` + +`MapleSpanFixes` fixes handoff tool names and streamed replies. It has to run before the OpenInference processor, so keep the order shown. + +`set_trace_processors` already stops the upload to OpenAI. Don't use `set_tracing_disabled(True)` or `OPENAI_AGENTS_DISABLE_TRACING=1` for that, or you get no spans. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add the OTLP exporter to it and pass it to `instrument()` instead of creating a second one. + +If an agent uses `OpenAIChatCompletionsModel` with a non-OpenAI base URL (OpenRouter, LiteLLM, vLLM, Ollama), give it `model_settings=ModelSettings(include_usage=True)`. Otherwise streamed turns report zero tokens. + + + + +Create an `instrumentation.ts` and import it as the first line of your entry point (`import "./instrumentation"`): + +```ts +// instrumentation.ts +import * as agents from "@openai/agents" +import { OpenAIAgentsInstrumentation } from "@arizeai/openinference-instrumentation-openai-agents" +import { NodeSDK } from "@opentelemetry/sdk-node" + +// Reads OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES and OTEL_EXPORTER_OTLP_* +export const sdk = new NodeSDK() +sdk.start() + +// Replaces the SDK's default processor, which uploads traces to OpenAI +new OpenAIAgentsInstrumentation().manuallyInstrument(agents) +``` + +Don't use `setTracingDisabled(true)` or `OPENAI_AGENTS_DISABLE_TRACING=1` to stop the upload to OpenAI, or you get no spans. + +If the app already starts OpenTelemetry (auto-instrumentation, Sentry, your own `NodeTracerProvider`), skip the `NodeSDK` lines and keep the `manuallyInstrument` call. The bridge sends its spans to the global tracer provider. + + + + +## Group each conversation into one session + + + + +Wrap every run in `using_session` with the conversation's id: + +```py +from agents import RunConfig, Runner, SQLiteSession +from openinference.instrumentation import using_session + + +async def handle_message(conversation_id: str, text: str) -> str: + with using_session(conversation_id): + result = await Runner.run( + agent, + text, + session=SQLiteSession(conversation_id, "chats.db"), + run_config=RunConfig(workflow_name="support workflow"), + ) + return result.final_output +``` + +Use the id your app already stores the chat under, and give the SDK session the same one. A new UUID per request gives you one session per message. + +For streaming, call `Runner.run_streamed` inside the `with` block. + +Keep `tool` out of `workflow_name`, or Maple counts a phantom tool call per run. + + + + +Run every call inside an OpenTelemetry context that carries the conversation id as `gen_ai.conversation.id`. The bridge copies it onto every span of the run: + +```ts +import { setAttributes } from "@arizeai/openinference-core" +import { run } from "@openai/agents" +import { context } from "@opentelemetry/api" + +export async function handleMessage(conversationId: string, text: string) { + const ctx = setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }) + const result = await context.with(ctx, () => run(agent, text)) + return result.finalOutput +} +``` + +Use the id your app already stores the chat under. A new UUID per request gives you one session per message. If you pass a `session` to `run`, key it by the same id. + +For streaming, call `run(agent, text, { stream: true })` inside the `context.with` callback. You can read the stream after it returns. + + + + +## Flush in short-lived processes + + + + +In serverless functions and notebooks, flush explicitly: + +```py +from tracing import provider + +try: + asyncio.run(handle_message("conv-42", "What's the weather in Berlin?")) +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +The SDK's `flush_traces()` doesn't flush these spans. + + + + +In a script, call `await sdk.shutdown()` in a `finally` block before exiting. + +A serverless handler has to flush after every invocation, so create the span processor yourself: + +```bash +npm install @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto +``` + +```ts +// instrumentation.ts, replacing `new NodeSDK()` +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" + +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) +export const sdk = new NodeSDK({ spanProcessors: [spanProcessor] }) +``` + +Then call `await spanProcessor.forceFlush()` in a `finally` block before the handler returns. Read streamed runs to the end first. + + + + +## Check that it works + +Send two or three messages with the same conversation id, one using a tool, then open **Agent Sessions**. You should see one session with one turn per run, the tool calls, and tokens on every model call. The framework shows as **OpenAI Agents SDK**. Cost shows as unpriced. + +## Troubleshooting + +- **No spans at all.** Tracing is disabled somewhere (`set_tracing_disabled` or `setTracingDisabled`, `OPENAI_AGENTS_DISABLE_TRACING`, or the run config), or the tracing setup ran after the first run. In TypeScript, `NODE_ENV=test` also turns tracing off; call `setTracingDisabled(false)` in tests. +- **One session per message.** The run isn't inside `using_session(...)` or the `context.with` callback, or the id changes per request. Group ids and SDK session ids don't reach Maple. +- **Streamed turns have zero tokens (Python).** Add `ModelSettings(include_usage=True)` to agents on a non-OpenAI base URL. +- **Every model call appears twice.** Another instrumentation of the OpenAI client (OpenInference, Logfire, Langfuse, `@opentelemetry/instrumentation-openai`) is also active. Keep one. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Tracing in the OpenAI Agents SDK](https://openai.github.io/openai-agents-python/tracing/) (Python) and [for TypeScript](https://openai.github.io/openai-agents-js/guides/tracing/): the switch that turns off prompt and reply capture. +- [OpenRouter](/docs/agent-tracing/openrouter) and [LiteLLM](/docs/agent-tracing/litellm): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/openrouter.mdx b/apps/landing/src/content/docs/agent-tracing/openrouter.mdx new file mode 100644 index 0000000000..124ed6ccde --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/openrouter.mdx @@ -0,0 +1,182 @@ +--- +title: "Trace OpenRouter calls in Maple with Broadcast" +description: "Send every OpenRouter model call to Maple with OpenRouter Broadcast, grouped into one Agent Session per conversation." +group: "AI Agents" +order: 41 +navLabel: "OpenRouter" +icon: "openrouter" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +OpenRouter Broadcast sends a trace of every request on your OpenRouter account to Maple, with tokens, cost, prompt and completion. You set it up once in the OpenRouter dashboard, then add a `session_id` to your requests. Without it, every model call is its own session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-openrouter](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openrouter) skill and follows it. + +```text +Set up Maple agent tracing for OpenRouter in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-openrouter -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. The agent prints the values for the OpenRouter dashboard, which you enter yourself as described in the next section. + +## Point Broadcast at Maple + +1. In OpenRouter, open [Settings → Observability](https://openrouter.ai/settings/observability) and turn on **Enable Broadcast**. In an organization account, only an admin can change this. +2. Click the edit icon next to **OpenTelemetry Collector** and set **Endpoint** to the full traces URL. OpenRouter doesn't append `/v1/traces`. EU organizations use `https://ingest.eu.maple.dev/v1/traces`. + + ```text + https://ingest.maple.dev/v1/traces + ``` + +3. Set **Headers** to your Maple ingest key: + + ```json + { "Authorization": "Bearer YOUR_INGEST_KEY" } + ``` + +4. Click **Test Connection**. OpenRouter only saves the destination if the test passes. + +Leave the sampling rate at 1.0, since a lower rate drops whole conversations. Leave the API key filter empty. If your app calls `eu.openrouter.ai`, add the Europe data region. + +## Send a session id with every request + +Send a `session_id` field in the request body (or an `x-session-id` header). Use the same id on every request of a conversation, such as your chat thread id, and a new one for each conversation. + + + + +With the `openai` SDK in TypeScript, the field isn't in the types, so it needs a `@ts-expect-error`: + +```ts +import OpenAI from "openai" + +const client = new OpenAI({ + baseURL: "https://openrouter.ai/api/v1", + apiKey: process.env.OPENROUTER_API_KEY, +}) + +const completion = await client.chat.completions.create({ + model: "openai/gpt-4o-mini", + messages, + // @ts-expect-error OpenRouter-only field + session_id: conversationId, +}) +``` + +With the Vercel AI SDK, pass it under `providerOptions.openrouter`: + +```ts +import { createOpenRouter } from "@openrouter/ai-sdk-provider" +import { streamText } from "ai" + +const openrouter = createOpenRouter({ apiKey: process.env.OPENROUTER_API_KEY }) + +const result = streamText({ + model: openrouter("openai/gpt-4o-mini"), + messages, + providerOptions: { openrouter: { session_id: conversationId } }, +}) +``` + + + + +In Python, use `extra_body`: + +```py +import os + +from openai import OpenAI + +client = OpenAI(base_url="https://openrouter.ai/api/v1", api_key=os.environ["OPENROUTER_API_KEY"]) + +completion = client.chat.completions.create( + model="openai/gpt-4o-mini", + messages=messages, + extra_body={"session_id": conversation_id}, +) +``` + + + + +OpenRouter's own SDKs have a typed field: `sessionId` in `@openrouter/sdk` and `session_id=` in the `openrouter` Python package. + +## Nest Broadcast under your own traces + +Skip this if OpenRouter is your only source of traces. If your app also sends its own traces to Maple, every model call is recorded twice. Pass the active span's ids in the `trace` field so OpenRouter places its spans inside your trace. + + + + +In TypeScript, wrap `fetch` and pass it to the client (`new OpenAI({ baseURL, apiKey, fetch: openRouterFetch })` or `createOpenRouter({ apiKey, fetch: openRouterFetch })`): + +```ts +import { trace } from "@opentelemetry/api" + +export const openRouterFetch: typeof fetch = (input, init) => { + const span = trace.getActiveSpan()?.spanContext() + if (span && typeof init?.body === "string") { + const body = JSON.parse(init.body) + body.trace = { ...body.trace, trace_id: span.traceId, parent_span_id: span.spanId } + init = { ...init, body: JSON.stringify(body) } + } + return fetch(input, init) +} +``` + + + + +In Python, build the body next to `session_id`: + +```py +from opentelemetry import trace + + +def openrouter_extra_body(conversation_id: str) -> dict: + body = {"session_id": conversation_id} + ctx = trace.get_current_span().get_span_context() + if ctx.is_valid: + body["trace"] = { + "trace_id": format(ctx.trace_id, "032x"), + "parent_span_id": format(ctx.span_id, "016x"), + } + return body + + +client.chat.completions.create(model=model, messages=messages, extra_body=openrouter_extra_body(conversation_id)) +``` + + + + +Use the same value for `session_id` as your framework's conversation id. + +Broadcast has no tool calls or agent names. For those, trace your app with its [framework guide](/docs/agent-tracing) and nest Broadcast under it as shown here. + +## Check that it works + +Run a conversation of three turns with the same `session_id`. After about a minute, open **Agent Sessions** and filter by service `openrouter`. + +You should see one session named after your `session_id`, with vendor **OpenRouter**, an `LLM Generation` span per model call, and tokens and cost in USD. To keep content out of Maple, turn on **Privacy Mode** on the destination. + +## Troubleshooting + +- **Test Connection fails.** Use the full `https://ingest.maple.dev/v1/traces` URL and valid JSON headers. +- **Test Connection passes but nothing arrives.** Check the destination's API key filter and data regions against the key and endpoint your app uses, and that you're not using a placeholder key like `MAPLE_TEST`. +- **Every call is its own session, named `trace:`.** The request has no `session_id`. Log the outgoing body, since some wrappers drop unknown fields. +- **Tokens or LLM calls are about double.** Your app and Broadcast both record each call. Nest Broadcast with the `trace` field as shown above. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [OpenRouter Broadcast](https://openrouter.ai/docs/guides/features/broadcast) +- [Vercel AI SDK](/docs/agent-tracing/vercel-ai-sdk), if you call OpenRouter through `@openrouter/ai-sdk-provider` diff --git a/apps/landing/src/content/docs/agent-tracing/opentelemetry.mdx b/apps/landing/src/content/docs/agent-tracing/opentelemetry.mdx new file mode 100644 index 0000000000..94ead98ffb --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/opentelemetry.mdx @@ -0,0 +1,268 @@ +--- +title: "Trace any AI agent with the OpenTelemetry GenAI conventions" +description: "Write the OpenTelemetry GenAI spans Maple reads by hand, in any language, so a custom agent loop shows up in Agent Sessions." +group: "AI Agents" +order: 51 +navLabel: "Any language (OTel GenAI)" +icon: "opentelemetry" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +Use this guide when no other guide covers your agent, for example a hand-written agent loop. You write the agent spans yourself with any OpenTelemetry SDK: an `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. + +Tested with the OpenTelemetry JS SDK 2.11 on Node.js 26 and the Python SDK 1.45 on Python 3.14, against the [GenAI conventions](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-opentelemetry](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-opentelemetry) skill and follows it. + +```text +Set up Maple agent tracing for my hand-rolled agent in this project, using the OpenTelemetry GenAI conventions. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-opentelemetry -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## The spans and attributes Maple reads + +One user message produces one trace: + +```text +invoke_agent support gen_ai.conversation.id = chat_42 +├── chat openai/gpt-4o-mini model call: asks for get_weather +├── execute_tool get_weather tool call +└── chat openai/gpt-4o-mini model call: final answer +``` + +Set `gen_ai.operation.name` on every span. Agent Sessions ignores spans without it. + +| Span (kind) | Attribute | Value | +| --- | --- | --- | +| `invoke_agent` (`INTERNAL`) | `gen_ai.operation.name` | `invoke_agent` | +| | `gen_ai.agent.name` | `support` | +| | `gen_ai.conversation.id` | your chat or thread id, see [sessions](#group-turns-into-one-session) | +| | `gen_ai.input.messages`, `gen_ai.output.messages` | optional, JSON string | +| `chat` (`CLIENT`) | `gen_ai.operation.name` | `chat` | +| | `gen_ai.provider.name` | the API you called: `openai`, `anthropic`, `gcp.gemini`, `openrouter`... | +| | `gen_ai.request.model`, `gen_ai.response.model` | model ids | +| | `gen_ai.response.id` | the provider's response id | +| | `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` | int | +| | `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` | int, optional | +| | `gen_ai.usage.cost` | double in USD, optional | +| | `gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages` | JSON string | +| | `gen_ai.response.finish_reasons` | string array, e.g. `["stop"]` | +| | `gen_ai.response.time_to_first_chunk` | double, in **seconds**, streamed calls | +| `execute_tool` (`INTERNAL`) | `gen_ai.operation.name` | `execute_tool` | +| | `gen_ai.tool.name` | `get_weather` | +| | `gen_ai.tool.call.id` | the id the model gave the tool call | +| | `gen_ai.tool.call.arguments` | JSON string of an object | +| | `gen_ai.tool.call.result` | the tool's result: a string as is, anything else as a JSON string | + +Mark a failed span with status `ERROR` and an `error.type` attribute, even when you return a tool's error to the model as its result. + +Put token usage on `chat` spans only. Input tokens include cached tokens and output tokens include reasoning: for Anthropic, add the cache reads and writes to `input_tokens`; for Gemini, add `thoughtsTokenCount` to `candidatesTokenCount`. On OpenAI's streaming API, set `stream_options: { include_usage: true }`, or streamed calls report no tokens. Maple doesn't price tokens, so a session without `gen_ai.usage.cost` shows as unpriced. OpenRouter returns the cost in `usage.cost`. + +### The message format + +`gen_ai.input.messages` and `gen_ai.output.messages` are JSON arrays of `{role, parts}` messages, serialized to a string: + +```json +[ + { "role": "user", "parts": [{ "type": "text", "content": "What's the weather in Berlin?" }] }, + { + "role": "assistant", + "parts": [{ "type": "tool_call", "id": "call_1", "name": "get_weather", "arguments": { "city": "Berlin" } }] + }, + { "role": "tool", "parts": [{ "type": "tool_call_response", "id": "call_1", "response": "{\"temperature_c\":21}" }] } +] +``` + +Output messages add a `finish_reason` to each message. `gen_ai.system_instructions` is an array of parts without a role: `[{"type":"text","content":"You are a concise assistant."}]`. + +Set the message attributes as JSON strings. Maple doesn't read span events, logs or indexed keys like `gen_ai.prompt.0.content`. Leave `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` and `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` unset, because a truncated message array no longer parses. + +To keep prompts and results out of Maple, skip the content attributes (`gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result`). The transcript is then empty and everything else still shows up. + +## Export spans to Maple + +Point the OTLP exporter at Maple with the standard variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporters append `/v1/traces` themselves. + + + + +TypeScript (Node.js 20 or newer): + +```bash +npm install @opentelemetry/api @opentelemetry/sdk-trace-node @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources openai +``` + +```ts +// tracing.ts: import this first in every entry point +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { resourceFromAttributes } from "@opentelemetry/resources" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { NodeTracerProvider } from "@opentelemetry/sdk-trace-node" + +export const provider = new NodeTracerProvider({ + resource: resourceFromAttributes({ + "service.name": "support-agent", + "deployment.environment.name": "production", + }), + // Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS + spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())], +}) +// Registers the global provider and the async context manager, so spans nest across awaits +provider.register() +``` + + + + +Python (3.10 or newer): + +```bash +pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" openai +``` + +```py +# tracing.py: import this first in every entry point +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) +``` + + + + +If your app already has a `TracerProvider` (Sentry, Datadog, `opentelemetry-instrument`, `NodeSDK`), add the `BatchSpanProcessor` to it instead of creating a second one. + +Name the tracer after your app, like `support-agent`. A tracer named after a framework or gateway, such as `openrouter` or `langsmith`, makes Maple treat your spans as that framework's. + +## Instrument the agent loop + +Complete loops that stream, call tools and record every attribute above are in the skill: [TypeScript](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/typescript.md) and [Python](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/python.md). Wrap your own loop's existing calls the same way. + +This is the `execute_tool` span from the TypeScript version. The `chat` span follows the same pattern around each model call: + +```ts +async function runTool(agent: Agent, call: ToolCall) { + const name = call.function.name + return tracer.startActiveSpan( + `execute_tool ${name}`, + { + kind: SpanKind.INTERNAL, + attributes: { + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.type": "function", + "gen_ai.tool.call.id": call.id, + "gen_ai.tool.call.arguments": call.function.arguments || "{}", + }, + }, + async (span) => { + try { + const result = await agent.tools[name]!.run(JSON.parse(call.function.arguments || "{}")) + const output = typeof result === "string" ? result : json(result ?? null) + span.setAttribute("gen_ai.tool.call.result", output) + return output + } catch (error) { + markFailed(span, error) + return json({ error: error instanceof Error ? error.message : String(error) }) + } finally { + span.end() + } + }, + ) +} +``` + +Start the `chat` and `execute_tool` spans inside the `invoke_agent` span's callback, so they land in the same trace. For a sub-agent, run its loop inside the delegating tool's `execute_tool` span and give it a distinct `gen_ai.agent.name`. + +## Group turns into one session + +Set `gen_ai.conversation.id` on the `invoke_agent` span of every turn, using the id your app already has for the conversation. A chat backend passes it once per user message: + +```ts +// One history per conversation. Store it in your database in a real backend. +const histories = new Map() + +export async function handleMessage(chatId: string, text: string, onText?: (delta: string) => void) { + const history = histories.get(chatId) ?? [] + histories.set(chatId, history) + history.push({ role: "user", content: text }) + return runAgent(assistant, history, { conversationId: chatId, onText }) +} +``` + +Without the id, or with one generated per request, each trace becomes its own one-turn session named `trace:`. Don't give sub-agents their own id; they inherit the session from the trace. + +If a framework writes its session id under a key Maple doesn't read, wrap each turn in your own span that carries `maple_ai.session.id`, and run the framework inside it: + +```ts +await tracer.startActiveSpan( + "invoke_agent support", + { + attributes: { + "gen_ai.operation.name": "invoke_agent", + "gen_ai.agent.name": "support", + "maple_ai.session.id": chatId, + }, + }, + async (span) => { + try { + return await frameworkAgent.run(message) // the framework's spans nest under this one + } finally { + span.end() + } + }, +) +``` + +Put `maple_ai.session.id` only on your own wrapper span, never on the framework's spans. + +## Flush before a short-lived process exits + +`BatchSpanProcessor` exports every few seconds, so a script, CLI or Lambda can exit before the last batch is sent. At the end of a script, call `await provider.shutdown()` (TypeScript) or `provider.shutdown()` (Python). In a serverless handler, call `provider.forceFlush()` (`force_flush()` in Python) before returning. + +## Check that it works + +Run one conversation with two messages and a tool call, then open **Agent Sessions** in Maple. After a few seconds you should see one session with your conversation id, one turn per message with its transcript, and `chat` and `execute_tool` spans nested under each turn's `invoke_agent` span. Hand-written spans show the framework as **Unidentified**. + +## Troubleshooting + +- **Nothing shows up in Agent Sessions, but the trace is in Traces.** No span has `gen_ai.operation.name`. Add it to every span. +- **Every message is its own session.** `gen_ai.conversation.id` is missing or changes per request. Pass the conversation's id on each turn's `invoke_agent` span. +- **The transcript is empty, but tokens are there.** The messages are plain text, in span events, or cut by an attribute length limit. Set them as JSON strings and unset `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT`. +- **Model and tool spans are separate traces.** The spans weren't started inside the `invoke_agent` span, the Node provider wasn't registered with `provider.register()`, or the work ran in a new Python thread (pass the context with `contextvars.copy_context().run(...)`). +- **Every model call appears twice.** A provider auto-instrumentation (OpenAI, Anthropic, OpenLLMetry, OpenInference) is also active. Keep your `chat` spans or the instrumentation, not both. See [provider SDKs](/docs/agent-tracing/provider-sdks). + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [All agent tracing guides](/docs/agent-tracing) +- [Provider SDKs](/docs/agent-tracing/provider-sdks), for auto-instrumented OpenAI, Anthropic and Gemini clients +- [GenAI spans](https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-spans.md) and [GenAI agent spans](https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-agent-spans.md) diff --git a/apps/landing/src/content/docs/agent-tracing/provider-sdks.mdx b/apps/landing/src/content/docs/agent-tracing/provider-sdks.mdx new file mode 100644 index 0000000000..3ee479319c --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/provider-sdks.mdx @@ -0,0 +1,375 @@ +--- +title: "Trace agents built on the OpenAI, Anthropic and Gemini SDKs" +description: "Trace your own agent loop on the OpenAI, Anthropic or Google Gen AI SDK so each conversation becomes one Maple Agent Session, in Python or TypeScript." +group: "AI Agents" +order: 50 +navLabel: "OpenAI, Anthropic & Gemini SDKs" +icon: "openai" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +Use this guide when your agent is your own loop around `client.chat.completions.create`, `client.messages.create` or `client.models.generate_content`. An instrumentation library records each model call, and you add the turn and tool spans and put the conversation id on the turn span. If you use an agent framework on top of these SDKs, use [that framework's guide](/docs/agent-tracing) instead. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-provider-sdks](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-provider-sdks) skill and follows it. + +```text +Set up Maple agent tracing for the OpenAI, Anthropic or Gemini SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-provider-sdks -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Configure the exporter + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT="SPAN_ONLY" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. `SPAN_ONLY` records prompts and replies for the transcript. Leave it unset to keep content out of Maple. `EVENT_ONLY` or `true` leaves the transcript empty. + +## Set up tracing + + + + +Install the `genai` packages below, not `opentelemetry-instrumentation-openai` or `opentelemetry-instrumentation-openai-v2`. + +```bash +pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ + "opentelemetry-instrumentation-genai-openai>=1.2b0" +# Anthropic: opentelemetry-instrumentation-genai-anthropic>=1.2b0 +# Gemini: opentelemetry-instrumentation-google-genai>=1.2b0 +``` + +```py +# tracing.py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.instrumentation.genai.openai import OpenAIInstrumentor +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider(resource=Resource.create({"service.name": "support-agent"})) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +OpenAIInstrumentor().instrument() +# from opentelemetry.instrumentation.genai.anthropic import AnthropicInstrumentor +# from opentelemetry.instrumentation.google_genai import GoogleGenAiSdkInstrumentor +``` + +Import `tracing` first in your entry point. If your app already has a `TracerProvider` (Sentry, Logfire, Datadog), add the `BatchSpanProcessor` to it instead. + + + + +No OpenTelemetry instrumentation works for `openai` 7 in TypeScript, so this helper records the turn, each tool call and each model call. + +```bash +npm install @opentelemetry/api @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-proto +``` + +```ts +// instrumentation.ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { NodeSDK, tracing } from "@opentelemetry/sdk-node" + +export const spanProcessor = new tracing.BatchSpanProcessor(new OTLPTraceExporter()) + +export const sdk = new NodeSDK({ serviceName: "support-agent", spanProcessors: [spanProcessor] }) +sdk.start() +``` + +```ts +// agent-tracing.ts +import { type Attributes, type Span, SpanKind, SpanStatusCode, trace } from "@opentelemetry/api" +import type OpenAI from "openai" + +const tracer = trace.getTracer("support-agent") + +const captureContent = ["SPAN_ONLY", "SPAN_AND_EVENT"].includes( + (process.env.OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT ?? "").toUpperCase(), +) + +export function agentSpan(agentName: string, conversationId: string | undefined, fn: () => Promise) { + const attributes: Attributes = { "gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agentName } + if (conversationId) attributes["gen_ai.conversation.id"] = conversationId + return withSpan(`invoke_agent ${agentName}`, SpanKind.INTERNAL, attributes, () => fn()) +} + +export function runTool(callId: string, name: string, args: string, tool: (args: any) => unknown) { + const attributes = { "gen_ai.operation.name": "execute_tool", "gen_ai.tool.name": name, "gen_ai.tool.call.id": callId } + return withSpan(`execute_tool ${name}`, SpanKind.INTERNAL, attributes, async (span) => { + if (captureContent) span.setAttribute("gen_ai.tool.call.arguments", args) + let result: string + try { + result = JSON.stringify(await tool(JSON.parse(args))) + } catch (error) { + markFailed(span, error) + result = JSON.stringify({ error: String(error) }) + } + if (captureContent) span.setAttribute("gen_ai.tool.call.result", result) + return result + }) +} + +type ChatParams = Omit + +/** One model call. Pass `onText` to stream; usage still arrives. */ +export function tracedChat(client: OpenAI, params: ChatParams, onText?: (delta: string) => void) { + const attributes = { "gen_ai.operation.name": "chat", "gen_ai.provider.name": "openai", "gen_ai.request.model": params.model } + return withSpan(`chat ${params.model}`, SpanKind.CLIENT, attributes, async (span) => { + if (captureContent) { + span.setAttribute("gen_ai.input.messages", JSON.stringify(params.messages.map(toGenAiMessage))) + } + let completion: OpenAI.Chat.ChatCompletion + if (onText) { + const started = performance.now() + let firstChunkAt: number | undefined + // Without include_usage, a streamed call reports no tokens at all. + const stream = client.chat.completions.stream({ ...params, stream_options: { include_usage: true } }) + stream.on("content", (delta) => { + firstChunkAt ??= performance.now() + onText(delta) + }) + completion = await stream.finalChatCompletion() + if (firstChunkAt !== undefined) { + span.setAttribute("gen_ai.response.time_to_first_chunk", (firstChunkAt - started) / 1000) + } + } else { + completion = await client.chat.completions.create(params) + } + span.setAttributes({ + "gen_ai.response.id": completion.id, + "gen_ai.response.model": completion.model, + "gen_ai.response.finish_reasons": completion.choices.map((c) => c.finish_reason), + }) + if (completion.usage) { + span.setAttributes({ + "gen_ai.usage.input_tokens": completion.usage.prompt_tokens, + "gen_ai.usage.output_tokens": completion.usage.completion_tokens, + "gen_ai.usage.cache_read.input_tokens": completion.usage.prompt_tokens_details?.cached_tokens ?? 0, + "gen_ai.usage.reasoning.output_tokens": completion.usage.completion_tokens_details?.reasoning_tokens ?? 0, + }) + // OpenRouter adds the call's price in USD to usage. Other providers don't send one. + const cost = (completion.usage as { cost?: number }).cost + if (cost !== undefined) span.setAttribute("gen_ai.usage.cost", cost) + } + if (captureContent) { + const output = completion.choices.map((c) => ({ ...toGenAiMessage(c.message), finish_reason: c.finish_reason })) + span.setAttribute("gen_ai.output.messages", JSON.stringify(output)) + } + return completion + }) +} + +// Converts OpenAI messages to Maple's transcript format. Text and tool calls only. +function toGenAiMessage(message: OpenAI.Chat.ChatCompletionMessageParam | OpenAI.Chat.ChatCompletionMessage) { + if (message.role === "tool") { + return { role: "tool", parts: [{ type: "tool_call_response", id: message.tool_call_id, response: message.content }] } + } + const parts: object[] = [] + if (typeof message.content === "string" && message.content) parts.push({ type: "text", content: message.content }) + if (message.role === "assistant") { + for (const call of message.tool_calls ?? []) { + if (call.type === "function") { + parts.push({ type: "tool_call", id: call.id, name: call.function.name, arguments: call.function.arguments }) + } + } + } + return { role: message.role, parts } +} + +function withSpan(name: string, kind: SpanKind, attributes: Attributes, fn: (span: Span) => Promise) { + return tracer.startActiveSpan(name, { kind, attributes }, async (span) => { + try { + return await fn(span) + } catch (error) { + markFailed(span, error) + throw error + } finally { + span.end() + } + }) +} + +function markFailed(span: Span, error: unknown) { + const err = error instanceof Error ? error : new Error(String(error)) + span.recordException(err) + span.setStatus({ code: SpanStatusCode.ERROR, message: err.message }) + span.setAttribute("error.type", err.name) +} +``` + + + + +## Wrap each turn and tool call + + + + +Wrap each turn in an `invoke_agent` span that carries `gen_ai.conversation.id`. Use the id your app already stores for the conversation (thread id, ticket id), not a new UUID per request. + +```py +# agent_tracing.py +import json +import os +from contextlib import contextmanager + +from opentelemetry import trace +from opentelemetry.trace import Status, StatusCode + +tracer = trace.get_tracer("support-agent") + +CAPTURE_CONTENT = os.environ.get( + "OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT", "" +).upper() in ("SPAN_ONLY", "SPAN_AND_EVENT") + + +@contextmanager +def agent_span(agent_name: str, conversation_id: str | None = None): + attributes = {"gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agent_name} + if conversation_id: + attributes["gen_ai.conversation.id"] = conversation_id + with tracer.start_as_current_span(f"invoke_agent {agent_name}", attributes=attributes) as span: + yield span + + +def run_tool(call_id: str, name: str, arguments: str, tool) -> str: + with tracer.start_as_current_span( + f"execute_tool {name}", + attributes={ + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.call.id": call_id, + }, + ) as span: + if CAPTURE_CONTENT: + span.set_attribute("gen_ai.tool.call.arguments", arguments) + try: + result = json.dumps(tool(**json.loads(arguments))) + except Exception as exc: + span.record_exception(exc) + span.set_status(Status(StatusCode.ERROR, str(exc))) + span.set_attribute("error.type", type(exc).__qualname__) + result = json.dumps({"error": str(exc)}) + if CAPTURE_CONTENT: + span.set_attribute("gen_ai.tool.call.result", result) + return result +``` + +In your loop, the tracing is the `with agent_span(...)` line and the `run_tool(...)` call: + +```py +# agent.py +import tracing # noqa: F401 (first import) +from openai import OpenAI + +from agent_tracing import agent_span, run_tool + +client = OpenAI() +MODEL = "gpt-4o-mini" +# get_weather, fetch_transport_data and TOOL_SCHEMAS are your own tools and their JSON schemas. +TOOLS = {"get_weather": get_weather, "fetch_transport_data": fetch_transport_data} + + +def chat_turn(conversation_id: str, history: list, user_text: str) -> str: + with agent_span("support_agent", conversation_id): + history.append({"role": "user", "content": user_text}) + while True: + response = client.chat.completions.create(model=MODEL, messages=history, tools=TOOL_SCHEMAS) + message = response.choices[0].message + history.append(message.model_dump(include={"role", "content", "tool_calls"}, exclude_none=True)) + if not message.tool_calls: + return message.content + for call in message.tool_calls: + result = run_tool(call.id, call.function.name, call.function.arguments, TOOLS[call.function.name]) + history.append({"role": "tool", "tool_call_id": call.id, "content": result}) +``` + +With Anthropic, run each `tool_use` block with `run_tool(block.id, block.name, json.dumps(block.input), ...)`. With Gemini's automatic function calling, the instrumentation records the tool spans itself, so wrap the turn in `agent_span` and skip `run_tool`. + + + + +Pass the conversation id your app already stores to `agentSpan`, and route model and tool calls through `tracedChat` and `runTool`: + +```ts +// agent.ts +import "./instrumentation.ts" +import OpenAI from "openai" +import { agentSpan, runTool, tracedChat } from "./agent-tracing.ts" + +const client = new OpenAI() +const model = "gpt-4o-mini" +// tools: Record unknown> and toolSchemas: OpenAI.Chat.ChatCompletionTool[] are your own. + +export function chatTurn( + conversationId: string, + history: OpenAI.Chat.ChatCompletionMessageParam[], + userText: string, + onText?: (delta: string) => void, +) { + return agentSpan("support_agent", conversationId, async () => { + history.push({ role: "user", content: userText }) + while (true) { + const completion = await tracedChat(client, { model, messages: history, tools: toolSchemas }, onText) + const message = completion.choices[0].message + history.push(message) + if (!message.tool_calls?.length) return message.content ?? "" + for (const call of message.tool_calls) { + if (call.type !== "function") continue + const result = await runTool(call.id, call.function.name, call.function.arguments, tools[call.function.name]) + history.push({ role: "tool", tool_call_id: call.id, content: result }) + } + } + }) +} +``` + +For `@anthropic-ai/sdk` or `@google/genai`, copy `tracedChat` and map that SDK's response fields. The [skill's TypeScript reference](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-provider-sdks/references/typescript.md) has the attribute table. + + + + +## Sub-agents, streaming and cost + +Run a sub-agent inside `run_tool` and wrap its loop in `agent_span` with its own name and no conversation id. + +When you stream OpenAI Chat Completions in Python, pass `stream_options={"include_usage": True}`, or the call shows 0 tokens. The TypeScript helper sets it for you. + +Python sessions show as unpriced. The TypeScript helper records cost only when you call through OpenRouter. + +## Flush before a short-lived process exits + +In Python, a serverless handler or notebook should call `provider.force_flush()` after each turn. In TypeScript, call `await sdk.shutdown()` before a script exits, or `await spanProcessor.forceFlush()` before a serverless handler returns. + +## Check that it works + +Run a conversation of two or three messages with a tool call, flush, and open **Agent Sessions** in Maple. You should see one session with your conversation id, one turn per user message, and a transcript with the replies and tool calls. The framework is **Unidentified** for this setup. + +## Troubleshooting + +- **Every model call is its own session.** The call ran outside `agent_span`. Wrap the whole turn, including the tool loop. +- **One session per turn.** The conversation id changes per request. Pass the id stored with the conversation. +- **The transcript is empty.** Set `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` in the process that makes the calls. +- **No model spans in Python.** `tracing` wasn't imported first, or you installed `opentelemetry-instrumentation-openai` instead of `opentelemetry-instrumentation-genai-openai`. +- **Every model call appears twice.** A second instrumentation (OpenLLMetry, OpenInference, `logfire.instrument_openai()`, Sentry's OpenAI integration) wraps the same SDK. Uninstall the extra one. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [OpenRouter](/docs/agent-tracing/openrouter), if your calls go through OpenRouter +- [OpenTelemetry GenAI instrumentations for Python](https://github.com/open-telemetry/opentelemetry-python-genai) diff --git a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md new file mode 100644 index 0000000000..bb183a4cc9 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md @@ -0,0 +1,171 @@ +--- +title: "Trace Pydantic AI agents with OpenTelemetry" +description: "Send Pydantic AI's built-in OpenTelemetry spans to Maple so each conversation shows up as one Agent Session." +group: "AI Agents" +order: 20 +navLabel: "Pydantic AI" +icon: "pydantic" +--- + +Pydantic AI already emits OpenTelemetry spans for runs, model calls and tool calls. You export them to Maple and pass a conversation id on every run. Without the id, each message becomes its own session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-pydantic-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-pydantic-ai) skill and follows it. + +```text +Set up Maple agent tracing for Pydantic AI in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-pydantic-ai -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Export spans to Maple + +Install the OpenTelemetry SDK and the OTLP/HTTP exporter next to Pydantic AI. Swap `[openai]` for the extras of the providers you use. + +```bash +pip install "pydantic-ai-slim[openai]>=2.51" "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Point the exporter at Maple. For an EU organization, use `https://ingest.eu.maple.dev`. + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +Set up tracing once, when your process starts: + +```py +# tracing.py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor +from pydantic_ai import Agent, InstrumentationSettings + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +Agent.instrument_all( + InstrumentationSettings( + tracer_provider=provider, + include_content=True, + include_binary_content=False, + ) +) +``` + +Import `tracing` at the top of your entry point (`main.py`, the FastAPI app module, the worker), before the first `agent.run()`. + +If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Sentry or your own setup), add the `BatchSpanProcessor` to it instead of creating a second one, and call `Agent.instrument_all(InstrumentationSettings(include_content=True, include_binary_content=False))` without `tracer_provider`. + +### If you already use Logfire + +Skip the `pip install` above, because it conflicts with Logfire's OpenTelemetry pins. Keep the three environment variables. Logfire exports to Maple whenever `OTEL_EXPORTER_OTLP_ENDPOINT` is set. + +```py +import logfire + +logfire.configure(service_name="support-agent", environment="production", send_to_logfire=False) +logfire.instrument_pydantic_ai() +``` + +Logfire scrubs tool arguments and results by default. If they arrive as `[Scrubbed due to ...]`, see Troubleshooting. + +## Pass the conversation id on every run + +Pass the chat or thread id your app already has as `conversation_id=` on every `run()`, `run_stream()` and `iter()`: + +```py +from pydantic_ai import Agent + +support = Agent("openai:gpt-4o-mini", name="support") + + +async def handle_message(chat_id: str, text: str, history: list) -> str: + result = await support.run(text, conversation_id=chat_id, message_history=history) + return result.output +``` + +The id must stay the same for the whole conversation and differ between conversations. + +When streaming, keep the `async with` block open until the stream is finished. The run's span ends when the block exits. + +```py +async def stream_reply(chat_id: str, text: str, history: list): + async with support.run_stream(text, conversation_id=chat_id, message_history=history) as run: + async for delta in run.stream_text(delta=True): + yield delta +``` + +### Sub-agents + +When a tool runs another agent, pass `conversation_id` and `usage` from the tool's context. Give every agent a `name=` so each one gets its own lane. + +```py +from pydantic_ai import Agent, RunContext + +weather_worker = Agent("openai:gpt-4o-mini", name="weather_worker", tools=[get_weather]) +orchestrator = Agent("openai:gpt-4o-mini", name="orchestrator") + + +@orchestrator.tool +async def research_weather(ctx: RunContext[None], city: str) -> str: + """Delegate to the weather worker.""" + result = await weather_worker.run( + f"What is the current weather in {city}?", + usage=ctx.usage, + conversation_id=ctx.conversation_id, + ) + return result.output +``` + +## Flush short-lived processes + +A Lambda, a killed worker or a notebook kernel doesn't exit normally, so flush explicitly: + +```py +import asyncio + +from opentelemetry import trace + + +def handler(event, context): + try: + return asyncio.run(handle_message(event["chat_id"], event["text"], [])) + finally: + trace.get_tracer_provider().force_flush() +``` + +In a script or CLI, call `provider.shutdown()` at the end. With Logfire, use `logfire.force_flush()` or `logfire.shutdown()`. + +## Check that it works + +If Pydantic AI prints an `observability: off` banner on the first run, `tracing.py` didn't run before your first `agent.run()`. + +Run a conversation with two messages and a tool call, then open **Agent Sessions**. You should see one session with your conversation id and framework **Pydantic AI**, one turn per `run()`, a transcript with prompts, replies and tool calls, and token counts and cost on every model call. + +## Troubleshooting + +- **Every message is its own session.** Pass `conversation_id=` on every `run()`, `run_stream()` and `iter()`. +- **A multi-agent run is split into several turns or sessions.** Pass `conversation_id=ctx.conversation_id` to every nested `run()`. +- **Tool arguments or results read `[Scrubbed due to ...]`.** Logfire's scrubbing matched a word like `session` or `auth`. Pass `scrubbing=logfire.ScrubbingOptions(callback=...)` that keeps `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, or `scrubbing=False`. +- **A failed tool shows as successful.** The tool returned an error value. Raise `ToolFailed("...")` so the call is marked failed and the model still sees the message. +- **Spans show up twice.** Another instrumentor (Logfire's `instrument_openai()`, OpenInference, OpenLLMetry) also traces the model client. Remove it and keep Pydantic AI's. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Instrument a Python application](/docs/guides/instrumentation-python) diff --git a/apps/landing/src/content/docs/agent-tracing/smolagents.md b/apps/landing/src/content/docs/agent-tracing/smolagents.md new file mode 100644 index 0000000000..3b524b5253 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/smolagents.md @@ -0,0 +1,148 @@ +--- +title: "Trace Hugging Face smolagents with OpenTelemetry" +description: "Send smolagents runs to Maple through the OpenInference instrumentor so each conversation shows up as one Agent Session." +group: "AI Agents" +order: 25 +navLabel: "smolagents" +icon: "huggingface" +--- + +smolagents is traced with OpenInference's `openinference-instrumentation-smolagents`. Turn on its GenAI attributes and wrap each run in `using_session(...)`. Without `using_session(...)`, every message becomes its own session. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-smolagents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-smolagents) skill and follows it. + +```text +Set up Maple agent tracing for smolagents in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-smolagents -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the instrumentor and export to Maple + +```bash +pip install "smolagents[openai]>=1.26" "openinference-instrumentation-smolagents>=0.1.40" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Use `[litellm]` instead of `[openai]` if you run `LiteLLMModel`. Skip the `smolagents[telemetry]` extra, which installs the Arize Phoenix server. + +Point the exporter at Maple. For an EU organization, use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +Add a `tracing.py` and import it at the top of your entry point, before the first `agent.run()`: + +```py +# tracing.py +import json + +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.smolagents import SmolagentsInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class SmolagentsForMaple(SpanProcessor): + """Fixes agent names, tool names and tool arguments for Maple.""" + + def on_start(self, span, parent_context=None): + if span.instrumentation_scope.name != "openinference.instrumentation.smolagents": + return + attrs = span.attributes + if span.name.endswith(".run"): + span.set_attribute("gen_ai.agent.name", span.name.removesuffix(".run")) + elif "tool.name" in attrs: + span.update_name(f"execute_tool {attrs['tool.name']}") + if attrs.get("input.value", "").startswith("{"): + call = json.loads(attrs["input.value"]) + span.set_attribute("gen_ai.tool.call.arguments", json.dumps(call["kwargs"] or call["args"])) + + +provider = TracerProvider() +provider.add_span_processor(SmolagentsForMaple()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +SmolagentsInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +Copy `SmolagentsForMaple` as is. It gives each agent its own lane and fixes tool names and arguments. + +If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add `SmolagentsForMaple()` and the exporter to that provider and pass it to `instrument()` instead of creating a second one. + +## Wrap each run in the conversation id + +Wrap every `agent.run()` in OpenInference's `using_session` with the id your app already stores the chat under: + +```py +from openinference.instrumentation import using_session +from smolagents import OpenAIServerModel, ToolCallingAgent + +# One agent per conversation. In a long-running server, evict idle ones or rebuild them from stored history. +agents: dict[str, ToolCallingAgent] = {} + + +def handle_message(conversation_id: str, text: str) -> str: + agent = agents.get(conversation_id) + if agent is None: + agent = agents[conversation_id] = ToolCallingAgent( + tools=[get_weather, calculate], + model=OpenAIServerModel(model_id="gpt-4o-mini"), + name="assistant", + ) + with using_session(conversation_id): + return str(agent.run(text, reset=False)) +``` + +The id must stay the same across the conversation and differ between conversations. + +Keep one agent object per conversation, as above. A single shared agent with `reset=False` mixes every user's memory into one conversation. + +Give every agent, including managed agents, a `name`. Unnamed agents share one lane. + +## Flush before the process exits + +Serverless functions, killed workers and notebooks don't exit normally, so flush after each run: + +```py +from tracing import provider + +try: + handle_message("conv-42", "What's the weather in Berlin?") +finally: + provider.force_flush() +``` + +## Check that it works + +Run two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions**. You should see one session with framework **smolagents**, one turn per `agent.run()`, a transcript, `OpenAIModel.generate` model calls with token counts, and `execute_tool ` tool calls. Each run ends with an `execute_tool final_answer` call. + +Cost shows as unpriced. + +## Troubleshooting + +- **One session per message.** Wrap every `agent.run()` in `using_session(...)` with the stored conversation id. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use the environment variable with the base URL, or pass the full path. +- **Every model call appears twice.** Remove `openinference-instrumentation-openai` or `openinference-instrumentation-litellm`; the smolagents instrumentor already covers model calls. +- **No tool spans with `CodeAgent`.** A remote executor (`e2b`, `docker`, `modal`) runs tools outside your process. Only the local executor produces tool spans. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter), if your models go through either gateway diff --git a/apps/landing/src/content/docs/agent-tracing/spring-ai.md b/apps/landing/src/content/docs/agent-tracing/spring-ai.md new file mode 100644 index 0000000000..f1bcae6d35 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/spring-ai.md @@ -0,0 +1,252 @@ +--- +title: "Trace Spring AI agents with OpenTelemetry" +description: "Send Spring AI ChatClient, model and tool spans to Maple, with one session per chat memory conversation." +group: "AI Agents" +order: 31 +navLabel: "Spring AI" +icon: "spring" +--- + +Spring AI already emits spans for `ChatClient`, model and tool calls, with token counts. You add Spring Boot's OpenTelemetry starter, set sampling to 100%, add one configuration class for the transcript, tool names and failed tools, and pass the conversation id on every call. + +Tested with Spring AI 2.0.1 on Spring Boot 4.1.1 and Java 21. Spring AI 1.1 on Boot 3.5 works too, with different dependencies and property names listed in the [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai). + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-spring-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai) skill and follows it. + +```text +Set up Maple agent tracing for Spring AI in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-spring-ai -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the OpenTelemetry starter + +Next to the `spring-ai-bom` import (2.0.1) and your model starter, such as `spring-ai-starter-model-openai`, add Boot's OpenTelemetry starter: + +```xml + + org.springframework.boot + spring-boot-starter-opentelemetry + +``` + +## Export to Maple + +Add to `application.properties`: + +```properties +spring.application.name=support-agent + +management.opentelemetry.tracing.export.otlp.endpoint=https://ingest.maple.dev/v1/traces +management.opentelemetry.tracing.export.otlp.headers.Authorization=Bearer ${MAPLE_INGEST_KEY:} +management.tracing.sampling.probability=1.0 +management.opentelemetry.resource-attributes.deployment.environment.name=production + +# The starter also exports metrics, to localhost:4318 unless told otherwise +management.otlp.metrics.export.url=https://ingest.maple.dev/v1/metrics +management.otlp.metrics.export.headers.Authorization=Bearer ${MAPLE_INGEST_KEY:} + +# Read by the configuration class below +maple.ai.capture-content=true +``` + +The empty default in `${MAPLE_INGEST_KEY:}` keeps the app starting when the key is missing. To avoid sending requests without a key, turn export off at the top of `main` in that case: + +```java +public static void main(String[] args) { + // A missing key disables export; it never stops the app. + var mapleKey = System.getenv("MAPLE_INGEST_KEY"); + if (mapleKey == null || mapleKey.isEmpty()) { + System.err.println("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled"); + System.setProperty("management.tracing.export.enabled", "false"); + System.setProperty("management.otlp.metrics.export.enabled", "false"); + } + SpringApplication.run(Application.class, args); +} +``` + +Spring Boot doesn't read `.env` files, so set `MAPLE_INGEST_KEY` in the environment that starts the JVM. + +For an EU organization, use `https://ingest.eu.maple.dev`. This property takes the full URL, so keep `/v1/traces` on the end. To skip metrics, replace the two metrics lines with `management.otlp.metrics.export.enabled=false`. + +Keep `management.tracing.sampling.probability=1.0`. Boot's default of `0.1` silently drops nine turns out of ten. + +## Add the attributes Maple reads + +Add this class in a package your application scans: + +```java +package com.example.agent; + +import java.util.ArrayList; +import java.util.List; +import java.util.Map; + +import io.micrometer.common.KeyValue; +import io.micrometer.observation.Observation; +import io.micrometer.observation.ObservationFilter; +import io.micrometer.observation.ObservationRegistry; +import tools.jackson.databind.json.JsonMapper; + +import org.springframework.ai.chat.client.observation.ChatClientObservationContext; +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.messages.Message; +import org.springframework.ai.chat.messages.ToolResponseMessage; +import org.springframework.ai.chat.observation.ChatModelObservationContext; +import org.springframework.ai.tool.execution.DefaultToolExecutionExceptionProcessor; +import org.springframework.ai.tool.execution.ToolExecutionExceptionProcessor; +import org.springframework.ai.tool.observation.ToolCallingObservationContext; +import org.springframework.beans.factory.annotation.Value; +import org.springframework.context.annotation.Bean; +import org.springframework.context.annotation.Configuration; + +@Configuration(proxyBeanMethods = false) +public class MapleAiObservationConfig { + + /** Agent name for a ChatClient: `.defaultAdvisors(a -> a.param(AGENT_NAME, "support_agent"))`. */ + public static final String AGENT_NAME = "gen_ai.agent.name"; + + @Bean + ObservationFilter mapleGenAiAttributes(@Value("${maple.ai.capture-content:false}") boolean captureContent) { + return context -> { + if (context instanceof ChatClientObservationContext client) { + client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.operation.name", "invoke_agent")); + if (client.getRequest().context().get(AGENT_NAME) instanceof String agent) { + client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.agent.name", agent)); + } + } + else if (context instanceof ToolCallingObservationContext tool) { + tool.addLowCardinalityKeyValue(KeyValue.of("gen_ai.tool.name", tool.getToolDefinition().name())); + if (tool.getToolCallId() != null) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.id", tool.getToolCallId())); + } + if (captureContent) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.arguments", tool.getToolCallArguments())); + if (tool.getToolCallResult() != null) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.result", tool.getToolCallResult())); + } + } + } + else if (captureContent && context instanceof ChatModelObservationContext chat) { + chat.addHighCardinalityKeyValue(KeyValue.of("gen_ai.input.messages", + JsonMapper.shared().writeValueAsString(chat.getRequest().getInstructions().stream().map(MapleAiObservationConfig::message).toList()))); + if (chat.getResponse() != null) { + chat.addHighCardinalityKeyValue(KeyValue.of("gen_ai.output.messages", + JsonMapper.shared().writeValueAsString(chat.getResponse().getResults().stream().map(g -> message(g.getOutput())).toList()))); + } + } + return context; + }; + } + + // Spring AI hands a failing tool's message back to the model and ends the tool span OK. + // Mark the span failed first, then keep the default behaviour. + @Bean + ToolExecutionExceptionProcessor toolExecutionExceptionProcessor(ObservationRegistry registry) { + ToolExecutionExceptionProcessor fallback = DefaultToolExecutionExceptionProcessor.builder().build(); + return exception -> { + Observation toolCall = registry.getCurrentObservation(); + if (toolCall != null) { + toolCall.error(exception); + } + return fallback.process(exception); + }; + } + + /** One message in the OpenTelemetry GenAI shape: {role, parts: [...]}. */ + private static Map message(Message message) { + List> parts = new ArrayList<>(); + if (message instanceof ToolResponseMessage toolResponse) { + toolResponse.getResponses().forEach(r -> parts.add( + Map.of("type", "tool_call_response", "id", r.id(), "response", r.responseData()))); + } + else { + if (message.getText() != null && !message.getText().isEmpty()) { + parts.add(Map.of("type", "text", "content", message.getText())); + } + if (message instanceof AssistantMessage assistant) { + assistant.getToolCalls().forEach(call -> parts.add( + Map.of("type", "tool_call", "id", call.id(), "name", call.name(), "arguments", call.arguments()))); + } + } + return Map.of("role", message.getMessageType().getValue(), "parts", parts); + } + +} +``` + +If your app already defines a `ToolExecutionExceptionProcessor`, add the `error()` call to it instead of adding a second bean. + +The transcript comes from `maple.ai.capture-content=true`. Spring AI's own `log-prompt` and `log-completion` settings only write to the application log. Set it to `false` to keep message and tool content out of Maple; everything else still shows up, with an empty transcript. A failed tool's exception message is still sent. + +## Group turns into one session + +Pass the `ChatMemory.CONVERSATION_ID` advisor parameter on every request. Spring AI writes it to the `spring.ai.chat.client.conversation.id` attribute, which Maple groups sessions by: + +```java +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.client.advisor.MessageChatMemoryAdvisor; +import org.springframework.ai.chat.memory.ChatMemory; +import org.springframework.stereotype.Service; + +@Service +public class ChatService { + + private final ChatClient chatClient; + + public ChatService(ChatClient.Builder builder, ChatMemory chatMemory, SupportTools tools) { + this.chatClient = builder + .defaultSystem("You are a concise support assistant.") + .defaultTools(tools) + .defaultAdvisors(MessageChatMemoryAdvisor.builder(chatMemory).build()) + .defaultAdvisors(a -> a.param(MapleAiObservationConfig.AGENT_NAME, "support_agent")) + .build(); + } + + public String reply(String conversationId, String userMessage) { + return chatClient.prompt() + .user(userMessage) + .advisors(a -> a.param(ChatMemory.CONVERSATION_ID, conversationId)) + .call() + .content(); + } + +} +``` + +Use the conversation id your app already stores. Set it on the request, as above, and never with `defaultAdvisors` on the builder, which puts every user in one session. The parameter works without a chat memory advisor. Sub-agent calls inside tools don't need it. + +## Exit command-line apps explicitly + +A web app needs nothing extra: Boot flushes pending spans on shutdown. In a `CommandLineRunner` app, exit explicitly, or the OpenAI client's threads hold the JVM and the unsent spans for about 60 seconds: + +```java +public static void main(String[] args) { + System.exit(SpringApplication.exit(SpringApplication.run(Application.class, args))); +} +``` + +On serverless platforms, inject `SdkTracerProvider` and call `tracerProvider.forceFlush().join(10, TimeUnit.SECONDS)` at the end of each invocation. + +## Check that it works + +Send two or three messages with the same conversation id, including one that calls a tool. After about a minute, **Agent Sessions** in Maple shows one session for that id with framework **Spring AI**, one turn per `ChatClient` call, the transcript, and tokens on every model call. Cost shows as unpriced, because Spring AI doesn't report it. + +## Troubleshooting + +- **Only some turns arrive.** Sampling is at Boot's default of 10%. Set `management.tracing.sampling.probability=1.0`. +- **Every message is its own session.** Pass `.advisors(a -> a.param(ChatMemory.CONVERSATION_ID, id))` on every `prompt()` call. +- **The transcript is empty.** `maple.ai.capture-content` isn't `true`, or `MapleAiObservationConfig` isn't in a scanned package. +- **Every span is its own trace.** The OpenTelemetry Java agent is attached next to the starter. Remove it, or follow the Java agent steps in the [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai). + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Spring AI observability reference](https://docs.spring.io/spring-ai/reference/observability/index.html) +- [Spring Boot tracing reference](https://docs.spring.io/spring-boot/reference/actuator/tracing.html) diff --git a/apps/landing/src/content/docs/agent-tracing/strands.mdx b/apps/landing/src/content/docs/agent-tracing/strands.mdx new file mode 100644 index 0000000000..db7a570991 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/strands.mdx @@ -0,0 +1,214 @@ +--- +title: "Trace Strands Agents with OpenTelemetry" +description: "Send Strands Agents traces to Maple with the full transcript and one Agent Session per conversation." +group: "AI Agents" +order: 24 +navLabel: "Strands Agents" +icon: "strands" +--- + +import LanguageTabs from "../../../components/docs/LanguageTabs.astro" +import LanguageTab from "../../../components/docs/LanguageTab.astro" + +Strands Agents has a built-in OpenTelemetry tracer, and Maple reads its spans without an extra instrumentation library. You set one environment variable so the transcript is recorded, and pass your conversation id as `session.id` on each agent. + +You need `strands-agents` 1.51 or newer, or the TypeScript SDK `@strands-agents/sdk` 1.19 or newer. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-strands](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-strands) skill and follows it. + +```text +Set up Maple agent tracing for Strands Agents in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-strands -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install and configure the exporter + + + + +Replace `openai` with your model provider's extra (`anthropic`, `litellm`; Bedrock needs none). + +```bash +pip install 'strands-agents[otel,openai]>=1.57' +``` + + + + +Install the SDK with its OpenTelemetry packages: + +```bash +npm install @strands-agents/sdk @opentelemetry/api @opentelemetry/sdk-trace-base @opentelemetry/sdk-trace-node @opentelemetry/resources @opentelemetry/exporter-trace-otlp-http @opentelemetry/sdk-metrics @opentelemetry/exporter-metrics-otlp-http +``` + + + + +Set these environment variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_SERVICE_NAME="support-agent" +export OTEL_RESOURCE_ATTRIBUTES="deployment.environment.name=production" +export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +Without `OTEL_SEMCONV_STABILITY_OPT_IN` the transcript is empty. Strands reads it when the first `Agent` is created, so set it in the environment rather than in code. + +To redact message content, append `gen_ai_unredacted_attributes=` with a `;`-separated allowlist of attributes to keep; everything else becomes `[REDACTED]`. + + + + +Then start the tracer once, before your first agent runs: + +```py +# telemetry.py: import this from your entry point before creating any Agent +from strands.telemetry import StrandsTelemetry + +telemetry = StrandsTelemetry().setup_otlp_exporter() +``` + +If your app already sets up a global `TracerProvider` (for example `opentelemetry-instrument` or the ADOT distro on AgentCore), skip `StrandsTelemetry()` and add a `BatchSpanProcessor(OTLPSpanExporter(...))` pointing at Maple to that provider. + + + + +Use the same `OTEL_EXPORTER_OTLP_*` variables, with `OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only"` (the SDK doesn't support the third value). Then: + +```ts +import { Agent, FileStorage, SessionManager } from "@strands-agents/sdk" +import { OpenAIModel } from "@strands-agents/sdk/models/openai" +import { setupTracer } from "@strands-agents/sdk/telemetry" + +const provider = setupTracer({ exporters: { otlp: true } }) // reads OTEL_EXPORTER_OTLP_* env vars +``` + + + + +## Pass the conversation id as session.id + + + + +Pass the conversation id in `trace_attributes`: + +```py +from strands import Agent +from strands.models.openai import OpenAIModel +from strands.session.file_session_manager import FileSessionManager + +model = OpenAIModel(model_id="gpt-4o-mini", params={"max_tokens": 600}) + + +def handle_message(conversation_id: str, text: str) -> str: + agent = Agent( + name="support_agent", + model=model, + tools=[get_weather, calculate], + system_prompt="You are a concise support assistant.", + session_manager=FileSessionManager(session_id=conversation_id, storage_dir="./sessions"), + trace_attributes={"session.id": conversation_id}, + callback_handler=None, + ) + return str(agent(text)) +``` + +The session manager restores history but doesn't put its `session_id` on the spans, so you need both arguments. Use the conversation id your app already has, never a fresh UUID per request. + +Create the agent per request, as above. A shared module-level `Agent` would carry one user's id into everyone's traces. Give every agent a `name`, or sub-agents merge into one lane. + +With `agent.as_tool()` sub-agents, only the orchestrator needs `session.id`. For a `Swarm`, pass `trace_attributes={"session.id": conversation_id}` to the `Swarm`. For a `Graph`, set `graph.trace_attributes = {"session.id": conversation_id}` after `builder.build()`. + + + + +```ts +const model = new OpenAIModel({ modelId: "gpt-4o-mini", params: { max_tokens: 600 } }) + +async function handleMessage(conversationId: string, text: string): Promise { + const agent = new Agent({ + name: "support_agent", + model, + tools: [getWeather], + systemPrompt: "You are a concise support assistant.", + traceAttributes: { "session.id": conversationId }, + sessionManager: new SessionManager({ + sessionId: conversationId, + storage: { snapshot: new FileStorage("./sessions") }, + }), + printer: false, + }) + return String(await agent.invoke(text)) +} +``` + +Create the agent per request here too. + + + + +## Flush in scripts and jobs + +A long-running server needs nothing extra. Scripts, notebooks and jobs lose their last spans unless they flush: + + + + +```py +from telemetry import telemetry + +try: + handle_message(conversation_id, "What's the weather in Berlin?") +finally: + telemetry.tracer_provider.force_flush() + telemetry.tracer_provider.shutdown() +``` + +On AWS Lambda, call `telemetry.tracer_provider.force_flush()` at the end of each invocation and don't call `shutdown()`. + + + + +```ts +try { + await handleMessage(conversationId, "What's the weather in Berlin?") +} finally { + await provider.forceFlush() + await provider.shutdown() +} +``` + + + + +## Check that it works + +Run a conversation of two or three messages, including one that calls a tool, then open **Agent Sessions** in Maple. You should see one session per conversation id with framework **Strands Agents**, one turn per `agent(...)` call, and a transcript with the user messages, replies and tool calls. + +Cost shows as unpriced because Strands doesn't report it. + +## Troubleshooting + +- **Transcript is empty, but tokens and tools show up.** Add `gen_ai_span_attributes_only` and `gen_ai_latest_experimental` to `OTEL_SEMCONV_STABILITY_OPT_IN`, set before the first `Agent` is created. +- **Every message is its own session.** Pass `trace_attributes={"session.id": conversation_id}` to the agent, or to the `Swarm` or `Graph` that runs it. +- **Two users' messages land in one session.** A shared `Agent` carries one `trace_attributes` dict for everyone. Create the agent per request. +- **Every model call appears twice.** Another instrumentation (OpenLIT, OpenLLMetry, OpenInference, an OpenAI or Bedrock instrumentor) wraps the same calls. Remove it. +- **Nothing arrives from a script.** The process exited before the batch was exported. Call `force_flush()` and `shutdown()` in a `finally` block. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Strands Agents traces documentation](https://strandsagents.com/docs/user-guide/observability-evaluation/traces/) diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md new file mode 100644 index 0000000000..0421b52634 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -0,0 +1,207 @@ +--- +title: "Trace Vercel AI SDK agents with OpenTelemetry" +description: "Send the Vercel AI SDK's OpenTelemetry spans to Maple and group each chat into one Agent Session." +group: "AI Agents" +order: 10 +navLabel: "Vercel AI SDK" +icon: "vercel" +--- + +The Vercel AI SDK emits OpenTelemetry spans for every `generateText`, `streamText` and `ToolLoopAgent` call, and Maple reads them without an extra instrumentation package. + +In AI SDK 7 nothing is traced until you call `registerTelemetry()` at startup. You also pass a conversation id on every call, or each message becomes its own session. + +You need `ai` 7.0.106 or newer and Node.js 22 or newer (Bun works too). On AI SDK 5 or 6, run `npx @ai-sdk/codemod v7` first. + +## Quick setup with a coding agent + +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-vercel-ai-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-vercel-ai-sdk) skill and follows it. + +```text +Set up Maple agent tracing for the Vercel AI SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-vercel-ai-sdk -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. + +## Install the packages + +```bash +npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/sdk-node +``` + +## Point the exporter at Maple + +```bash +export OTEL_SERVICE_NAME="support-agent" +export OTEL_RESOURCE_ATTRIBUTES="deployment.environment.name=production" +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +## Register the AI SDK integration + +Create an `instrumentation.ts` and import it as the first line of your entry point (`import "./instrumentation"`): + +```ts +// instrumentation.ts +import { OpenTelemetry } from "@ai-sdk/otel" +import { NodeSDK } from "@opentelemetry/sdk-node" +import { registerTelemetry } from "ai" + +// Reads OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES and OTEL_EXPORTER_OTLP_* +export const sdk = new NodeSDK() +sdk.start() + +registerTelemetry( + new OpenTelemetry({ + usage: true, + runtimeContext: true, + }), +) +``` + +Call `registerTelemetry()` exactly once. Each call adds another integration, and each one emits its own copy of every span. + +### Serverless, or an app that already uses OpenTelemetry + +Two setups need a handle on the span processor: a serverless handler that flushes after every invocation, and an app that already starts its own OpenTelemetry SDK (auto-instrumentation, Sentry, your own `NodeTracerProvider`), where you shouldn't start a second one. Install the exporter packages and create the processor yourself: + +```bash +npm install @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto +``` + +```ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" + +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) +``` + +Pass it as `new NodeSDK({ spanProcessors: [spanProcessor] })`, or add it to your existing provider instead of creating a `NodeSDK`. Keep the `registerTelemetry()` call either way. + +### Next.js + +In Next.js, register both in the `register()` function of `instrumentation.ts`, using `@vercel/otel` as in the [Next.js guide](/docs/guides/instrumentation-nextjs): + +```ts +// instrumentation.ts (project root, or src/instrumentation.ts) +import { OpenTelemetry } from "@ai-sdk/otel" +import { OTLPHttpProtoTraceExporter, registerOTel } from "@vercel/otel" +import { registerTelemetry } from "ai" + +export function register() { + const mapleKey = process.env.MAPLE_INGEST_KEY + if (!mapleKey) { + // A missing key disables export; it never stops the app. + console.warn("MAPLE_INGEST_KEY is not set; Maple telemetry export is disabled") + return + } + registerOTel({ + serviceName: "support-chat", + traceExporter: new OTLPHttpProtoTraceExporter({ + url: "https://ingest.maple.dev/v1/traces", // EU: https://ingest.eu.maple.dev/v1/traces + headers: { authorization: `Bearer ${mapleKey}` }, + }), + }) + + registerTelemetry( + new OpenTelemetry({ + usage: true, + runtimeContext: true, + }), + ) +} +``` + +Set `MAPLE_INGEST_KEY` to your ingest key; this exporter doesn't read the `OTEL_EXPORTER_OTLP_*` variables. + +Return AI SDK streams as the response (`toUIMessageStreamResponse()` or `createAgentUIStreamResponse()`). `@vercel/otel` ends all open spans when the request ends, so a stream read later, for example in `after()`, loses its reply and token counts. + +## Pass the conversation id on every call + +Pass the id as `runtimeContext.conversationId`. It only reaches telemetry if you also list it in `includeRuntimeContext`: + +```ts +import { streamText } from "ai" + +const result = streamText({ + model, + messages, + runtimeContext: { conversationId: chatId }, + telemetry: { functionId: "support_agent", includeRuntimeContext: { conversationId: true } }, +}) +``` + +Use the chat or thread id your app already stores. It must stay the same for the whole conversation and differ between conversations. `functionId` becomes the agent name in Maple, so give each agent a distinct one. + +`ToolLoopAgent` doesn't take `runtimeContext` per call. Declare the id as a call option and turn it into runtime context in `prepareCall`: + +```ts +// agent.ts +import { ToolLoopAgent } from "ai" +import { z } from "zod" + +export const assistant = new ToolLoopAgent({ + // ...model, instructions, tools + callOptionsSchema: z.object({ conversationId: z.string() }), + prepareCall: ({ options, ...rest }) => ({ + ...rest, + runtimeContext: { conversationId: options.conversationId }, + }), + telemetry: { functionId: "support_agent", includeRuntimeContext: { conversationId: true } }, +}) + +const result = await assistant.generate({ messages, options: { conversationId: chatId } }) +``` + +With `useChat`, the request body already carries a stable chat id as `id`. Pass it through: + +```ts +// app/api/chat/route.ts +import { createAgentUIStreamResponse, type UIMessage } from "ai" +import { assistant } from "@/agent" + +export async function POST(req: Request) { + const { id, messages }: { id: string; messages: UIMessage[] } = await req.json() + + return createAgentUIStreamResponse({ + agent: assistant, + uiMessages: messages, + options: { conversationId: id }, + }) +} +``` + +Sub-agents called from a tool's `execute` don't need the id, only their own `functionId`. + +To keep a call's prompts and replies out of Maple, set `recordInputs: false` and `recordOutputs: false` in its `telemetry`. + +## Flush in scripts and serverless functions + +A short-lived process can exit before its spans are exported. In a script, call `await sdk.shutdown()` in a `finally` block before exiting. In a serverless handler, call `await spanProcessor.forceFlush()` in a `finally` instead (see [the variant above](#serverless-or-an-app-that-already-uses-opentelemetry)). + +Read streams to the end (`await result.consumeStream()`) before flushing, since a stream's spans only end when it has been read. On Vercel, `@vercel/otel` flushes after each request, but queue consumers and cron jobs need their own `forceFlush()`. + +## Check that it works + +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session named after your conversation id with framework **Vercel AI SDK**, one turn per call, and a transcript with the prompts, replies and tool calls. + +Cost shows as unpriced because the AI SDK doesn't report it. + +## Troubleshooting + +- **No AI spans at all.** `registerTelemetry()` never ran. `experimental_telemetry: { isEnabled: true }` alone does nothing in AI SDK 7. +- **Every message is its own session.** Check all three parts: `runtimeContext: true` in `registerTelemetry()`, `runtimeContext: { conversationId }` on the call (or `prepareCall`), and `includeRuntimeContext: { conversationId: true }`. +- **Every span shows up twice.** `registerTelemetry()` ran twice, or a second OpenTelemetry SDK (such as Sentry without `skipOpenTelemetrySetup: true`) exports the same spans. +- **A failed tool shows as successful.** The tool returned an error value. Throw from `execute` instead. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [AI SDK telemetry](https://ai-sdk.dev/docs/ai-sdk-core/telemetry) diff --git a/apps/landing/src/content/docs/getting-started/ai-agents.md b/apps/landing/src/content/docs/getting-started/ai-agents.md index 142bbfb52d..c23f060820 100644 --- a/apps/landing/src/content/docs/getting-started/ai-agents.md +++ b/apps/landing/src/content/docs/getting-started/ai-agents.md @@ -113,10 +113,11 @@ When no tool fits, `describe_warehouse_tables` lists the tables and columns, and ## Instrument with a coding agent -Two open-source skills teach a coding agent how to set up OpenTelemetry for Maple: +Open-source skills teach a coding agent how to set up OpenTelemetry for Maple: - [maple-onboard](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-onboard) instruments every app and service in a repository: traces, logs and metrics, using the native OpenTelemetry SDK for each language. - [maple-audit](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-audit) reviews an existing setup, reports gaps per service (missing service map edges, missing `service.version`, errors without exceptions) and fixes them. +- [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) traces an AI agent so each conversation shows up as one [Agent Session](/docs/agent-sessions/overview). It detects the framework and installs a skill for that framework only. See [Trace your AI agent](/docs/agent-tracing). Install them together with the per-language guides they read: diff --git a/apps/landing/src/content/docs/getting-started/quickstart.mdx b/apps/landing/src/content/docs/getting-started/quickstart.mdx index 2d6754b64a..5cc7282bb9 100644 --- a/apps/landing/src/content/docs/getting-started/quickstart.mdx +++ b/apps/landing/src/content/docs/getting-started/quickstart.mdx @@ -140,13 +140,14 @@ func main() { ctx := context.Background() // Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. - exporter, err := otlptracehttp.New(ctx) - if err != nil { - log.Fatal(err) + // A setup error only disables tracing; it never stops the app. + if exporter, err := otlptracehttp.New(ctx); err != nil { + log.Printf("telemetry disabled: %v", err) + } else { + tp := sdktrace.NewTracerProvider(sdktrace.WithBatcher(exporter)) + defer tp.Shutdown(ctx) + otel.SetTracerProvider(tp) } - tp := sdktrace.NewTracerProvider(sdktrace.WithBatcher(exporter)) - defer tp.Shutdown(ctx) - otel.SetTracerProvider(tp) mux := http.NewServeMux() mux.HandleFunc("GET /hello", func(w http.ResponseWriter, r *http.Request) { diff --git a/apps/landing/src/content/docs/guides/instrumentation-go.md b/apps/landing/src/content/docs/guides/instrumentation-go.md index 8a649e0aa1..50beb95cfc 100644 --- a/apps/landing/src/content/docs/guides/instrumentation-go.md +++ b/apps/landing/src/content/docs/guides/instrumentation-go.md @@ -118,11 +118,12 @@ func initTelemetry(ctx context.Context) (func(context.Context) error, error) { func main() { ctx := context.Background() - shutdown, err := initTelemetry(ctx) - if err != nil { - log.Fatal(err) + // A setup error only disables telemetry; it never stops the app. + if shutdown, err := initTelemetry(ctx); err != nil { + log.Printf("telemetry disabled: %v", err) + } else { + defer shutdown(ctx) } - defer shutdown(ctx) // Your application code here } diff --git a/apps/landing/src/content/docs/instrumentation.mdx b/apps/landing/src/content/docs/instrumentation.mdx index 1524a4688b..7a89120399 100644 --- a/apps/landing/src/content/docs/instrumentation.mdx +++ b/apps/landing/src/content/docs/instrumentation.mdx @@ -53,6 +53,10 @@ Maple ingests OpenTelemetry over OTLP/HTTP, so any OTLP exporter works without a +## AI agents and LLM calls + +Agent frameworks and LLM gateways have their own guides, so each conversation shows up as one [Agent Session](/docs/agent-sessions/overview) with its transcript, tool calls, tokens and cost: Vercel AI SDK, OpenAI Agents SDK, LangChain and LangGraph, Mastra, Pydantic AI, CrewAI, Google ADK, OpenRouter, LiteLLM and more. See [Trace your AI agent](/docs/agent-tracing). +
## Infrastructure and data sources diff --git a/apps/landing/src/layouts/DocsLayout.astro b/apps/landing/src/layouts/DocsLayout.astro index 11a0a6070a..819c119744 100644 --- a/apps/landing/src/layouts/DocsLayout.astro +++ b/apps/landing/src/layouts/DocsLayout.astro @@ -161,6 +161,7 @@ const hasToc = !wide && headings.some((h) => h.depth >= 2 && h.depth <= 3);