From 6834bef789b52b6aa5aa9d73134d9963eda56f12 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Mon, 28 Sep 2026 19:15:08 +0200 Subject: [PATCH 01/39] docs(agent-tracing): per-framework agent tracing guides and skills (WIP) --- .../components/docs/DocsCategoryIcon.astro | 2 +- .../src/components/docs/DocsSidebar.astro | 5 +- .../src/components/docs/GuideGrid.astro | 12 +- apps/landing/src/content.config.ts | 4 + .../content/docs/agent-sessions/overview.md | 227 ++--- .../src/content/docs/agent-tracing.mdx | 69 ++ .../src/content/docs/agent-tracing/agno.md | 305 ++++++ .../docs/agent-tracing/claude-agent-sdk.md | 354 +++++++ .../src/content/docs/agent-tracing/crewai.md | 272 ++++++ .../src/content/docs/agent-tracing/dspy.md | 343 +++++++ .../content/docs/agent-tracing/google-adk.md | 295 ++++++ .../content/docs/agent-tracing/haystack.md | 380 ++++++++ .../content/docs/agent-tracing/langchain.md | 311 ++++++ .../src/content/docs/agent-tracing/litellm.md | 399 ++++++++ .../content/docs/agent-tracing/llamaindex.md | 324 +++++++ .../src/content/docs/agent-tracing/mastra.md | 325 +++++++ .../microsoft-agent-framework.md | 394 ++++++++ .../docs/agent-tracing/openai-agents.md | 318 +++++++ .../content/docs/agent-tracing/openrouter.md | 277 ++++++ .../docs/agent-tracing/opentelemetry.md | 883 ++++++++++++++++++ .../docs/agent-tracing/provider-sdks.md | 475 ++++++++++ .../content/docs/agent-tracing/pydantic-ai.md | 301 ++++++ .../content/docs/agent-tracing/smolagents.md | 233 +++++ .../content/docs/agent-tracing/spring-ai.md | 421 +++++++++ .../src/content/docs/agent-tracing/strands.md | 308 ++++++ .../docs/agent-tracing/vercel-ai-sdk.md | 409 ++++++++ apps/landing/src/lib/agent-tracing-guides.ts | 69 ++ apps/landing/src/lib/brand-marks.ts | 85 ++ apps/landing/src/lib/docs-nav.ts | 5 +- skills/maple-agent-tracing-agno/SKILL.md | 185 ++++ .../SKILL.md | 221 +++++ skills/maple-agent-tracing-crewai/SKILL.md | 215 +++++ skills/maple-agent-tracing-dspy/SKILL.md | 261 ++++++ .../maple-agent-tracing-google-adk/SKILL.md | 172 ++++ skills/maple-agent-tracing-haystack/SKILL.md | 258 +++++ skills/maple-agent-tracing-langchain/SKILL.md | 204 ++++ skills/maple-agent-tracing-litellm/SKILL.md | 252 +++++ .../maple-agent-tracing-llamaindex/SKILL.md | 227 +++++ skills/maple-agent-tracing-mastra/SKILL.md | 181 ++++ .../SKILL.md | 244 +++++ .../SKILL.md | 233 +++++ .../maple-agent-tracing-openrouter/SKILL.md | 166 ++++ .../SKILL.md | 144 +++ .../references/go.md | 153 +++ .../references/python.md | 275 ++++++ .../references/typescript.md | 308 ++++++ .../SKILL.md | 130 +++ .../references/python.md | 147 +++ .../references/typescript.md | 260 ++++++ .../maple-agent-tracing-pydantic-ai/SKILL.md | 202 ++++ .../maple-agent-tracing-smolagents/SKILL.md | 180 ++++ skills/maple-agent-tracing-spring-ai/SKILL.md | 262 ++++++ skills/maple-agent-tracing-strands/SKILL.md | 183 ++++ .../SKILL.md | 237 +++++ skills/maple-agent-tracing/SKILL.md | 55 ++ 55 files changed, 13005 insertions(+), 155 deletions(-) create mode 100644 apps/landing/src/content/docs/agent-tracing.mdx create mode 100644 apps/landing/src/content/docs/agent-tracing/agno.md create mode 100644 apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md create mode 100644 apps/landing/src/content/docs/agent-tracing/crewai.md create mode 100644 apps/landing/src/content/docs/agent-tracing/dspy.md create mode 100644 apps/landing/src/content/docs/agent-tracing/google-adk.md create mode 100644 apps/landing/src/content/docs/agent-tracing/haystack.md create mode 100644 apps/landing/src/content/docs/agent-tracing/langchain.md create mode 100644 apps/landing/src/content/docs/agent-tracing/litellm.md create mode 100644 apps/landing/src/content/docs/agent-tracing/llamaindex.md create mode 100644 apps/landing/src/content/docs/agent-tracing/mastra.md create mode 100644 apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md create mode 100644 apps/landing/src/content/docs/agent-tracing/openai-agents.md create mode 100644 apps/landing/src/content/docs/agent-tracing/openrouter.md create mode 100644 apps/landing/src/content/docs/agent-tracing/opentelemetry.md create mode 100644 apps/landing/src/content/docs/agent-tracing/provider-sdks.md create mode 100644 apps/landing/src/content/docs/agent-tracing/pydantic-ai.md create mode 100644 apps/landing/src/content/docs/agent-tracing/smolagents.md create mode 100644 apps/landing/src/content/docs/agent-tracing/spring-ai.md create mode 100644 apps/landing/src/content/docs/agent-tracing/strands.md create mode 100644 apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md create mode 100644 apps/landing/src/lib/agent-tracing-guides.ts create mode 100644 skills/maple-agent-tracing-agno/SKILL.md create mode 100644 skills/maple-agent-tracing-claude-agent-sdk/SKILL.md create mode 100644 skills/maple-agent-tracing-crewai/SKILL.md create mode 100644 skills/maple-agent-tracing-dspy/SKILL.md create mode 100644 skills/maple-agent-tracing-google-adk/SKILL.md create mode 100644 skills/maple-agent-tracing-haystack/SKILL.md create mode 100644 skills/maple-agent-tracing-langchain/SKILL.md create mode 100644 skills/maple-agent-tracing-litellm/SKILL.md create mode 100644 skills/maple-agent-tracing-llamaindex/SKILL.md create mode 100644 skills/maple-agent-tracing-mastra/SKILL.md create mode 100644 skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md create mode 100644 skills/maple-agent-tracing-openai-agents/SKILL.md create mode 100644 skills/maple-agent-tracing-openrouter/SKILL.md create mode 100644 skills/maple-agent-tracing-opentelemetry/SKILL.md create mode 100644 skills/maple-agent-tracing-opentelemetry/references/go.md create mode 100644 skills/maple-agent-tracing-opentelemetry/references/python.md create mode 100644 skills/maple-agent-tracing-opentelemetry/references/typescript.md create mode 100644 skills/maple-agent-tracing-provider-sdks/SKILL.md create mode 100644 skills/maple-agent-tracing-provider-sdks/references/python.md create mode 100644 skills/maple-agent-tracing-provider-sdks/references/typescript.md create mode 100644 skills/maple-agent-tracing-pydantic-ai/SKILL.md create mode 100644 skills/maple-agent-tracing-smolagents/SKILL.md create mode 100644 skills/maple-agent-tracing-spring-ai/SKILL.md create mode 100644 skills/maple-agent-tracing-strands/SKILL.md create mode 100644 skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md create mode 100644 skills/maple-agent-tracing/SKILL.md diff --git a/apps/landing/src/components/docs/DocsCategoryIcon.astro b/apps/landing/src/components/docs/DocsCategoryIcon.astro index f9a509ba2b..ddea378586 100644 --- a/apps/landing/src/components/docs/DocsCategoryIcon.astro +++ b/apps/landing/src/components/docs/DocsCategoryIcon.astro @@ -110,7 +110,7 @@ const base = { )} -{name === "Agent Sessions" && ( +{(name === "Agent Sessions" || name === "AI Agents") && ( diff --git a/apps/landing/src/components/docs/DocsSidebar.astro b/apps/landing/src/components/docs/DocsSidebar.astro index ae51b5b460..8d28c97b71 100644 --- a/apps/landing/src/components/docs/DocsSidebar.astro +++ b/apps/landing/src/components/docs/DocsSidebar.astro @@ -5,6 +5,7 @@ // Language rows carry their logo; Effect's platform pages nest under Effect. import { getCollection } from "astro:content"; import DocsCategoryIcon from "./DocsCategoryIcon.astro"; +import BrandMarkIcon from "../BrandMarkIcon.astro"; import LanguageLogo from "./LanguageLogo.astro"; import { groupRank, isDocGroup, sectionForGroup } from "../../lib/docs-nav"; import { getDocSections } from "../../lib/docs-order"; @@ -73,8 +74,10 @@ const rowClass = (active: boolean) => aria-current={active ? "page" : undefined} class:list={["flex items-center gap-2 px-2.5 py-1.5 text-[13px] leading-5 transition-colors focus-visible:outline-none focus-visible:ring-1 focus-visible:ring-inset focus-visible:ring-primary", rowClass(active)]} > - {doc.data.sdk && ( + {doc.data.sdk ? ( + ) : ( + doc.data.icon && )} {label(doc)} diff --git a/apps/landing/src/components/docs/GuideGrid.astro b/apps/landing/src/components/docs/GuideGrid.astro index 393057ec48..32feee9721 100644 --- a/apps/landing/src/components/docs/GuideGrid.astro +++ b/apps/landing/src/components/docs/GuideGrid.astro @@ -1,17 +1,19 @@ --- -// One ecosystem's cards on /docs/instrumentation. Styled with scoped CSS on +// One ecosystem's cards on /docs/instrumentation (or a given card list, as on /docs/frontend). Styled with scoped CSS on // purpose: the page sits inside `.docs-content`, whose descendant rules for // `a`/`p` beat plain Tailwind utilities (see `local/InstallTabs.astro`). -import { GUIDE_SECTIONS } from "../../lib/instrumentation-guides"; +import { type GuideCard, GUIDE_SECTIONS } from "../../lib/instrumentation-guides"; import BrandMarkIcon from "../BrandMarkIcon.astro"; import LanguageLogo from "./LanguageLogo.astro"; interface Props { - section: string; + /** A section of GUIDE_SECTIONS, or `cards` for a list shown elsewhere. */ + section?: string; + cards?: readonly GuideCard[]; } -const { section } = Astro.props; -const data = GUIDE_SECTIONS.find((s) => s.id === section); +const { section, cards } = Astro.props; +const data = cards ? { cards } : GUIDE_SECTIONS.find((s) => s.id === section); if (!data) throw new Error(`Unknown guide section: ${section}`); --- diff --git a/apps/landing/src/content.config.ts b/apps/landing/src/content.config.ts index 961b5ef88b..cba585c4bd 100644 --- a/apps/landing/src/content.config.ts +++ b/apps/landing/src/content.config.ts @@ -1,5 +1,6 @@ import { defineCollection, reference, z } from "astro:content" import { glob } from "astro/loaders" +import { BRAND_MARKS, type BrandMarkId } from "./lib/brand-marks" import { LANGUAGE_IDS } from "./lib/docs-languages" import { CHANGELOG_CATEGORIES, CONTRIBUTOR_IDS } from "./lib/changelog-meta" @@ -28,6 +29,9 @@ const docs = defineCollection({ // ("Node.js" for "Node.js Instrumentation"). navLabel: z.string().optional(), sdk: z.enum(LANGUAGE_IDS).optional(), + // Brand mark on the sidebar row, for guides that aren't a language (`sdk`), + // like the frontend framework guides. + icon: z.enum(Object.keys(BRAND_MARKS) as [BrandMarkId, ...BrandMarkId[]]).optional(), }), }) diff --git a/apps/landing/src/content/docs/agent-sessions/overview.md b/apps/landing/src/content/docs/agent-sessions/overview.md index 22bcabcdaf..91f956bd58 100644 --- a/apps/landing/src/content/docs/agent-sessions/overview.md +++ b/apps/landing/src/content/docs/agent-sessions/overview.md @@ -1,197 +1,132 @@ --- title: "Agent Sessions" -description: "An AI agent conversation as one session: every turn, model call and tool call with its cost, timing and failures, built from OpenTelemetry GenAI traces. What Maple records, and how to connect an agent in any language or framework." +description: "See an AI agent conversation as one session: every turn, model call and tool call, with its tokens, cost, timing and failures, built from the OpenTelemetry traces your agent already sends." group: "Agent Sessions" order: 1 navLabel: "Overview" --- -**Agent Sessions** shows an AI agent conversation as one session: every turn, model call and tool call, with its cost, timing and failures. Maple builds sessions from the OpenTelemetry traces your agent already sends, using the OpenTelemetry GenAI semantic conventions. There is no extra SDK. +A trace shows you one request. An agent conversation is rarely one request: a chat backend handles each user message separately, so a ten-message conversation is ten traces, and the model calls, tool calls and handoffs that matter are scattered across them. -
- No extra SDK - OpenTelemetry GenAI semconv - 20+ frameworks - Any language -
- -For each session you get: - -- an overview that splits the wall clock into model time, tool time and idle, and rolls up cost and tokens per model; -- the transcript, with each model call's model, tokens, cost and finish reason; -- the trace, grouped by turn; -- tool pages that rank every tool by volume, failure rate and latency across sessions. - -The same data is on the [MCP server](/docs/reference/mcp) as `list_agent_sessions`, `get_agent_session`, `get_agent_tools_overview` and `get_agent_tool_error`. - -## Connect your agent - -### Step 1: traces are flowing +**Agent Sessions** puts them back together. Maple groups the traces of one conversation into a session and shows it the way you think about it: turn by turn, with the transcript, every model and tool call, what it cost, where the time went and what failed. Sessions are built from OpenTelemetry traces, so there is no Maple SDK to add. To get your agent's traces in, pick your framework in [Trace your AI agent](/docs/agent-tracing). -Your service exports OTLP to `https://ingest.maple.dev` (`https://ingest.eu.maple.dev` for EU organizations) with an ingest key. If it does not yet, pick your language in [Instrument your application](/docs/instrumentation) and come back. Agent Sessions adds nothing to that setup; it reads the traces that already arrive. +## What is an agent session? -### Step 2: emit GenAI spans +A session is one conversation between a user (or a job) and your agent. Maple uses four levels: -There are two ways to get there. If your agent runs on a [framework Maple recognizes](#frameworks-maple-recognizes-automatically), turn on that framework's OpenTelemetry export. Otherwise, emit the **OpenTelemetry GenAI semantic conventions** directly. We recommend this path: every framework integration is normalized into these conventions anyway, and they work in every language. +| Level | What it is | Where it comes from | +| --- | --- | --- | +| **Session** | One conversation, from the first message to the last. | Every trace that carries the same session id, such as `gen_ai.conversation.id`. | +| **Turn** | One user message and everything the agent did to answer it. | Usually one trace, rooted at an `invoke_agent` span. | +| **Model call** | One request to an LLM: the prompt, the reply, tokens, finish reason. | A `chat` (or `generate_content`, `text_completion`) span. | +| **Tool call** | One function the model asked to run, with its arguments and result. | An `execute_tool` span. | -The conventions live at [opentelemetry.io/docs/specs/semconv/gen-ai](https://opentelemetry.io/docs/specs/semconv/gen-ai/). The pages you will actually use: +Sub-agents sit inside a turn. When an orchestrator hands work to a `researcher` agent, the researcher's model and tool calls show up as their own lane, labeled with its `gen_ai.agent.name`. -- [Inference spans](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-spans/): the `chat`, `text_completion` and `embeddings` spans, the request and response attributes, and the token usage attributes. -- [Agent and tool spans](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-agent-spans/): `invoke_agent`, `create_agent` and `execute_tool`, and the `gen_ai.tool.*` attributes. -- [Message content](https://opentelemetry.io/docs/specs/semconv/gen-ai/gen-ai-events/): the `{role, parts}` shape of `gen_ai.input.messages`, `gen_ai.output.messages` and `gen_ai.system_instructions`. -- [Attribute registry](https://opentelemetry.io/docs/specs/semconv/registry/attributes/gen-ai/): every `gen_ai.*` key with its type and examples. +A background job that never talks to a user is still a session: one run, usually one trace. -The session id is `gen_ai.conversation.id`: put the same value on every span of a conversation and its traces become one session. Everything else Maple reads is in [the attribute table](#the-attributes-maple-reads) below. +## What a session shows you -### Step 3: open Explore → Agent Sessions - -Run one conversation and open the list. The session appears as soon as its first trace lands, and the overview and transcript fill in as the rest of the turns arrive. If it does not look like the screenshots in [A session, step by step](#a-session-step-by-step), [When it does not look right](#when-it-does-not-look-right) covers the five usual reasons. - -### The attributes Maple reads - -You do not need all of these. The first two rows make a session; the rest make it useful. - -| Attribute | Used for | -| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `gen_ai.operation.name` | Marks the span as AI and says what kind: `chat`, `text_completion`, `embeddings`, `execute_tool`, `invoke_agent`, `create_agent`, `invoke_workflow`. | -| `gen_ai.conversation.id` | Groups traces into one session: the same value on every span of a conversation. | -| `gen_ai.provider.name`, `gen_ai.request.model`, `gen_ai.response.model` | Provider and model facets, per-model token and cost roll-ups. The older `gen_ai.system` is accepted too. | -| `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_creation.input_tokens`, `gen_ai.usage.reasoning.output_tokens` | The five token buckets. Maple knows which providers nest one bucket inside another and does not double count. | -| `gen_ai.usage.cost` | Cost in USD, if your instrumentation prices calls. The conventions define no cost attribute, so Maple does not price semconv calls itself; this is the OpenLLMetry key, and `llm.cost.total` from OpenInference is read too. | -| `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions` | The transcript. | -| `gen_ai.agent.name`, `gen_ai.agent.id`, `gen_ai.agent.description` | Agent facet, and sub-agent handoffs inside a session. | -| `gen_ai.tool.name`, `gen_ai.tool.call.id`, `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result`, `gen_ai.tool.definitions` | Tool calls in the transcript, and everything on the tool pages. | -| `gen_ai.response.id`, `gen_ai.response.finish_reasons`, `gen_ai.response.time_to_first_chunk` | Response metadata and time to first token. | -| `error.type` plus span status | Failed model and tool calls, and the failure groups on the tool pages. | - -Retrieval, memory, embeddings and evaluation attributes from the conventions are stored and shown on the span, but do not change how the session is built. - -## Frameworks Maple recognizes automatically - -If your agent runs on one of these, use the framework's own OpenTelemetry exporter or the instrumentation listed and point it at Maple. Maple recognizes the framework at ingest, normalizes its attribute dialect into the `gen_ai.*` fields above, and takes the session id from wherever that framework keeps it. - -| Framework | Instrumentation Maple recognizes | Session id | -| -------------------------------- | --------------------------------------------------- | --------------------------------------------------------- | -| Vercel AI SDK | The SDK's `experimental_telemetry` | One per trace | -| OpenAI Agents SDK | OpenInference `openai_agents` instrumentation | `session.id` or `gen_ai.conversation.id` | -| Claude Code and Claude Agent SDK | Built-in OpenTelemetry telemetry | `session.id` | -| LangChain and LangGraph | LangSmith OpenTelemetry export | `langsmith.metadata.thread_id` | -| LlamaIndex | LlamaIndex's OpenTelemetry observability package | One per trace | -| Pydantic AI | Built-in (`instrument=True`) or Logfire | `gen_ai.conversation.id` | -| Mastra | `@mastra/otel-exporter` | `gen_ai.conversation.id` | -| Google ADK | Built-in OpenTelemetry tracing | `gen_ai.conversation.id` or `gcp.vertex.agent.session_id` | -| Microsoft Agent Framework | Built-in OpenTelemetry tracing | `gen_ai.conversation.id` | -| Semantic Kernel | Built-in OpenTelemetry tracing | One per trace | -| Spring AI | Spring Boot observability over OTLP | `spring.ai.chat.client.conversation.id` | -| CrewAI | Built-in telemetry or OpenInference instrumentation | `session.id` | -| DSPy | OpenInference `dspy` instrumentation | `session.id` | -| smolagents | OpenInference `smolagents` instrumentation | `session.id` | -| Agno | OpenInference `agno` instrumentation | `session.id` | -| Strands Agents | Built-in OpenTelemetry tracing | `session.id` | -| Haystack | Built-in OpenTelemetry tracer | One per trace | -| LiteLLM | Built-in OpenTelemetry callback | One per trace | -| OpenRouter | Broadcast to an OTLP endpoint | `session.id` | -| Effect AI | `@effect/opentelemetry` | One per trace | -| OpenAI SDK via OpenInference | OpenInference `openai` instrumentation | `session.id` | - -### Claude Code - -Claude Code's tracing is a beta behind its own flag, and its content is redacted unless you opt in. Set these in the shell that runs `claude`, or under `env` in `~/.claude/settings.json` to cover every session including the desktop app: - -```bash -export CLAUDE_CODE_ENABLE_TELEMETRY=1 -export CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1 # traces; without it there are no spans to build a session from -export OTEL_TRACES_EXPORTER=otlp -export OTEL_LOGS_EXPORTER=otlp # events: prompts, responses, per-request cost -export OTEL_METRICS_EXPORTER=otlp # optional: cost and token counters -export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf -export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" # https://ingest.eu.maple.dev for EU orgs -export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" - -# Content. Each is off by default; turn on what you are allowed to store. From Claude Code -# v2.1.193 an unset OTEL_LOG_ASSISTANT_RESPONSES follows OTEL_LOG_USER_PROMPTS: set it to 0 -# explicitly if prompts are approved and replies are not. -export OTEL_LOG_USER_PROMPTS=1 # the prompt that opens each turn -export OTEL_LOG_TOOL_DETAILS=1 # tool arguments: Bash commands, file paths, MCP tool names -export OTEL_LOG_TOOL_CONTENT=1 # tool results for Read, Bash, Edit and Write (not MCP tools) -export OTEL_LOG_ASSISTANT_RESPONSES=1 # the assistant's replies, on the assistant_response event -``` - -What each span becomes: - -| Claude Code span | In Agent Sessions | -| ---------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `claude_code.interaction` | A turn, titled with the prompt when `OTEL_LOG_USER_PROMPTS=1`. | -| `claude_code.llm_request` | A model call: model, the four token buckets (Anthropic's input count excludes the cache buckets, and Maple counts it that way), TTFT, finish reason and failures. | -| `claude_code.tool` | A tool call, with the command or file path as its arguments and, for Read, Bash, Edit and Write, the `tool.output` content as its result. | -| `claude_code.tool.execution`, `claude_code.tool.blocked_on_user` | Shown in the trace as part of their tool call rather than as calls of their own. A failed run marks the tool call failed, with its error. | - -Cost and the assistant's reply text are only on Claude Code's log events (`api_request`, `assistant_response`), not on its spans. They are stored and searchable under Logs; the session views read spans, so cost reads as unpriced there for now. - -For the frameworks that give you one session per trace, stamp `gen_ai.conversation.id` on every span of the conversation (a span processor is the usual place) and Maple groups them into one session. - -Two dialects that are not frameworks are recognized as well: any **OpenInference** emitter (`openinference.span.kind`, `llm.*`, `input.value`) and any **OpenLLMetry / Traceloop** emitter (`traceloop.*`, `llm.*`). Their spans land as sessions without a framework name attached: one per trace, or one per conversation when the spans carry `gen_ai.conversation.id`. - -**Don't see yours?** Send us the framework and a sample trace at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP). Adding a framework is a detection rule and a dialect map on our side, not a new SDK, so it is usually quick. In the meantime, anything that emits `gen_ai.operation.name` already works through the generic path above. - -## A session, step by step - -One conversation from a support agent instrumented with the OpenTelemetry GenAI conventions: the customer asks to change a delivery address, gives one in Paris, and ends up canceling the order. This is what Maple recorded. +This is one conversation with a support agent that uses the OpenTelemetry GenAI conventions. The customer asks to change a delivery address, gives one in Paris, and ends up canceling the order.
A session's overview page: a time breakdown bar, a findings list with a failed tool call, a tools table with calls, failures and a timeline, and a right column with cost by model and token buckets. -
The session overview. The failed tool call leads the page: update_shipping_address returned unsupported_destination in turn 2.
+
The overview. The failed tool call leads the page: update_shipping_address returned unsupported_destination in turn 2.
-The overview splits the wall clock into model time, tool time and idle, and rolls cost and tokens up per model. Here the agent was busy for 14 seconds of 2 minutes 25 seconds. The rest was the customer typing. +The **overview** splits the wall clock into model time, tool time and idle time, and rolls cost and tokens up per model. Here the agent was busy for 14 seconds of a 2 minute 25 second session. The rest was the customer typing. + +Under it, Maple runs a set of checks over the session and gives it a verdict: it completed cleanly, completed with warnings, or failed, and if it failed, which span ended it. The checks that need attention come first, each with what to do about it: + +- **Completion**, **Context window**, **Reply length** and **Refusals**: did the model finish its answers, or did it run out of context, hit `max_tokens` or refuse? +- **Rate limits** and **Provider errors**: did the LLM provider fail calls, and did the session survive them? +- **Tool errors**, **Tool arguments**, **Tool timeouts** and **Tool availability**: which tools failed, and was it the tool or the arguments the model sent? +- **Repeated calls** and **Stalls**: is the agent looping, or did it stop making progress? +- **Prompt cache**: is the prompt prefix being reused across calls, or are you paying full price every turn? +- **Structured output**: did the model's JSON parse? + +When your instrumentation doesn't record something a check needs, like message content or tool arguments, the check says it was skipped and what to capture, instead of passing quietly.
The transcript view of a session: system instructions, user and assistant messages in sequence, each model call annotated with its model, tokens, cost and finish reason, and a tool call row with its latency and payload sizes.
The transcript. Each model call carries its model, tokens, cost and finish reason; tool calls sit where the model made them.
+The **transcript** is the conversation as the model saw it: system instructions, user and assistant messages, and tool calls with their arguments and results, in order. It needs message content on your spans, which most instrumentations leave off by default. Each framework guide shows the switch. +
The trace view of a session: three turns, each with an invoke_agent span, chat spans labeled with their model and token counts, and execute_tool spans, on a time axis with the idle gaps between turns removed. One tool span is marked with its error type. -
The trace. Spans grouped by turn, 2m 10s of idle cut from the axis, the failed tool span flagged with its error.type.
+
The trace. Spans grouped by turn, with 2 minutes 10 seconds of idle time cut from the axis and the failed tool span flagged.
+The **trace** view is every span of every turn on one time axis, with the idle gaps between turns removed so a slow tool doesn't hide next to a user who went for coffee. +
The Agent Sessions list in Maple, one row per session with services, model, duration, LLM call and tool call counts, tokens, cost, errors and start time, and a filter sidebar on the left. -
The list. One row per conversation for your retention period, filterable by framework, service, model, agent and tool; failed ones marked.
+
The list. One row per conversation, filterable by framework, service, environment, model, agent and tool.
-Tools get the same treatment across sessions: ranked by volume, failure rate and latency, with each failure grouped by error type and the arguments and results that produced it. See [Debug and monitor tools](#debug-and-monitor-tools). +The **list** has one row per session with its model, duration, call counts, tokens, cost and errors. Sort by cost to find the expensive conversations, or filter to sessions that called a given tool. -## Debug and monitor tools +## Debug the tools your agent calls -Every `execute_tool` span is ranked, charted and grouped across sessions, so the three questions behind a misbehaving agent each take one click. +When an agent misbehaves, the cause is usually a tool: it failed, it was slow, or the model called it with arguments it couldn't use. The **Tools** tab looks at every tool call across all sessions.
The Tools tab: a metric strip with tool calls, sessions, error rate and duration, a chart of calls over time, and a table ranking each tool by calls, p50, p90, p95, error rate, errors, sessions and last call. -
Which tool is failing? The Tools tab ranks every tool by calls, latency percentiles and error rate for the window. Failing only keeps just the ones that have failed.
+
Which tool is failing? Every tool ranked by calls, latency percentiles and error rate. Failing only keeps just the ones that have failed.
- A tool's detail page: four charts for tool calls, error rate, duration percentiles and calls per session, then an errors table with one row per error type showing trend, share, count, sessions and last seen. -
Why? A tool's page charts its calls, error rate, duration and calls per session, then groups its failures by error.type with a trend, share and the sessions hit.
+ A tool's detail page: four charts for tool calls, error rate, duration percentiles and calls per session, then an errors table with one row per error showing trend, share, count, sessions and last seen. +
Why? A tool's page charts its calls, error rate, duration and calls per session, then groups its failures with a trend and the sessions they hit.
The error group dialog for a tool: the error type, how many calls and sessions it affected, a failed-calls-per-day chart, a Where it happens panel listing the model and service, a sessions list, and a sample failed call with its JSON arguments and JSON result side by side. -
What exactly happened? An error group opens on the failed calls themselves: arguments on the left, the result on the right, and a jump into the trace.
+
What exactly happened? An error group opens on the failed calls themselves: arguments on the left, the result on the right, and a link to the trace.
-
- A tool span opened from the session overview: an ERROR banner with the error type and message, the result JSON, the span's timing and identifiers, and its AI attributes including operation, conversation id and tool name. -
Inside a session, a failed tool is a finding on the overview; opening it shows the error, the result and every attribute the span carried.
-
+Failures are grouped by what went wrong: Maple fingerprints the failed result (or the span's status message) with ids and numbers masked, so a thousand `order 12345 not found` errors are one group, not a thousand. A tool call counts as failed when its span has an `ERROR` status or an `error.type` attribute. Many frameworks catch a tool's exception and hand the message back to the model without marking the span, which makes a broken tool look healthy; the framework guides cover how to avoid that. + +## How Maple builds a session from your traces + +You don't need this to use Agent Sessions, but it explains what the framework guides ask you to configure. + +1. **Maple recognizes AI spans at ingest.** A span counts if it carries `gen_ai.operation.name` (the OpenTelemetry GenAI conventions) or matches the fingerprint of a framework Maple knows: Vercel AI SDK, OpenAI Agents SDK, LangChain and LangGraph, Mastra, Pydantic AI, CrewAI, Google ADK, Strands, Claude Code, Spring AI and others. Traces with no AI span don't appear. +2. **It reads the session id that framework uses.** For most frameworks that's `gen_ai.conversation.id`; a few use `session.id`, LangGraph uses its thread id. One span per trace is enough. A trace without one becomes a session of its own, which is why "every message is its own session" is the most common setup problem. +3. **It splits the session into turns**, normally one per trace, and decodes each model call's model, tokens and content, and each tool call's name, arguments and result. +4. **It counts tokens once.** Providers disagree on whether cached and reasoning tokens are included in the input and output counts. Maple resolves that per provider, and when a framework records usage on both an agent span and the model calls inside it, Maple keeps the model calls' numbers. + +Two limits are worth knowing up front: + +- **Maple reads span attributes.** Prompts and replies that a framework writes only to span events or OpenTelemetry logs are still stored (on the span, or under [Logs](/docs/explore/logs)), but they don't show up in the transcript. The framework guides say where each framework puts its content and how to move it onto spans when that's possible. +- **Maple shows cost your instrumentation reports; it doesn't price tokens itself.** Cost appears when spans carry `gen_ai.usage.cost` (or OpenInference's `llm.cost.total`). OpenRouter and some instrumentations send it. Otherwise the session shows tokens and reads as unpriced. + +A session view loads up to 2,000 spans. Longer sessions show the first 2,000 and say they were cut. + +## Get your agent into Agent Sessions + +Maple ingests OpenTelemetry over OTLP/HTTP, so the work is turning on your framework's tracing and pointing it at `https://ingest.maple.dev` (`https://ingest.eu.maple.dev` for EU organizations) with an ingest key from **Settings → Ingestion**. [Trace your AI agent](/docs/agent-tracing) has a guide for each framework, and a prompt that has a coding agent do the setup for you. + +The guides all aim for the same result: + +- one session per conversation, not one per message; +- the prompt, reply and tool arguments and results in the transcript; +- tokens on every model call, including streamed ones; +- failed tools marked as failed; +- sub-agents named, so they get their own lanes. + +## Query sessions from your coding agent + +The same data is on the [MCP server](/docs/reference/mcp): `list_agent_sessions` finds sessions by cost, model, tool or failure, `get_agent_session` returns a session's verdict, checks and turns, and `get_agent_tools_overview` and `get_agent_tool_error` return the tool rankings and failure groups. Ask your coding agent "why did the last expensive session fail?" and it can answer from the traces. -A tool span needs `gen_ai.operation.name` of `execute_tool`, `gen_ai.tool.name`, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, and on failure a span status of `ERROR` with a stable `error.type`. Groups are keyed on `error.type`, with ids and timestamps masked so one failure is one row. `get_agent_tools_overview` and `get_agent_tool_error` on the [MCP server](/docs/reference/mcp) return the same ranking and groups. +## When a session doesn't look right -## When it does not look right +- **Every message is its own session.** No span in the trace carried a session id Maple reads for that framework. The framework's guide shows where to set it. +- **The transcript is empty.** Message content isn't on the spans: either capture is off (the default for most instrumentations), or the framework writes content to span events or logs only. Content that is on the spans must be a JSON string, such as a `[{role, parts}]` array; a plain string is ignored. +- **The framework shows as "Unidentified".** The spans follow the GenAI conventions but match no framework fingerprint. Sessions, transcripts and tools all work; only the framework facet is missing. Tell us which framework it is. +- **Token totals look doubled.** Two instrumentations recorded the same model call, typically the framework's and a provider SDK instrumentor like OpenAI's. Turn one off. +- **Nothing appears at all.** Check that ordinary traces from the service show up under **Explore → Traces** first. If they don't, the exporter isn't reaching Maple; a short-lived script that exits before flushing is the usual cause. If they do, none of the spans is recognized as AI; check the framework's tracing is actually on. -- **Every turn is its own session.** No span carried a session id Maple recognizes. If the framework table lists a session key for your framework, check that its spans carry it; otherwise add `gen_ai.conversation.id` to every span of the conversation. -- **The transcript is empty.** Message content is not on the spans. Most official instrumentations leave it off by default, and some only ever write it to log events, which Maple does not read. If content is on the spans and still missing, check that the attribute holds a JSON array of `{role, parts}` objects rather than a plain string. -- **The framework shows as "Unidentified".** The spans carry `gen_ai.*` attributes but no fingerprint of a known framework. Sessions, transcripts and tool pages all work; only the framework facet is missing. Tell us which framework it is and we will add the rule. -- **Token totals look too high or too low.** Providers disagree on whether cached and reasoning tokens are included in the input and output counts. Maple resolves that per `gen_ai.provider.name`, so if the provider name is missing or unexpected, set it and the totals correct themselves. -- **Nothing appears at all.** Confirm ordinary traces from the service show under **Explore → Traces** first. If they do, no span in them carries `gen_ai.operation.name`; if they do not, the problem is the exporter, and the [instrumentation guide](/docs/instrumentation) for your language covers it. +Using a framework we don't cover? Send us the framework and a sample trace at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP). Adding one is a detection rule on our side, not a new SDK. Until then, [the OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry) works for any agent in any language. diff --git a/apps/landing/src/content/docs/agent-tracing.mdx b/apps/landing/src/content/docs/agent-tracing.mdx new file mode 100644 index 0000000000..fa8b2280dc --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing.mdx @@ -0,0 +1,69 @@ +--- +title: "Trace your AI agent" +description: "Setup guides for every agent framework and LLM gateway Maple supports, so each conversation shows up as one Agent Session with its transcript, tool calls, tokens and cost." +group: "AI Agents" +order: 0 +navLabel: "Overview" +--- + +import GuideGrid from "../../components/docs/GuideGrid.astro" +import { AGENT_GUIDE_SECTIONS } from "../../lib/agent-tracing-guides" + +Most agent frameworks can export OpenTelemetry traces, and almost none of them export good ones by default. The usual result of turning tracing on is a list of disconnected traces, one per user message, with no prompts in them, and a tool that failed forty times marked green. + +Each guide below gets a framework from there to a proper [Agent Session](/docs/agent-sessions/overview): one session per conversation, the transcript, every model and tool call with its tokens, failures marked as failures, and sub-agents in their own lanes. Maple ingests OTLP directly, so there is no Maple SDK to add. You turn on the framework's own instrumentation and point it at Maple. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) skill, which detects your framework and installs the skill for it, so the agent only reads the steps for your stack. + +```text +Set up Maple agent tracing in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. Each guide below has the same prompt for its own framework. + +## Choose your framework + +{AGENT_GUIDE_SECTIONS.map((section) => ( + <> +

{section.title}

+ + +))} + +If you use a gateway like OpenRouter or LiteLLM **and** a framework, pick one of the two for model calls. Instrumenting both records every call twice; Maple dedupes calls that share a response id, but the framework guide is the one that gives you turns, tools and sub-agents. + +## What every guide sets up + +The frameworks differ, the goal doesn't. A finished setup has: + +- **One session per conversation.** Chat backends handle each message as its own request, so each message is its own trace. A session id on the spans (`gen_ai.conversation.id` for most frameworks) ties them together. Almost no framework sets one by default, and without it every message is a separate session. +- **The transcript.** Prompts, replies and tool arguments and results, recorded on the spans. Most instrumentations leave content off by default for privacy, and some write it only to logs, which Maple doesn't read for the transcript. +- **Tokens on every model call**, streamed ones included. Streaming responses often report no usage unless you ask for it. +- **Failed tools marked as failed.** Frameworks usually catch a tool's exception and hand it back to the model, so the span looks successful unless the instrumentation marks it. +- **Named agents.** `gen_ai.agent.name` on each agent, so a handoff to a sub-agent shows up as its own lane. +- **Spans that actually arrive.** Exporters batch spans in the background. Scripts, CLIs, notebooks and serverless functions have to flush before they exit, or the last turn never reaches Maple. + +## Connection details + +Every guide exports OTLP over HTTP to your organization's region with an ingest key from **Settings → Ingestion**: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" # https://ingest.eu.maple.dev for EU organizations +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_SERVICE_NAME="support-agent" +``` + +The exporter appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. If you set the traces-only variable `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` or pass an `endpoint=` argument in code, most SDKs use it as-is, so include `/v1/traces` yourself. Maple's endpoint is OTLP over HTTP only, so an exporter left on its gRPC default fails; use `http/protobuf` or `http/json`. + +If your service samples traces, a sampled-out turn is a gap in the session, so keep agent traffic at 100%. See [Sampling and throughput](/docs/concepts/sampling-throughput). + +## Not listed? + +Anything that emits the OpenTelemetry GenAI conventions works, in any language: [Any language](/docs/agent-tracing/opentelemetry) lists exactly what Maple reads. Frameworks instrumented with OpenInference or OpenLLMetry also land as sessions without a dedicated guide, with the framework shown as unidentified. Tell us what you run at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP) and we'll add a guide. diff --git a/apps/landing/src/content/docs/agent-tracing/agno.md b/apps/landing/src/content/docs/agent-tracing/agno.md new file mode 100644 index 0000000000..7dafa92aec --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/agno.md @@ -0,0 +1,305 @@ +--- +title: "Trace Agno agents and teams with OpenTelemetry" +description: "Send Agno agent, team and tool spans to Maple with the OpenInference instrumentor, grouped into one Agent Session per conversation with transcripts, tokens and failed tools." +group: "AI Agents" +order: 26 +navLabel: "Agno" +icon: "agno" +--- + +Agno's tracing is built on OpenInference. The `openinference-instrumentation-agno` package wraps every agent and team run, every model call and every tool call, and Agno's own `setup_tracing()` uses the same instrumentor. The catch is where `setup_tracing()` sends the spans: into your AgentOS database, not to an OpenTelemetry endpoint. To get them into Maple you install the instrumentor yourself with an OTLP exporter. + +The thing that goes wrong by default is the session id. Agno always stamps `session.id` on the run span, but when you don't pass `session_id=`, it generates one and keeps it on the `Agent` instance. A chat server with one module-level agent then puts every user's conversation into the same Maple session. This guide covers Agno 3.0 (tested with 3.0.11) and `openinference-instrumentation-agno` 1.0.10 on Python 3.10 to 3.14. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-agno](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-agno) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Agno in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-agno -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install the instrumentor and export to Maple + +```bash +pip install -U "agno>=3.0" "openinference-instrumentation-agno>=1.0.10" \ + opentelemetry-sdk opentelemetry-exporter-otlp-proto-http +``` + +Version 1.0.8 of the instrumentor is the first that traces resumed human-in-the-loop runs, and 1.0.10 adds cost. Older versions work, with the gaps listed under [Troubleshooting](#troubleshooting). + +Configure the exporter with the standard OpenTelemetry variables: + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +export AGNO_TELEMETRY=false +``` + +EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` to the endpoint itself. `AGNO_TELEMETRY=false` turns off Agno's anonymous usage pings to agno.com, which have nothing to do with your traces. + +Then create one tracer provider at startup: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.agno import AgnoInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +# Resource comes from OTEL_SERVICE_NAME / OTEL_RESOURCE_ATTRIBUTES, +# endpoint and key from OTEL_EXPORTER_OTLP_*. +provider = TracerProvider() +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +AgnoInstrumentor().instrument( + tracer_provider=provider, + # Also write gen_ai.* attributes, which Maple's session detail page reads + config=TraceConfig(enable_genai_semconv=True), +) +``` + +Import `tracing` at the top of your entry point, before you build agents or an `AgentOS`. The instrumentor patches Agno's run functions and every model class in `agno.models`, so agents created later are traced without further changes. + +`enable_genai_semconv=True` is not optional for Maple. Without it the spans only carry OpenInference attributes (`llm.input_messages.0.message.content`, `llm.token_count.prompt`). The Agent Sessions list still shows tokens and models from those, but the session detail page shows no transcript. With it, each span also gets `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.agent.name`. The environment variable `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same thing if you can't change the `instrument()` call. + +### If you use AgentOS tracing + +`AgentOS(tracing=True)` and `agno.tracing.setup_tracing(db=...)` install the same instrumentor with a `DatabaseSpanExporter`, which is what feeds the AgentOS traces view. Both skip their setup when a real `TracerProvider` is already registered. So if `tracing.py` runs first, AgentOS stops writing traces to its database. + +To keep both, add Agno's database exporter to your provider next to the OTLP one, with the same `db` you give AgentOS: + +```py +from agno.tracing.exporter import DatabaseSpanExporter + +provider.add_span_processor(BatchSpanProcessor(DatabaseSpanExporter(db=db))) +``` + +Don't call `AgnoInstrumentor().instrument()` twice. The second call logs "Attempting to instrument while already instrumented" and does nothing, so its `config` is ignored. + +## Group each conversation into one session + +Every `agent.run()`, `team.run()` and their async versions start a new trace. Maple joins those traces into a session using the `session.id` attribute on the run span (`support_agent.run`). Agno sets it from the `session_id` argument, so pass your conversation id on every call: + +```py +from agno.agent import Agent +from agno.db.sqlite import SqliteDb +from agno.models.openrouter import OpenRouter + +# Built once at import time and shared by every request +agent = Agent( + name="support_agent", + model=OpenRouter(id="openai/gpt-4o-mini"), + db=SqliteDb(db_file="tmp/agno.db"), + tools=[get_weather, calculate], + add_history_to_context=True, +) + + +def chat(conversation_id: str, user_id: str, message: str) -> str: + response = agent.run(message, session_id=conversation_id, user_id=user_id) + return response.content + + +async def chat_stream(conversation_id: str, user_id: str, message: str): + async for event in agent.arun( + message, stream=True, session_id=conversation_id, user_id=user_id + ): + if getattr(event, "content", None): + yield event.content +``` + +If you leave it out, Agno generates a UUID on the first run and assigns it to `agent.session_id`, and every later run of that `Agent` object reuses it. That's right for a script that creates one agent per conversation. It's wrong for a server that builds the agent once at import time, where every user's messages land in a single Maple session that grows forever. + +`session_id` is also how Agno finds the conversation's history in the agent's `db`, so the id you pass for tracing is the same one you need for memory. Members of a `Team` inherit the team's session id, and `continue_run()` takes `session_id=` too. + +Maple reads `session.id` for Agno spans. With `enable_genai_semconv=True` the root span also carries `gen_ai.conversation.id` with the same value, which Maple ignores for Agno. You don't need `using_session()` from `openinference.instrumentation`: Agno already sets the id on the run span, and one span per trace is enough. + +## Record prompts, responses and tool calls + +The OpenInference instrumentor records content by default. Each model span carries the full message list sent to the model (system prompt, history, tool results) and the model's reply, including tool calls. Each tool span carries the arguments and the result. With `enable_genai_semconv=True` they are written as `gen_ai.input.messages` and `gen_ai.output.messages` in the `[{role, parts}]` format Maple renders as a transcript. + +Maple reads span attributes only. The instrumentor emits no span events or OTLP logs, so nothing else needs enabling. + +To keep content out of Maple, set OpenInference's masking variables before the instrumentor starts: + +```bash +export OPENINFERENCE_HIDE_INPUT_MESSAGES=true # prompts sent to the model +export OPENINFERENCE_HIDE_OUTPUT_MESSAGES=true # model replies +export OPENINFERENCE_HIDE_INPUTS=true # run and tool inputs +export OPENINFERENCE_HIDE_OUTPUTS=true # run and tool outputs +``` + +Masked values are replaced with `__REDACTED__` before the span is exported, so they never leave your process. Sessions still show models, tokens, tools and failures, with an empty transcript. One exception: tool arguments (`tool.parameters` and `gen_ai.tool.call.arguments`) are not covered by any of these variables and are still exported. To drop them, or for finer control such as redacting emails but keeping the rest, run an OpenTelemetry Collector with a `redaction` or `transform` processor between your app and Maple. + +Tool results are recorded as `str(result)`, which is also what Agno sends back to the model. A tool that returns a `dict` shows up as a Python repr (`{'city': 'Berlin'}`), which isn't valid JSON, so Maple shows it as plain text. Return a JSON string from tools that produce structured data: + +```py +import json + +from agno.tools import tool + + +@tool +def get_weather(city: str) -> str: + """Get the current weather for a city.""" + return json.dumps({"city": city, "temperature_c": 21, "condition": "partly cloudy"}) +``` + +## Tools, errors and team members + +Each tool call is its own span, named after the function (`get_weather`), with `gen_ai.operation.name=execute_tool` and `gen_ai.tool.name`. When a tool raises, Agno catches the exception and hands the message back to the model, but the instrumentor still marks the tool span as failed with status `ERROR` and the exception text as its message. Maple counts it as a failed tool call and groups repeated failures by that message. The run span stays OK because the run itself continued. + +A tool that returns an error string instead of raising looks like a success. If you want a failure to count, raise. + +A `Team` delegates to its members through a `delegate_task_to_member` tool: + +```py +from agno.team import Team, TeamMode + +weather_worker = Agent(name="weather_worker", model=primary, tools=[get_weather]) +budget_worker = Agent(name="budget_worker", model=secondary, tools=[calculate]) +transport_worker = Agent(name="transport_worker", model=primary, tools=[fetch_transport_data]) + +team = Team( + name="travel_team", + mode=TeamMode.coordinate, + model=primary, + members=[weather_worker, budget_worker, transport_worker], +) + +await team.arun("Produce a mini briefing about Amsterdam.", session_id=conversation_id) +``` + +The whole team run is one trace. The members' runs sit directly under the leader's run, next to the `delegate_task_to_member` tool spans rather than inside them: + +```text +travel_team.arun invoke_agent (team leader) +├─ OpenRouter.ainvoke chat +├─ delegate_task_to_member execute_tool +├─ delegate_task_to_member execute_tool +├─ weather_worker.arun invoke_agent +│ ├─ OpenRouter.ainvoke chat +│ ├─ get_weather execute_tool +│ └─ OpenRouter.ainvoke chat +├─ transport_worker.arun invoke_agent +│ ├─ OpenRouter.ainvoke chat +│ ├─ fetch_transport_data execute_tool (ERROR) +│ └─ OpenRouter.ainvoke chat +└─ OpenRouter.ainvoke chat (leader's final answer) +``` + +Maple opens a lane for each member because each member span has its own `gen_ai.agent.name`. The `delegate_task_to_member` calls show up as ordinary tool calls on the leader, with the member id and task as arguments. Give every `Agent` and `Team` a `name=`. Unnamed ones produce spans called `Agent.run` or `Team.run` with no agent name, and Maple can't give them a lane. With `team.arun()` in `coordinate` mode, members the leader calls in one step run concurrently, and their spans overlap in time. The sync `team.run()` runs them one after another. + +Instrumentor 1.0.10 leaves the team's span attached to the current context after `team.run()` or `team.arun()` returns. Any agent run that follows in the same thread or asyncio task becomes a child of the finished team run, in its trace. Web frameworks that handle each request in its own task or copied context, such as FastAPI, are not affected. In scripts, workers and loops, give each team run its own context: + +```py +import asyncio +import contextvars + +# async +result = await asyncio.create_task(team.arun(prompt, session_id=conversation_id)) + +# sync +result = contextvars.copy_context().run(team.run, prompt, session_id=conversation_id) +``` + +### Human-in-the-loop approvals + +Agno's confirmation flow pauses the run, and `continue_run()` resumes it. Pass the same `session_id` to both: + +```py +@tool(requires_confirmation=True) +def delete_file(path: str) -> str: + """Delete a file from disk. Destructive: requires user approval.""" + ... + + +response = agent.run(message, session_id=conversation_id) +if response.is_paused: + for requirement in response.active_requirements: + requirement.confirm() # or requirement.reject(note="...") + response = agent.continue_run( + run_id=response.run_id, + requirements=response.requirements, + session_id=conversation_id, + ) +``` + +The paused run and the resumed run are two traces, `support_agent.run` and `support_agent.continue_run`, both carrying the conversation's `session.id`. Maple shows them as two turns of the same session, and the approved tool call appears once, in the second. `continue_run()` needs a `db` on the agent. + +Workflows (`agno.workflow`) are instrumented too and accept `session_id=` the same way. + +## Tokens and cost + +Every model span carries input and output tokens, plus cache read and cache write tokens when the provider reports them. Streaming runs (`stream=True`) record usage as well, because Agno reads it from the final chunk. The run span has no token counts of its own, so nothing is counted twice. + +Reasoning tokens are not broken out into their own attribute, so Maple can't show them separately. Model spans also carry no `gen_ai.response.id` and no time-to-first-token, and tool spans carry no `gen_ai.tool.call.id` (the id is only inside the message parts). Maple still links tool results to calls through the conversation history, but it can't show streaming latency for Agno. + +Cost appears when the model provider returns a price with the response. OpenRouter does, and the instrumentor records it as `llm.cost.total` in USD. Maple never prices tokens itself, so calls to providers that don't return a cost, such as OpenAI or Anthropic called directly, show as unpriced. + +## Flush before short-lived processes exit + +`BatchSpanProcessor` exports every 5 seconds. A long-running server needs nothing extra: the SDK flushes on normal shutdown. Scripts, CLIs, notebooks, queue workers and serverless handlers need an explicit flush, or the last spans are lost: + +```py +from tracing import provider + +try: + agent.run("Summarize today's tickets", session_id=conversation_id) +finally: + provider.force_flush() # serverless: flush at the end of every invocation + provider.shutdown() # scripts: flush and stop at the end of the process +``` + +In a Jupyter notebook, call `provider.force_flush()` after the cell that runs the agent. On AWS Lambda and similar platforms, call `force_flush()` before returning from the handler and never `shutdown()`, since the next invocation reuses the process. + +## Check that it works + +Run one conversation of two or three turns with the same `session_id`, including one tool call. Within a minute, open **Agent Sessions** in Maple and filter by your service name. You should see: + +- **One session per conversation**, with your `session_id` as its id and **Agno** as the framework. A session called `trace:...` means the run span had no `session.id`. +- **One turn per `run()` call**, labelled with the user's message. +- **A transcript** with the system prompt, user messages, assistant replies and tool calls. An empty transcript with non-zero tokens means `enable_genai_semconv` is off. +- **Model calls** named after the Agno model class (`OpenRouter.invoke`, `OpenAIChat.ainvoke`, `Claude.invoke_stream`), with the model id (`openai/gpt-4o-mini`) as the model. +- **Tool calls** named after your functions, with arguments and results, and failed ones marked. +- **Agents** named after your `Agent(name=...)` and `Team(name=...)`, with a lane per team member. +- **Tokens** on every model call, and a cost if your provider returns one. + +A human-in-the-loop approval shows up as two turns in the same session: the run that paused (`support_agent.run`) and the resumed run (`support_agent.continue_run`), each in its own trace. + +## Troubleshooting + +- **Nothing arrives in Maple.** The spans went to AgentOS's database instead. You called `setup_tracing()` or created `AgentOS(tracing=True)` before `tracing.py` ran, so the global provider is Agno's, and `set_tracer_provider` in `tracing.py` logged "Overriding of current TracerProvider is not allowed". Run `tracing.py` first. If only the last few runs are missing, see [Flush before short-lived processes exit](#flush-before-short-lived-processes-exit). +- **Every user's conversation is in one giant session.** The `Agent` is shared and no `session_id=` is passed, so Agno reuses its auto-generated id. Pass `session_id=` on every `run()`, `arun()` and `continue_run()`. +- **Every turn is its own session.** You pass a new id per request, often a fresh `uuid4()`. Use the conversation or thread id your app already stores. +- **Sessions list shows tokens, but the detail page has no transcript or model.** `enable_genai_semconv` is off, so the spans only have OpenInference attributes, which Maple's detail page doesn't decode for Agno yet. Pass `TraceConfig(enable_genai_semconv=True)` or set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true`, and check that `instrument()` isn't being called a second time by other code. +- **Resumed runs show up as loose model and tool calls with no session.** Your instrumentor is older than 1.0.8, which didn't wrap `continue_run()`. Upgrade to 1.0.10. +- **Team members appear as `Agent.run` with no lane.** The member has no `name=`. +- **Every model call appears twice.** A second instrumentor wraps the same client, usually `openinference-instrumentation-openai`, OpenLIT, or Phoenix's `register(auto_instrument=True)` picking up every installed OpenInference package. Keep only the Agno instrumentor. +- **An agent run lands inside the previous team run's trace.** The instrumentor leaves a finished team run's span attached to the thread or task ([agno#5573](https://github.com/agno-agi/agno/issues/5573)). Run each team run in its own context with `asyncio.create_task(...)` or `contextvars.copy_context().run(...)`, as shown in [Tools, errors and team members](#tools-errors-and-team-members). Plain agent runs, streamed or not, don't leak. +- **Tool results show as text instead of JSON.** The tool returns a `dict` or `list`, which Agno records as a Python repr. Return `json.dumps(...)`. +- **"Failed to detach context" in the logs after a streamed team run.** A known instrumentor issue with async generators ([agno#5208](https://github.com/agno-agi/agno/issues/5208)). The spans are still exported; it's log noise. +- **`SqliteDb` fails with "requires that the Python 'greenlet' library is installed".** Agno 3's SQLite store imports SQLAlchemy's asyncio support. SQLAlchemy doesn't pull in `greenlet` on every platform (Apple silicon, for one), so install `agno[sqlite]` and `greenlet`. This matters for tracing because `continue_run()` needs a `db`. +- **A failed tool shows as successful.** The tool returned an error message instead of raising. Raise an exception so the span gets status `ERROR`. +- **No cost on any session.** Your provider doesn't return prices. Maple shows the tokens and marks the session unpriced. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Agent tracing guides](/docs/agent-tracing) +- [Agno tracing documentation](https://docs.agno.com/agent-os/tracing/overview) +- [openinference-instrumentation-agno on PyPI](https://pypi.org/project/openinference-instrumentation-agno/) +- [OpenInference configuration variables](https://arize.com/docs/phoenix/tracing/how-to-tracing/advanced/masking-span-attributes) diff --git a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md new file mode 100644 index 0000000000..c767d41872 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md @@ -0,0 +1,354 @@ +--- +title: "Trace Claude Agent SDK agents and Claude Code sessions with OpenTelemetry" +description: "Turn on the tracing Claude Code has built in, so each Agent SDK conversation or Claude Code session shows up in Maple as one Agent Session with its prompts, model calls, tool calls and tokens." +group: "AI Agents" +order: 14 +navLabel: "Claude Agent SDK & Claude Code" +icon: "claude" +--- + +The Claude Agent SDK has no telemetry of its own. Every `query()` starts the Claude Code CLI as a child process, and the CLI has OpenTelemetry built in: a span per user turn (`claude_code.interaction`), per model request (`claude_code.llm_request`) and per tool call (`claude_code.tool`), plus log events and metrics. Maple recognizes these spans and turns them into Agent Sessions, so there is nothing to install besides the SDK. You configure it with environment variables, the same ones whether you run the TypeScript SDK, the Python SDK or `claude` in your terminal. + +Tracing is the part that goes wrong. Spans are a beta behind `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`, and without it the CLI exports metrics and logs and not a single span, so Agent Sessions stays empty while everything looks configured. The second surprise comes later: Claude's replies and the per-request cost are only on log events, which Maple's session views don't read, so sessions show the prompts and tool calls but no assistant text and no cost. + +This guide covers `@anthropic-ai/claude-agent-sdk` 0.3.283 (TypeScript), `claude-agent-sdk` 0.2.160 (Python) and Claude Code 2.1.283, which both SDKs bundle. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-claude-agent-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-claude-agent-sdk) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for the Claude Agent SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-claude-agent-sdk -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Export Claude Code telemetry to Maple + +Every setup below sets the same variables. What each one does: + +- `CLAUDE_CODE_ENABLE_TELEMETRY=1` turns telemetry on. `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` turns on spans, which are what Agent Sessions are built from. +- `OTEL_TRACES_EXPORTER=otlp` sends the spans. The logs and metrics exporters are optional: they carry cost and Claude's replies, which you can search under **Logs** but which don't appear in the session views. +- `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf` is required. Claude Code has no default protocol, and Maple ingest speaks OTLP over HTTP. +- `OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev` with `OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer YOUR_INGEST_KEY`. The CLI appends `/v1/traces`, `/v1/logs` and `/v1/metrics` itself. EU organizations use `https://ingest.eu.maple.dev`. +- `OTEL_SERVICE_NAME` names the service. Without it every agent reports as `claude-code`. + +Never set an exporter to `console` in an SDK app. The SDK reads the CLI's standard output as its message stream, and console telemetry corrupts it. + +### TypeScript Agent SDK + +```bash +npm install @anthropic-ai/claude-agent-sdk zod +``` + +In TypeScript, `options.env` **replaces** the child's environment instead of adding to it. Spread `process.env` so the CLI keeps `PATH` and `ANTHROPIC_API_KEY`, and drop any inherited trace context (see [Keep each turn in its own trace](#keep-each-turn-in-its-own-trace)): + +```ts +// maple-env.ts +const inherited: Record = { ...process.env } +delete inherited.TRACEPARENT +delete inherited.TRACESTATE + +export const mapleEnv: Record = { + ...inherited, + CLAUDE_CODE_ENABLE_TELEMETRY: "1", + CLAUDE_CODE_ENHANCED_TELEMETRY_BETA: "1", // spans; without it there are none + OTEL_TRACES_EXPORTER: "otlp", + OTEL_LOGS_EXPORTER: "otlp", // optional: cost and replies, under Logs + OTEL_METRICS_EXPORTER: "otlp", // optional: token and cost counters + OTEL_EXPORTER_OTLP_PROTOCOL: "http/protobuf", + OTEL_EXPORTER_OTLP_ENDPOINT: "https://ingest.maple.dev", + OTEL_EXPORTER_OTLP_HEADERS: `Authorization=Bearer ${process.env.MAPLE_INGEST_KEY}`, + OTEL_SERVICE_NAME: "support-agent", + OTEL_RESOURCE_ATTRIBUTES: "deployment.environment.name=production", + OTEL_TRACES_EXPORT_INTERVAL: "1000", + OTEL_LOGS_EXPORT_INTERVAL: "1000", + // Content, off by default. See "Record prompts and tool calls" below. + OTEL_LOG_USER_PROMPTS: "1", + OTEL_LOG_TOOL_DETAILS: "1", + OTEL_LOG_TOOL_CONTENT: "1", +} +``` + +Pass it on every call: + +```ts +import { query } from "@anthropic-ai/claude-agent-sdk" +import { mapleEnv } from "./maple-env" + +for await (const message of query({ prompt: "What changed in the last commit?", options: { env: mapleEnv } })) { + if (message.type === "result") console.log(message.subtype === "success" ? message.result : message.subtype) +} +``` + +### Python Agent SDK + +```bash +pip install claude-agent-sdk +``` + +In Python, `ClaudeAgentOptions.env` is merged on top of the inherited environment, so pass only the telemetry variables. Because it merges, removing an inherited `TRACEPARENT` has to happen on `os.environ`: + +```py +# maple_env.py +import os + +os.environ.pop("TRACEPARENT", None) +os.environ.pop("TRACESTATE", None) + +MAPLE_ENV = { + "CLAUDE_CODE_ENABLE_TELEMETRY": "1", + "CLAUDE_CODE_ENHANCED_TELEMETRY_BETA": "1", # spans; without it there are none + "OTEL_TRACES_EXPORTER": "otlp", + "OTEL_LOGS_EXPORTER": "otlp", # optional: cost and replies, under Logs + "OTEL_METRICS_EXPORTER": "otlp", # optional: token and cost counters + "OTEL_EXPORTER_OTLP_PROTOCOL": "http/protobuf", + "OTEL_EXPORTER_OTLP_ENDPOINT": "https://ingest.maple.dev", + "OTEL_EXPORTER_OTLP_HEADERS": f"Authorization=Bearer {os.environ['MAPLE_INGEST_KEY']}", + "OTEL_SERVICE_NAME": "support-agent", + "OTEL_RESOURCE_ATTRIBUTES": "deployment.environment.name=production", + "OTEL_TRACES_EXPORT_INTERVAL": "1000", + "OTEL_LOGS_EXPORT_INTERVAL": "1000", + # Content, off by default. See "Record prompts and tool calls" below. + "OTEL_LOG_USER_PROMPTS": "1", + "OTEL_LOG_TOOL_DETAILS": "1", + "OTEL_LOG_TOOL_CONTENT": "1", +} +``` + +```py +import asyncio +from claude_agent_sdk import ClaudeAgentOptions, ResultMessage, query +from maple_env import MAPLE_ENV + + +async def main(): + async for message in query( + prompt="What changed in the last commit?", + options=ClaudeAgentOptions(env=MAPLE_ENV), + ): + if isinstance(message, ResultMessage): + print(message.result) + + +asyncio.run(main()) +``` + +Because the CLI inherits the process environment in both SDKs, you can also set these variables in your Dockerfile or deployment manifest and skip `env` entirely. That is what Anthropic recommends for production. You still need to make sure no `TRACEPARENT` is set there. + +Settings files win over `env`. Unless you pass `settingSources` (Python `setting_sources`), the CLI loads `~/.claude/settings.json` and the project's `.claude/settings.json`, and an `env` block in either overrides the same variable in `options.env`. In our test, `OTEL_SERVICE_NAME` from user settings replaced the one the app passed. That matters on a developer machine that also sends its own Claude Code sessions to Maple. A server app that doesn't need file settings can pass `settingSources: []`. + +### Claude Code in your terminal, IDE or the desktop app + +To see your own Claude Code sessions in Maple, put the variables under `env` in `~/.claude/settings.json`: + +```json +{ + "env": { + "CLAUDE_CODE_ENABLE_TELEMETRY": "1", + "CLAUDE_CODE_ENHANCED_TELEMETRY_BETA": "1", + "OTEL_TRACES_EXPORTER": "otlp", + "OTEL_LOGS_EXPORTER": "otlp", + "OTEL_METRICS_EXPORTER": "otlp", + "OTEL_EXPORTER_OTLP_PROTOCOL": "http/protobuf", + "OTEL_EXPORTER_OTLP_ENDPOINT": "https://ingest.maple.dev", + "OTEL_EXPORTER_OTLP_HEADERS": "Authorization=Bearer YOUR_INGEST_KEY", + "OTEL_LOG_USER_PROMPTS": "1", + "OTEL_LOG_TOOL_DETAILS": "1", + "OTEL_LOG_TOOL_CONTENT": "1" + } +} +``` + +Start a new `claude` session to pick it up. Terminal sessions report `service.name=claude-code`, and sessions from the desktop app's Code tab report `claude-code-desktop`. + +It has to be your user settings, your shell, or managed settings. Since 2.1.282, Claude Code ignores the variables that turn export on, set the endpoint or capture content (`CLAUDE_CODE_ENABLE_TELEMETRY`, `OTEL_LOG_*` and the like) in a repository's `.claude/settings.json` and `.claude/settings.local.json`, so a repo can't turn export on or pick where it goes. `/status` lists any it ignored. To roll this out to a team, put the same `env` block in [managed settings](https://code.claude.com/docs/en/managed-settings). + +## Group turns into one session + +Maple groups Claude Code spans by their `session.id` attribute, which the CLI puts on every span. One Claude Code session becomes one Maple session, with one turn per `claude_code.interaction` span. Each turn is its own trace. + +In the terminal that needs no setup. `/clear` starts a new session, `claude --resume` and `--continue` keep the old one, and `--fork-session` starts a new one. + +In the SDK, **every `query()` call starts a new session** unless you resume one. A chat backend that calls `query()` once per user message and doesn't resume shows every message in Maple as its own one-turn session, and the agent has no memory of the previous message either. + +Store a session UUID with each conversation. Pass it as `sessionId` on the first turn, then as `resume` on every turn after that: + +```ts +import { randomUUID } from "node:crypto" +import { query } from "@anthropic-ai/claude-agent-sdk" +import { mapleEnv } from "./maple-env" + +type Conversation = { claudeSessionId?: string } + +export async function reply(conversation: Conversation, text: string) { + const firstTurn = !conversation.claudeSessionId + const sessionId = conversation.claudeSessionId ?? randomUUID() + conversation.claudeSessionId = sessionId + + for await (const message of query({ + prompt: text, + options: { env: mapleEnv, ...(firstTurn ? { sessionId } : { resume: sessionId }) }, + })) { + if (message.type === "result") return message.subtype === "success" ? message.result : undefined + } +} +``` + +```py +import uuid +from claude_agent_sdk import ClaudeAgentOptions, ResultMessage, query +from maple_env import MAPLE_ENV + + +async def reply(conversation: dict, text: str) -> str | None: + first_turn = "claude_session_id" not in conversation + session_id = conversation.setdefault("claude_session_id", str(uuid.uuid4())) + session = {"session_id": session_id} if first_turn else {"resume": session_id} + + async for message in query(prompt=text, options=ClaudeAgentOptions(env=MAPLE_ENV, **session)): + if isinstance(message, ResultMessage): + return message.result + return None +``` + +The session UUID is the Maple session id, so if your conversation ids are UUIDs you can use them directly and find a conversation in Maple by its own id. + +A few things break this: + +- `resume` reads the session's transcript from `~/.claude/projects/` on the machine that ran the earlier turns. If the next message can land on another host, mirror transcripts with the SDK's [`sessionStore`](https://code.claude.com/docs/en/agent-sdk/session-storage) option. +- `forkSession: true` (Python `fork_session=True`) gives the resumed conversation a new session id, so it becomes a new Maple session. +- `OTEL_METRICS_INCLUDE_SESSION_ID=false` removes `session.id` from spans too, and every turn becomes its own session. + +A long-lived process that keeps one session open needs none of this: a TypeScript `query()` fed an async iterable of messages, or a Python `ClaudeSDKClient`, sends every turn in the same session. + +Adding `gen_ai.conversation.id` or `maple_ai.session.id` doesn't help here. You can't add attributes to the CLI's spans, and Maple reads `session.id` for Claude Code. + +### Keep each turn in its own trace + +When your application has an active OpenTelemetry span, both SDKs pass its context to the CLI as `TRACEPARENT`, and each turn's `claude_code.interaction` span becomes a child of your span. That is useful: the agent turn shows up inside the HTTP request that triggered it. + +An **inherited** `TRACEPARENT` is the problem. Claude Code sets one on every command its Bash tool runs, CI systems set them, and the SDK passes the environment through. If you ask Claude Code to run your agent script, every turn of your agent nests inside that Claude Code session's trace. The snippets above drop `TRACEPARENT` and `TRACESTATE` from the inherited environment, and the SDK still injects your own active span's context on top. + +Interactive `claude` sessions ignore an inbound `TRACEPARENT`. Only SDK and `claude -p` runs read it. + +## Record prompts and tool calls + +Claude Code redacts content by default. Each flag adds one kind: + +| Variable | What it adds to the spans | What you see in Maple | +| --- | --- | --- | +| `OTEL_LOG_USER_PROMPTS=1` | `user_prompt` on `claude_code.interaction` (otherwise ``) | Turns titled with the prompt, and the user messages in the transcript | +| `OTEL_LOG_TOOL_DETAILS=1` | `full_command` for Bash, `file_path` for Read, Edit and Write, `subagent_type` for the Agent tool, and the full error message of a failed tool | The Bash command or file path as the tool call's arguments, and the real error on failed calls | +| `OTEL_LOG_TOOL_CONTENT=1` | A `tool.output` span event with what the tool returned | The tool call's result, for Read, Bash, Edit and Write (Edit and Write also need `OTEL_LOG_TOOL_DETAILS`), and from 2.1.283 for MCP tools, including your SDK tools, WebFetch and WebSearch | + +Some content never reaches the session views, whatever you set: + +- **Claude's replies.** They are only on the `assistant_response` log event (`OTEL_LOG_ASSISTANT_RESPONSES`, which follows `OTEL_LOG_USER_PROMPTS` when unset). With the logs exporter on, you can read them under **Logs**. The transcript shows the prompts and tool calls, not the answers. +- **The system prompt.** +- **Arguments of other tools.** Maple shows the Bash command and the file path. The arguments of MCP tools, including the ones you define with the SDK, are only in the `tool_result` log event. + +Anthropic's detailed beta tracing (`ENABLE_BETA_TRACING_DETAILED` with `BETA_TRACING_ENDPOINT`) adds more content attributes to spans, but it also redirects your logs and traces to that endpoint, and Maple doesn't read the attributes it adds. Leave it off. + +### Privacy + +The flags are per category, so `OTEL_LOG_TOOL_CONTENT=1` sends whatever files Claude reads and whatever commands print, secrets included. Turn on only what your Maple organization is allowed to store. Content is truncated at 60 KB per attribute (`CLAUDE_CODE_OTEL_CONTENT_MAX_LENGTH`), with a `[TRUNCATED ...]` marker. + +Terminal sessions signed in with a Claude account also carry `user.email` on every span and event. To drop or mask attributes before they leave your network, send the data through an OpenTelemetry Collector with a `redaction` or `attributes` processor and point the Collector at Maple. + +The ingest key can only write telemetry, so it is safe in an environment variable or settings file. Keep private `maple_sk_` keys out of both. + +## Tools, errors and sub-agents + +Every `claude_code.tool` span is one tool call, named by its `tool_name`: `Bash`, `Read` and `Edit` for built-ins, `mcp____` for MCP tools and the ones you define with `createSdkMcpServer`, and `Agent` for a sub-agent. The two phases inside a tool call, `claude_code.tool.blocked_on_user` (the permission wait) and `claude_code.tool.execution` (the run), are shown in the trace but not counted as tool calls of their own. + +A tool fails when its execution does. When a tool handler throws, or returns `isError: true`, the CLI marks `claude_code.tool.execution` with `success=false`, and Maple marks the tool call failed with `error.type` set to the CLI's `error_class` (`McpToolCallError` for SDK tools) and the error as its result. Without `OTEL_LOG_TOOL_DETAILS=1` the error is only that class name; with it, it is the full message your tool threw, such as `transport data service unavailable (503)`. + +A call the user or `canUseTool` rejects has a `blocked_on_user` span with `decision=reject` and no execution. Maple shows it as a call without a result, not as a failure. + +A failed model request (`success=false` on `claude_code.llm_request`, with `status_code` and `error`) counts as a failed LLM call. + +Sub-agents, defined with the `agents` option or in `.claude/agents/`, run through the `Agent` tool. Their model and tool calls nest under that `Agent` tool call in the same trace, so a whole delegation is one turn, and parallel sub-agents show up as overlapping calls. + +Claude Code can also run sub-agents in the background. In our tests, 2.1.283 did that for SDK `agents` even though the model never asked for it. The parent turn then ends right away, and each sub-agent that finishes starts a new turn in the same session, with a `` block as its prompt. Your `query()` loop receives one `result` per turn instead of one in total, and SDK tool calls made by background sub-agents failed with "The tool call was interrupted before a result was received", which Maple counts as failed tool calls. If your app expects one answer per `query()`, keep sub-agents in the foreground: + +```ts +for await (const message of query({ + prompt, + options: { agents, env: { ...mapleEnv, CLAUDE_CODE_DISABLE_BACKGROUND_TASKS: "1" } }, +})) { + if (message.type === "result") console.log(message.subtype === "success" ? message.result : message.subtype) +} +``` + +With it, the whole delegation is one turn and one trace, and parallel sub-agents still run side by side. + +What you don't get is a lane per sub-agent. Maple opens a lane for each distinct `gen_ai.agent.name`, and Claude Code puts no agent name on its spans, only an opaque `agent_id` and, with `OTEL_LOG_TOOL_DETAILS=1`, the `subagent_type` on the `Agent` call. The agent filter in Agent Sessions stays empty for Claude Code sessions for the same reason. + +## Tokens and cost + +Each `claude_code.llm_request` span carries `input_tokens`, `output_tokens`, `cache_read_tokens` and `cache_creation_tokens`, plus `ttft_ms` and the stop reason. Maple maps them to its token buckets and time to first token. Anthropic's `input_tokens` excludes both cache buckets, and Maple counts it that way, so total input is the sum of the three. The CLI always streams and still records usage, so there is no streaming gap to work around. + +Maple doesn't show cost for these sessions. Maple never prices tokens itself, and Claude Code puts cost only on the `api_request` log event (`cost_usd`) and the `claude_code.cost.usage` metric, never on a span. Sessions read as **unpriced**. + +To track spend anyway, keep the logs and metrics exporters on. The `claude_code.api_request` records are searchable under **Logs** with their `cost_usd` and `session.id`, and `claude_code.cost.usage` can go on a dashboard. Both are Claude Code's client-side estimate at list price, unless your organization sets `modelPricing` in managed settings. In an SDK app, the result message's `total_cost_usd` is the same estimate for one `query()`. + +## Flush short-lived processes + +The CLI exports spans every 5 seconds and flushes when it exits cleanly, but that last flush has a short timeout. In the SDK, a `query()` with a string prompt starts a CLI process that exits when the turn ends, so every turn ends with that flush. + +- Set `OTEL_TRACES_EXPORT_INTERVAL` and `OTEL_LOGS_EXPORT_INTERVAL` to `1000`, as in the snippets above, so most spans leave during the turn. +- Let the message loop run to the `result` message. Returning from the loop at `result`, as `reply()` does, is fine. Calling `close()` on a query, aborting it, or breaking out of the loop before that kills the CLI before it flushes. +- In a script, keep the process alive for about 5 seconds after the last `query()` finishes, so the CLI's final export completes. +- On serverless platforms, finish the loop before you return the response. A frozen instance can't flush. +- For `claude -p` in CI, set the same two intervals. + +## Check that it works + +Run one conversation with two turns and at least one tool call, for example with the `reply()` function above. Then open **Agent Sessions** in Maple. You should see: + +- **One session** for the conversation, with the framework shown as **Claude Agent SDK** (Claude Code sessions from the terminal show the same name) and your `OTEL_SERVICE_NAME` as the service. +- **One turn per message**, titled with the user's prompt. +- The model ID as Claude Code sent it (for example `anthropic/claude-haiku-4.5` when calling through OpenRouter), the LLM call count, and the four token buckets. +- The tool calls by name, with arguments and results as described in [Record prompts and tool calls](#record-prompts-and-tool-calls), and failed tools marked failed. +- Cost shown as **unpriced**, and no assistant text in the transcript. Both are expected, as explained above. + +How the spans map: + +| Claude Code span | In Agent Sessions | +| --- | --- | +| `claude_code.interaction` | A turn, titled with the prompt | +| `claude_code.llm_request` | A model call: model, tokens, time to first token, finish reason, failure | +| `claude_code.tool` | A tool call: name, arguments, result, failure | +| `claude_code.tool.blocked_on_user`, `claude_code.tool.execution` | Part of their tool call in the trace view, not calls of their own | + +With the logs exporter on, **Logs** has one `claude_code.user_prompt`, one `claude_code.api_request` per model call and one `claude_code.tool_result` per tool run for the same `session.id`. + +## Troubleshooting + +- **Metrics and logs arrive, but no traces and no sessions.** `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` or `OTEL_TRACES_EXPORTER=otlp` is missing. Both are required for spans. +- **Nothing arrives at all.** `OTEL_EXPORTER_OTLP_PROTOCOL` is unset (there is no default) or set to `grpc`. Set `http/protobuf`. To see export errors, set `CLAUDE_CODE_OTEL_DIAG_STDERR=1` and read the SDK's `stderr` callback, or run `claude --debug-file /tmp/claude.log` and look for `[3P telemetry]` lines. +- **The CLI ignores your settings.** They are in the repository's `.claude/settings.json`, which Claude Code ignores for telemetry since 2.1.282. Move them to `~/.claude/settings.json` or your shell. `/status` lists what it ignored. +- **401 errors, or data going to another backend.** Managed settings, or an org-distributed `~/.claude/remote-settings.json`, set `OTEL_EXPORTER_OTLP_ENDPOINT` or `OTEL_EXPORTER_OTLP_HEADERS`, and managed values win over yours. Ask whoever manages Claude Code in your organization, or add Maple to the managed configuration. +- **Every message is its own one-turn session.** Each `query()` starts a new session. Resume the conversation's session id as in [Group turns into one session](#group-turns-into-one-session), and check that `forkSession` is off and `OTEL_METRICS_INCLUDE_SESSION_ID` isn't `false`. +- **Your agent's turns appear inside another trace.** The process inherited a `TRACEPARENT`, typically because Claude Code or CI started it. Drop `TRACEPARENT` and `TRACESTATE` from the environment you pass to the SDK. +- **TypeScript: the CLI fails to start or can't authenticate after adding `env`.** `options.env` replaced the whole environment. Spread `process.env` into it. +- **SDK output is garbled or `query()` throws a JSON parse error.** An exporter is set to `console`. Use `otlp` or `none`. +- **Turns are untitled and the transcript has no user messages.** `OTEL_LOG_USER_PROMPTS=1` is missing, so the prompt is ``. +- **Your SDK tools have no results.** You need Claude Code 2.1.283 or later (Agent SDK 0.3.283 for TypeScript, 0.2.160 for Python) and `OTEL_LOG_TOOL_CONTENT=1`. +- **The last turn of a script is missing, or ends at a model call.** The process exited before the CLI's final export. See [Flush short-lived processes](#flush-short-lived-processes). +- **Extra turns titled ``, and sub-agent tool calls failed with "interrupted before a result was received".** Claude Code ran the sub-agents in the background. Set `CLAUDE_CODE_DISABLE_BACKGROUND_TASKS=1` as in [Tools, errors and sub-agents](#tools-errors-and-sub-agents). +- **The service name, endpoint or content flags aren't the ones you passed in `env`.** An `env` block in `~/.claude/settings.json` or `.claude/settings.json` overrides `options.env`. Pass `settingSources: []`, or remove the keys from the settings file. +- **Every model call appears twice.** You also installed a hook-based instrumentor for the SDK, such as OpenInference's. Remove it; the CLI's own spans are the ones Maple groups by session. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Trace your AI agent](/docs/agent-tracing): guides for other frameworks. +- [Provider SDKs](/docs/agent-tracing/provider-sdks): tracing direct calls with the Anthropic SDK instead of the Agent SDK. +- Claude Code [Monitoring reference](https://code.claude.com/docs/en/monitoring-usage): every variable, span attribute and event. +- Agent SDK [Observability with OpenTelemetry](https://code.claude.com/docs/en/agent-sdk/observability). diff --git a/apps/landing/src/content/docs/agent-tracing/crewai.md b/apps/landing/src/content/docs/agent-tracing/crewai.md new file mode 100644 index 0000000000..7fd6ba2e48 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/crewai.md @@ -0,0 +1,272 @@ +--- +title: "Trace CrewAI crews and flows with OpenTelemetry" +description: "Send CrewAI crews and flows to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools and an agent lane per crew member." +group: "AI Agents" +order: 21 +navLabel: "CrewAI" +icon: "crewai" +--- + +CrewAI sends nothing to your OpenTelemetry backend on its own. Its built-in telemetry is anonymous usage analytics that goes to CrewAI on a private tracer provider, and its OpenTelemetry export is a CrewAI AMP feature. The traces come from OpenInference: `openinference-instrumentation-crewai` records crews, flows, tasks and tools, and a second instrumentor for the SDK CrewAI calls records the model calls, prompts and tokens. + +Most broken CrewAI traces are missing that second instrumentor. With only the CrewAI one, you get agent and tool spans with no model, no tokens and no transcript. The other gap is the conversation: CrewAI has no chat thread, so every `kickoff()` is its own trace with nothing linking it to the previous message. This guide covers CrewAI 1.15 with `openinference-instrumentation-crewai` 1.1.18 and `openinference-instrumentation-openai` 0.1.61, on Python 3.10 to 3.13. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-crewai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-crewai) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for CrewAI in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-crewai -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install the instrumentors and export to Maple + +```bash +pip install "crewai>=1.15" "openinference-instrumentation-crewai>=1.1.18" \ + "openinference-instrumentation-openai>=0.1.61" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +The OpenAI instrumentor is right for most apps because CrewAI 1.x calls most providers through the `openai` SDK. Which instrumentor you need depends on the model string you pass to `LLM(...)`: + +| Model string | SDK CrewAI calls | Instrumentor | +| --- | --- | --- | +| `openai/…`, `openrouter/…`, `deepseek/…`, `ollama/…`, `custom_openai=True`, or a bare name like `gpt-4.1-mini` | `openai` | `openinference-instrumentation-openai` | +| `anthropic/…` or a bare `claude-…` | `anthropic` | `openinference-instrumentation-anthropic` | +| `gemini/…` or a bare `gemini-…` | `google-genai` | `openinference-instrumentation-google-genai` | +| `bedrock/…` | `boto3` | `openinference-instrumentation-bedrock` | +| Anything else (needs `crewai[litellm]`) | `litellm` | `openinference-instrumentation-litellm` | + +Instrument only the SDKs your crews actually call. The Arize Phoenix CrewAI page still recommends the LiteLLM instrumentor, which records nothing when CrewAI uses its native providers. + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_SERVICE_NAME=support-crew +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export CREWAI_DISABLE_TELEMETRY=true +export CREWAI_TRACING_ENABLED=false +``` + +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL: `OTLPSpanExporter()` with no arguments appends `/v1/traces`. An `endpoint=` passed in code is used as is, so it must end in `/v1/traces`. + +The last two variables turn off CrewAI's own pipelines. `CREWAI_DISABLE_TELEMETRY` stops the anonymous analytics export to `telemetry.crewai.com`. `CREWAI_TRACING_ENABLED=false` stops the CrewAI AMP trace uploader and its first-run "view your traces" prompt, which waits for input at the end of a run. Don't use `OTEL_SDK_DISABLED=true`, which CrewAI's telemetry docs also mention: it disables the OpenTelemetry SDK for the whole process, and your Maple traces with it. + +Then add a `tracing.py` and import it at the top of your entry point, before you build any crew: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.crewai import CrewAIInstrumentor +from openinference.instrumentation.openai import OpenAIInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class CrewAIAgentNames(SpanProcessor): + """Copies each CrewAI agent's role to gen_ai.agent.name, which Maple uses for agent lanes.""" + + def on_start(self, span, parent_context=None): + # The instrumentor records the role (graph.node.id) just after the agent span starts, + # so name the agent span when its first child starts, while it's still open. + parent = trace.get_current_span(parent_context) + attrs = getattr(parent, "attributes", None) or {} + role = attrs.get("graph.node.id") + if role and "gen_ai.agent.name" not in attrs and parent.is_recording(): + parent.set_attribute("gen_ai.agent.name", role) + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(CrewAIAgentNames()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +config = TraceConfig(enable_genai_semconv=True) +CrewAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) +OpenAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) +``` + +What each part does: + +- **`enable_genai_semconv=True`** makes both instrumentors write the OpenTelemetry GenAI attributes (`gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*`, `gen_ai.conversation.id`) next to their OpenInference ones when each span ends. Maple's session page reads the GenAI names for CrewAI's agent and tool spans, so without it those spans have no operation or tool details, and model calls have no message transcript. The environment variable `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same if it's set before `instrument()` runs. +- **`CrewAIAgentNames`** fills a gap in the CrewAI instrumentor, which puts the agent's role in the span name and in `graph.node.id` but never in an agent-name attribute. Without it, every agent in a crew shares one lane in Maple. +- **`skip_dep_check=True`** stops an instrumentor from skipping itself, with only an error log, when its version check disagrees with your installed CrewAI or `openai` package. + +Both instrumentors patch classes in place, so `instrument()` only has to run before the first `kickoff()`. If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second one. Add `CrewAIAgentNames()` and the OTLP exporter to the existing provider and pass that provider to `instrument()`. + +## Group a conversation into one session + +A `Crew` runs its tasks once and returns. There's no thread or session id, and the instrumentors set none: `crew_id` changes with every `Crew` object and `crew_key` is the same for every user of the same crew, so neither works as a conversation id. Maple groups traces into a session by `session.id`, which the instrumentors set only inside OpenInference's `using_session` context manager. + +In a chat backend, build the crew per message, put the user's message first in the task description, and run it inside `using_session` with the conversation id your app already stores: + +```py +import tracing # noqa: F401 (first import) + +from crewai import LLM, Agent, Crew, Task +from openinference.instrumentation import using_session + +llm = LLM(model="openai/gpt-4o-mini", temperature=0) + + +def build_crew(text: str, history: str, stream: bool = False) -> Crew: + assistant = Agent( + role="assistant", + goal="Answer the user's questions", + backstory="You are a concise, helpful assistant.", + llm=llm, + tools=[get_weather, calculate], + ) + task = Task( + description=f"{text}\n\nConversation so far:\n{history}", + expected_output="A short, direct reply to the user.", + agent=assistant, + name="reply", + ) + return Crew(name="support", agents=[assistant], tasks=[task], stream=stream) + + +def handle_message(conversation_id: str, text: str, history: str) -> str: + with using_session(conversation_id): + return build_crew(text, history).kickoff().raw +``` + +Each `kickoff()` is one trace and one turn in the session. CrewAI doesn't remember earlier messages, so pass the history yourself, as above. Maple labels each turn with the first line of the prompt CrewAI builds, `Current Task: `, which is why the user's message goes first. + +Give the crew a `name`. An unnamed crew's root span is `Crew_.kickoff`, a different name on every request. + +If you skip `using_session`, every message shows up in **Agent Sessions** as its own one-turn session named after its trace id. Setting `gen_ai.conversation.id` on your own spans doesn't help, because Maple reads `session.id` for CrewAI. + +`using_session` also works for conversational flows, where CrewAI's own session id is the flow's `state.id`. Use the same value for both: + +```py +with using_session(conversation_id): + reply = support_flow.handle_turn(text, session_id=conversation_id) +``` + +Each `handle_turn` runs one `kickoff()`, so each message is a trace under the flow's `.kickoff` span. + +### Async kickoffs + +Use `kickoff()` or `await crew.kickoff_async()`. Don't use `await crew.akickoff()`: it runs a separate native-async code path that the CrewAI instrumentor doesn't patch, so there's no crew or agent span and every model call and tool call becomes its own trace. `kickoff_async()` runs the instrumented `kickoff()` in a thread and keeps the session and trace. + +### Streaming + +`Crew(stream=True)` returns a `CrewStreamingOutput` right away and runs the crew again in a thread once you iterate it. The instrumentor records both calls, so each streamed message produces two traces: an empty `support.kickoff` and the real one. Wrap the turn in one span of your own so both land in one trace and one turn: + +```py +from opentelemetry import trace + +tracer = trace.get_tracer("chat") + + +def stream_message(conversation_id: str, text: str, history: str, send) -> None: + with using_session(conversation_id), tracer.start_as_current_span( + "invoke_agent support", + attributes={ + "gen_ai.operation.name": "invoke_agent", + "gen_ai.conversation.id": conversation_id, + }, + ): + for chunk in build_crew(text, history, stream=True).kickoff(): + send(chunk.content) +``` + +`gen_ai.operation.name` makes Maple treat the wrapper as the turn's agent span, so the two crew spans under it don't count as two turns. `LLM(stream=True)` without `Crew(stream=True)` streams inside a single `kickoff()` and needs none of this. + +## Record prompts, responses and tool calls + +Content capture is on by default. Each model call carries the messages CrewAI sent (the agent's role, goal and backstory as the system message, then `Current Task: …` with the context from earlier tasks) and the model's reply, including tool calls. Agent spans carry the task and its output, and tool spans carry the tool's arguments and result. With the GenAI dual-write on, Maple renders the model calls as the session transcript. + +To keep prompts and outputs out of your traces, add the hide switches to the config both instrumentors share: + +```py +config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) +``` + +`hide_inputs` drops the input messages and replaces `input.value` with `__REDACTED__`, and `hide_outputs` does the same for outputs. The session keeps its turns, model and tool calls, tokens and failures, with an empty transcript. `hide_input_text` and `hide_output_text` keep the message structure but redact the text. Each switch also has an `OPENINFERENCE_HIDE_*` environment variable. + +Agent roles, task names and tool names are always recorded, because they're span names. CrewAI's analytics, if you leave them on, also collect roles and tool names, so keep personal data out of both. + +## Tools, errors and agents + +Each tool call is a `.run` span with `gen_ai.operation.name` `execute_tool` and the tool's name in `gen_ai.tool.name`. A tool that raises is marked failed without extra code: the span ends with status `ERROR` and the exception message, for example `transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. CrewAI then sends `Error executing tool: …` back to the model, and the agent and crew spans stay `OK` because the crew itself carried on. + +Tool spans come from CrewAI's native function calling, which every provider in the table above uses. A model that CrewAI drives with text-based ReAct prompts instead (some custom `BaseLLM` subclasses and LiteLLM models without function calling) runs tools through a path the instrumentor doesn't patch, so those calls have no tool span. + +Each task is an agent span named `.._execute_core`, with `gen_ai.operation.name` `invoke_agent`. `CrewAIAgentNames` gives it the role as `gen_ai.agent.name`, and Maple opens a lane for each agent in the crew. Tasks with `async_execution=True` run in threads; the instrumentor copies the trace context into them, so parallel tasks stay in the same trace as siblings under the crew span. + +In a hierarchical crew (`process=Process.hierarchical`), CrewAI adds a manager agent called `Crew Manager` that delegates with the `Delegate work to coworker` and `Ask question to coworker` tools. The delegated work runs through `Agent.execute_task`, which the instrumentor doesn't patch, so the coworker's model calls appear inside the delegation's tool span with no agent span of their own and no lane. The manager's own task is an agent span like any other. + +In a flow, `.kickoff` is the root and each `@start`, `@listen` and `@router` method is a `.` span, with any crew or `Agent.kickoff()` it runs nested inside. A flow paused by `@human_feedback` and continued with `flow.resume(...)` continues in a new request, and `resume()` isn't patched, so wrap it in `using_session` with the same id and a span of your own, like the streaming wrapper above. + +## Tokens and cost + +Every model call carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals, along with cached and reasoning tokens when the provider reports them. Crew and agent spans carry no usage of their own, so nothing is counted twice. The model is the one the provider returned, such as `openai/gpt-4o-mini` behind OpenRouter. + +Streamed calls keep their token counts: CrewAI's OpenAI provider requests `stream_options={"include_usage": True}` whenever it streams. + +Maple shows cost only when a span carries one. The OpenAI, Anthropic and Gemini instrumentors never record cost, so those sessions show as **unpriced**, with token counts. LiteLLM computes a price for the models it knows, and the LiteLLM instrumentor records it as `llm.cost.total`, which Maple reads. + +`memory=True` and `planning=True` make model calls of their own (memory analysis, embeddings, a planning agent). They appear in the session because they're real, billed calls. + +## Flush spans before the process exits + +`BatchSpanProcessor` exports every 5 seconds, and the `TracerProvider` flushes from an `atexit` handler on a normal interpreter exit. That covers most scripts and CLIs, but not a killed process, `os._exit`, a frozen serverless instance or a notebook. Flush yourself in those cases: + +```py +from tracing import provider + +try: + handle_message("conv-42", "What's the weather in Berlin?", history="") +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. + +## Check that it works + +Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: + +- **One session** for the conversation, with one turn per `kickoff()`. Each turn's trace starts at `support.kickoff` (your crew's name), or `.kickoff` for a flow. +- **The transcript**: the system message built from the agent's role, goal and backstory, `Current Task: …` with your message, and the model's replies. +- **Model calls** named `ChatCompletion` (from the OpenAI instrumentor), each with a model and input and output tokens. +- **Tool calls** named `get_weather.run` and `calculate.run`, with results. +- **Agents**: one lane per role, from `assistant.reply._execute_core` and its siblings. +- **Cost**: unpriced, unless your models go through LiteLLM. + +A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. + +## Troubleshooting + +- **Agent and tool spans, but no model calls or tokens.** The instrumentor for your provider's SDK isn't installed. Match it to the model string with the table above; `openai/…` and `openrouter/…` models need `openinference-instrumentation-openai`, not the LiteLLM one. +- **No spans at all.** `instrument()` never ran, ran after the first kickoff, or skipped itself on its version check. Import `tracing` first, keep `skip_dep_check=True`, and look for an exporter error in the logs. Check that `OTEL_SDK_DISABLED` isn't set. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. +- **One session per message.** The kickoff isn't inside `using_session(...)`, or the id changes per request. Wrap every `kickoff()` and use the stored conversation id. +- **Every model call and tool call is its own trace.** The crew runs with `akickoff()`. Use `kickoff()` or `kickoff_async()`. +- **An empty extra turn for each streamed message.** `Crew(stream=True)` runs the crew twice. Wrap the turn in one span, as in [Streaming](#streaming). +- **Agent and tool spans have no details on the session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)` to both instrumentors. +- **All agents in one lane.** `CrewAIAgentNames` isn't on the provider, or it was added to a different provider than the one passed to `instrument()`. +- **Tool arguments show the tool's input schema.** The dual-write copies `tool.parameters`, which is the schema, into `gen_ai.tool.call.arguments`. The call's actual arguments are in the tool span's `input.value`. This is an OpenInference mapping bug with no workaround in the span processor, because the arguments are set after the span starts. +- **A delegated coworker has no lane.** Hierarchical delegation runs the coworker through `Agent.execute_task`, which isn't instrumented. Its model calls are inside the `Delegate work to coworker` tool span. +- **Every model call appears twice.** Two model-layer instrumentors cover the same call, for example LiteLLM's and OpenAI's with a LiteLLM model that calls the `openai` SDK, or `litellm.callbacks=["otel"]` next to an OpenInference instrumentor. Keep one. +- **The process hangs at exit asking about traces.** CrewAI's first-run trace prompt. Set `CREWAI_TRACING_ENABLED=false`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. +- [CrewAI telemetry](https://docs.crewai.com/en/telemetry): what CrewAI's own analytics collect and how to turn them off. +- [openinference-instrumentation-crewai](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-crewai): the instrumentor's source. +- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/dspy.md b/apps/landing/src/content/docs/agent-tracing/dspy.md new file mode 100644 index 0000000000..bd2e2ff4ed --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/dspy.md @@ -0,0 +1,343 @@ +--- +title: "Trace DSPy programs and ReAct agents with OpenTelemetry" +description: "Send DSPy programs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, cost and failed tools, including modules that run in dspy.Parallel." +group: "AI Agents" +order: 27 +navLabel: "DSPy" +icon: "python" +--- + +DSPy has no OpenTelemetry code of its own. Spans come from OpenInference's `openinference-instrumentation-dspy`, which patches `Module.__call__`, `Predict.forward`, the adapters, `LM.__call__` and `Tool.__call__`. That gives you the shape of a program (which module called which predictor, which tool ran, what the model was sent), but no token counts, no tool names and no conversation id. + +The usual fix for the missing tokens, adding `openinference-instrumentation-litellm`, stopped working in DSPy 3.4. Most models now run on DSPy's own `lm15` engine instead of LiteLLM, so the LiteLLM instrumentor records nothing: in our test, a call to `openrouter/openai/gpt-4o-mini` produced zero LiteLLM spans. This guide fills the gaps with a DSPy callback instead, which works on both engines. It covers DSPy 3.4 with `openinference-instrumentation-dspy` 0.1.45 on Python 3.10 or later, for `dspy.ReAct` agents and your own `dspy.Module` programs. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-dspy](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-dspy) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for DSPy in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-dspy -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install the instrumentor and export to Maple + +```bash +pip install "dspy>=3.4" "openinference-instrumentation-dspy>=0.1.45" "openinference-instrumentation>=0.1.66" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ + "opentelemetry-instrumentation-threading>=0.66b0" +``` + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL: `OTLPSpanExporter()` with no arguments appends `/v1/traces`. An `endpoint=` passed in code is used as is and has to end in `/v1/traces`. + +Add a `tracing.py` and import it at the top of your entry point, before your DSPy modules are defined: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.dspy import DSPyInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.instrumentation.threading import ThreadingInstrumentor +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +DSPyInstrumentor().instrument(tracer_provider=provider, config=TraceConfig(enable_genai_semconv=True)) +ThreadingInstrumentor().instrument() +``` + +- **`enable_genai_semconv=True`** makes the instrumentor also write the OpenTelemetry GenAI attributes (`gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.provider.name`, `gen_ai.request.model`, `gen_ai.tool.call.result`) when each span ends. Maple's session page reads those for DSPy, not the OpenInference originals, so without it the session page has no transcript and no model. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same if it's set before `instrument()` runs. +- **`ThreadingInstrumentor`** carries the trace context into threads. `dspy.Parallel`, `Evaluate` and a plain `ThreadPoolExecutor` start every worker without it, so each worker becomes its own trace with no parent and no session (see [Parallel modules](#parallel-modules-stay-in-one-trace)). + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), don't create a second one. Add the OTLP exporter to the existing provider and pass that provider to `instrument()`. + +### Add the Maple callback + +The instrumentor leaves out four things Maple needs: tokens and cost on model spans, tool names and arguments on tool spans, an agent span per program, and a way to tell DSPy's adapter spans apart from model calls. A DSPy callback runs inside each of those spans, so it can add them. Save this as `maple_dspy.py`: + +```py +# maple_dspy.py +import json + +import dspy +from dspy.utils.callback import BaseCallback +from openinference.instrumentation import TraceConfig +from opentelemetry import trace + +# Same switches as the instrumentor: OPENINFERENCE_HIDE_INPUTS / OPENINFERENCE_HIDE_OUTPUTS. +_config = TraceConfig() + + +def _message(role, values): + text = "\n".join(v for v in values if isinstance(v, str)) + return json.dumps([{"role": role, "parts": [{"type": "text", "content": text}]}]) if text else None + + +class MapleCallback(BaseCallback): + """Adds what Maple reads and the OpenInference DSPy instrumentor leaves out: + an agent span per program, tool names and arguments, tokens and cost.""" + + def __init__(self): + self._agents = set() + self._lms = {} + + def on_module_start(self, call_id, instance, inputs): + if type(instance).__module__.startswith("dspy."): + return # Predict, ChainOfThought, ReAct: building blocks, not agents + self._agents.add(call_id) + span = trace.get_current_span() + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", type(instance).__name__) + user = _message("user", [*inputs.get("args", ()), *inputs.get("kwargs", {}).values()]) + if user and not _config.hide_inputs: + span.set_attribute("gen_ai.input.messages", user) + + def on_module_end(self, call_id, outputs, exception): + if call_id not in self._agents: + return + self._agents.discard(call_id) + if isinstance(outputs, dspy.Prediction) and not _config.hide_outputs: + reply = _message("assistant", [v for k, v in outputs.items() if k != "reasoning"]) + if reply: + trace.get_current_span().set_attribute("gen_ai.output.messages", reply) + + def on_adapter_format_start(self, call_id, instance, inputs): + # The span is "ChatAdapter.__call__"; without an operation, "chat" in the name reads as a model call. + trace.get_current_span().set_attribute("gen_ai.operation.name", "invoke_workflow") + + def on_tool_start(self, call_id, instance, inputs): + span = trace.get_current_span() + span.set_attribute("gen_ai.tool.name", instance.name) + span.set_attribute("gen_ai.tool.description", instance.desc or "") + if not _config.hide_inputs: + span.set_attribute("gen_ai.tool.call.arguments", json.dumps(inputs.get("kwargs", {}), default=str)) + + def on_lm_start(self, call_id, instance, inputs): + self._lms[call_id] = instance + + def on_lm_end(self, call_id, outputs, exception): + lm = self._lms.pop(call_id, None) + # The LM's history holds the provider response. Threads share the LM, so match ours by identity. + entry = next((e for e in reversed(lm.history[-16:]) if e["outputs"] is outputs), None) if lm else None + if entry is None or getattr(entry["response"], "cache_hit", False): + return # history is off, or a cache hit that cost nothing + usage = entry["usage"] or {} + attributes = { + "gen_ai.response.id": getattr(entry["response"], "id", None), + "gen_ai.response.model": entry.get("response_model"), + "gen_ai.usage.input_tokens": usage.get("prompt_tokens"), + "gen_ai.usage.output_tokens": usage.get("completion_tokens"), + "gen_ai.usage.cache_read.input_tokens": (usage.get("prompt_tokens_details") or {}).get("cached_tokens"), + "gen_ai.usage.reasoning.output_tokens": (usage.get("completion_tokens_details") or {}).get("reasoning_tokens"), + "gen_ai.usage.cost": entry.get("cost"), + } + trace.get_current_span().set_attributes({k: v for k, v in attributes.items() if v is not None}) +``` + +Register it with the rest of your DSPy configuration: + +```py +import tracing # first: sets up the provider and patches DSPy + +import dspy +from maple_dspy import MapleCallback + +dspy.configure(lm=dspy.LM("openai/gpt-4o-mini", temperature=0), callbacks=[MapleCallback()]) +``` + +The callback works because the instrumentor patches DSPy's classes from the outside, so DSPy's own callback hooks run while the instrumentor's span is the current one. Any `dspy.Module` subclass you write becomes an agent in Maple, named after its class. DSPy's built-in modules (`Predict`, `ChainOfThought`, `ReAct`) stay plain steps inside it. The GenAI dual-write never overwrites a key that's already set, so the callback's values win. + +### Why not MLflow or the OpenTelemetry DSPy package + +DSPy's documented tracing path is `mlflow.dspy.autolog()`. MLflow can export OTLP and translate to GenAI attributes (`MLFLOW_ENABLE_OTEL_GENAI_SEMCONV`, MLflow 3.11 and later), but it pulls in the whole MLflow package, its GenAI mapping is documented for provider SDK spans, and it has no conversation id Maple can read. + +The OpenTelemetry project published `opentelemetry-instrumentation-genai-dspy` on 24 September 2026 as a 1.2 beta. It emits GenAI attributes directly, but only for tools, ReAct loops and retrieval. It records no model calls, and relies on a provider instrumentor for those, which DSPy 3.4's native engine bypasses. Maple also identifies DSPy by the OpenInference instrumentor, so its spans would show as an unidentified framework. We'll revisit it once it covers model calls. + +## Group a conversation into one session + +DSPy has no session, thread or conversation id. A chat program takes the conversation as a `dspy.History` input field, and every call to your module is a new root span and a new trace. Maple groups a DSPy program's traces into one session by the `session.id` attribute, which the instrumentor only sets inside OpenInference's `using_session` context manager: + +```py +import dspy +from openinference.instrumentation import using_session + + +class ChatAssistant(dspy.Module): + def __init__(self): + super().__init__() + self.respond = dspy.ReAct( + "question: str, conversation: dspy.History -> answer: str", + tools=[get_weather, calculate], + ) + + def forward(self, question, conversation): + return self.respond(question=question, conversation=conversation) + + +assistant = ChatAssistant() + + +def handle_message(conversation_id: str, question: str, turns: list[dict]) -> str: + with using_session(conversation_id): + answer = assistant(question=question, conversation=dspy.History(messages=turns)).answer + turns.append({"question": question, "answer": answer}) + return answer +``` + +Use your app's own conversation id, the one it stores the chat under. A new UUID per request gives you one session per message, and a constant gives every user one shared session. `using_session` puts the id in the OpenTelemetry context, and the instrumentor copies it onto every span it creates inside the block, including spans in worker threads once `ThreadingInstrumentor` is on. + +If you skip this, every call to your module shows up in **Agent Sessions** as its own one-turn session named after its trace id. The GenAI dual-write also copies `session.id` into `gen_ai.conversation.id`, but that key alone doesn't group anything: Maple reads `session.id` for DSPy. + +Don't store the conversation on the module as `self.history`. DSPy appends every model call to a list attribute named `history` on each calling module, so a `dspy.History` there crashes the first model call with `TypeError: object of type 'History' has no len()`. Pass it as an input field, or name the attribute something else. + +## Stream replies with dspy.streamify + +A streamed turn is traced like any other: same session, same spans, tokens and cost on every model call. Two details decide whether that holds. Here is a streaming handler you can return from an SSE or WebSocket endpoint: + +```py +stream_assistant = dspy.streamify(assistant, stream_listeners=[dspy.streaming.StreamListener("answer")]) + + +async def stream_message(conversation_id: str, question: str, turns: list[dict]): + with using_session(conversation_id): + async for chunk in stream_assistant(question=question, conversation=dspy.History(messages=turns)): + if isinstance(chunk, dspy.streaming.StreamResponse): + yield chunk.chunk + elif isinstance(chunk, dspy.Prediction): + turns.append({"question": question, "answer": chunk.answer}) +``` + +- **Call `dspy.streamify` after `dspy.configure(callbacks=[MapleCallback()])`.** `streamify` copies the callback list at the moment you call it. A streamed program built at import time, before `configure` runs, streams without the callback: no tokens, no tool names, and every `ChatAdapter.__call__` counted as a model call. +- **Put `using_session` around the loop that reads the stream.** The program doesn't start until the first chunk is requested. If the `with` block only wraps the `stream_assistant(...)` call, for example in an endpoint that returns `StreamingResponse(stream_assistant(...))`, it has already exited when the model runs, and the turn lands in a session of its own. + +DSPy asks the provider for usage on streamed calls, so the streamed `LM.__call__` span has input and output tokens and cost. In our test, the streamed answer to "What is the capital of France?" recorded 341 input and 47 output tokens. Two things are missing on streamed calls: `gen_ai.response.id`, because DSPy's native engine doesn't keep the id from a stream, and a time-to-first-token attribute, which nothing in this setup records. + +## Record prompts, responses and tool calls + +Content capture is on by default. Every `LM.__call__` span carries the messages DSPy's adapter sent to the model (the signature's instructions as the system message, then the formatted fields) and the raw reply, and every tool span carries the tool's arguments and result. The callback adds a short user and assistant message on each agent span, built from your module's string inputs and outputs, which Maple uses as the turn's title. + +The model messages look like DSPy's prompt format, not like a chat: the user message starts with `[[ ## question ## ]]` and the reply with `[[ ## next_thought ## ]]` or `[[ ## answer ## ]]`. That's what the model actually saw. A `dspy.History` input is expanded into earlier user and assistant messages, so each call repeats the whole conversation so far. + +To keep prompts and outputs out of your traces, set the instrumentor's switches, which the callback also reads: + +```bash +export OPENINFERENCE_HIDE_INPUTS=true +export OPENINFERENCE_HIDE_OUTPUTS=true +``` + +The session still shows its turns, model and tool calls, tokens and failures, with an empty transcript and no tool arguments or results. Set them before `tracing.py` runs. Narrower switches (`OPENINFERENCE_HIDE_INPUT_TEXT`, `OPENINFERENCE_HIDE_OUTPUT_TEXT`, `OPENINFERENCE_HIDE_LLM_INVOCATION_PARAMETERS`) redact parts of each message; the callback's agent messages follow only the two above. + +## Tools, errors and sub-agents + +Each tool call is a span named `.__call__`, of OpenInference kind `TOOL` and `gen_ai.operation.name` `execute_tool`. The callback adds `gen_ai.tool.name`, the tool's docstring as `gen_ai.tool.description`, and its arguments. `dspy.ReAct` also calls a built-in `finish` tool when it's done, so every ReAct run ends with a `finish.__call__` span, and `finish` shows up in Maple's tool list. + +A tool that raises is marked failed with no extra code. The instrumentor ends the tool span with status `ERROR` and the exception as the status message, for example `RuntimeError: transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. `ReAct` catches the exception and hands `Execution error in fetch_transport_data: ...` back to the model, so the `ReAct.forward` span and your module's span stay `OK`, and the program's return value doesn't tell you the tool failed. + +Tool spans have no `gen_ai.tool.call.id`. `ReAct` asks the model for the next tool as text fields (`next_tool_name`, `next_tool_args`), not through the provider's tool-calling API, so there's no id to link. + +DSPy's idiom for sub-agents is module composition: an orchestrator module that calls worker modules. With the callback, every module class you wrote is an `invoke_agent` span with its class name as `gen_ai.agent.name`, and Maple opens a lane for each agent whose name differs from its caller's. Name the classes after what they do (`WeatherWorker`, `BudgetWorker`); two instances of one class share a name and a lane. + +### Parallel modules stay in one trace + +`dspy.Parallel` runs modules in a `ThreadPoolExecutor` and copies only DSPy's own settings into each worker thread, not the OpenTelemetry context. Without `ThreadingInstrumentor`, each worker's spans start a new trace with no parent and no `session.id`: in our test, a three-worker fan-out became four traces, with 61 of 73 spans outside the trace that started them, including the only failed tool. With it, the same program is one trace and one session: + +```py +class Briefing(dspy.Module): + def __init__(self): + super().__init__() + self.weather = WeatherWorker() + self.transport = TransportWorker() + + def forward(self, city): + weather, transport = dspy.Parallel(num_threads=2)( + [(self.weather, {"city": city}), (self.transport, {"city": city})] + ) + return dspy.Prediction(briefing=f"{weather.findings}\n{transport.findings}") +``` + +The same applies to your own `ThreadPoolExecutor` and `threading.Thread`. `ThreadingInstrumentor` doesn't reach `multiprocessing`; a module that runs in another process starts its own trace. + +## Tokens and cost + +The instrumentor records no token counts. The callback reads them from the provider response DSPy keeps in `lm.history` and writes them on the `LM.__call__` span: `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, cached input tokens and reasoning tokens when the provider reports them, plus `gen_ai.response.id` and the model the provider answered with. + +Three cases have no tokens on the span: + +- **Cache hits.** `dspy.LM` caches responses by default (`cache=True`). A repeated prompt is answered from the cache with no provider call, so the callback records nothing for it. Pass `cache=False` while you're checking the numbers. +- **History turned off.** `dspy.configure(disable_history=True)` or `max_history_size=0` stops DSPy from keeping the response the callback reads. +- **A custom engine.** An `LM` with your own `engine=` reports whatever usage your engine puts on its response. + +Cost is DSPy's own estimate, the `cost` field of each history entry, written as `gen_ai.usage.cost`. On the `lm15` engine DSPy prices tokens from its bundled model metadata; on the LiteLLM engine it's LiteLLM's `response_cost`. Maple never prices tokens itself, so when DSPy has no price for a model, that call shows as **unpriced**. Treat the figure as an estimate, not your bill. + +Don't add `openinference-instrumentation-litellm` or `-openai` next to this setup. On DSPy 3.4's native engine they record nothing. On the LiteLLM engine (`dspy.LM(..., engine="litellm")`, and Anthropic models by default) they add a second model span under every `LM.__call__`, and Maple counts each call twice. + +## Flush spans before the process exits + +`BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` flushes on a normal interpreter exit, which covers most scripts and CLIs. That doesn't happen when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: + +```py +from tracing import provider + +try: + handle_message("conv-42", "What's the weather in Berlin?", turns=[]) +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. DSPy optimizers (`MIPROv2`, `GEPA`, `BootstrapFewShot`) and `dspy.Evaluate` make hundreds of calls; trace them in a separate service name or not at all, so they don't bury your production sessions. + +## Check that it works + +Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, with `cache=False` on the LM. Then open **Agent Sessions** in Maple. You should see: + +- **One session** for the conversation, framework **DSPy**, with one turn per call to your module. Each turn's trace starts at `ChatAssistant.forward` (your module's class name), titled with the question you asked. +- **Model calls** named `LM.__call__`, each with a model, input and output tokens. Every model call sits under `Predict.forward`, `Predict(StringSignature).forward` and `ChatAdapter.__call__` spans; those are DSPy's steps, not extra calls. +- **The transcript**: the messages DSPy sent, in its `[[ ## field ## ]]` format, and the replies. +- **Tool calls** named `get_weather.__call__` and `finish.__call__`, with arguments and results. +- **Agents**: one per module class you wrote, with a lane for each worker module. +- **Cost** per call where DSPy has a price for the model. + +A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. + +## Troubleshooting + +- **No spans at all.** `instrument()` never ran, or the exporter can't reach Maple. Import `tracing` first in the entry point and look for an `OTLPSpanExporter` error in the logs. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. +- **Tokens in the list, empty session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)` to `instrument()`. +- **No tokens anywhere.** `MapleCallback` isn't registered, a later `dspy.configure(callbacks=[...])` or `dspy.context(callbacks=[...])` replaced it, or the calls were cache hits. +- **One session per message.** The call isn't inside `using_session(...)`, or the id changes per request. +- **A streamed turn has no tokens, or is a session of its own.** `dspy.streamify` ran before `dspy.configure(callbacks=[MapleCallback()])`, or `using_session` exits before the stream is read. See [Stream replies](#stream-replies-with-dspystreamify). +- **`dspy.Parallel` workers are separate traces with no session.** `ThreadingInstrumentor().instrument()` didn't run. It has to run before the worker threads start. +- **`TypeError: object of type 'History' has no len()`.** A module attribute is named `history`. Rename it, or pass the `dspy.History` as an input field. +- **Every model call counted twice.** A LiteLLM or OpenAI instrumentor is also installed, or the callback isn't registered and `ChatAdapter.__call__` is read as a model call by its name. Remove the extra instrumentor and register the callback. +- **A `finish` tool in every run.** That's `dspy.ReAct`'s built-in tool for ending the loop, not one of yours. +- **Your module shows `OK` after a tool failed.** `ReAct` turns tool exceptions into text for the model. The failed tool span is still marked `ERROR` and counted. +- **Two spans for one step after a parse error.** When `ChatAdapter` can't parse a reply, DSPy retries with `JSONAdapter`, so one `Predict` span holds two adapter spans, each with its own `LM.__call__`. Both calls happened and both are billed. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. +- [DSPy observability](https://dspy.ai/tutorials/observability/): DSPy's own tracing page, built on MLflow. +- [openinference-instrumentation-dspy](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-dspy): the instrumentor's source. +- [DSPy's `BaseCallback`](https://github.com/stanfordnlp/dspy/blob/main/dspy/utils/callback.py): the hook API `MapleCallback` builds on. +- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/google-adk.md b/apps/landing/src/content/docs/agent-tracing/google-adk.md new file mode 100644 index 0000000000..43d6d2910b --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/google-adk.md @@ -0,0 +1,295 @@ +--- +title: "Trace Google ADK agents with OpenTelemetry" +description: "Send Google Agent Development Kit (ADK) traces to Maple with the full transcript, tool arguments and results, tokens, and one session per ADK session." +group: "AI Agents" +order: 22 +navLabel: "Google ADK" +icon: "googleadk" +--- + +Google's Agent Development Kit (ADK) creates OpenTelemetry spans itself, with no instrumentation package: one span per run, per agent, per model call and per tool call, under the instrumentation scope `gcp.vertex.agent`. Every span carries the ADK session id as `gen_ai.conversation.id`, so a multi-turn chat groups into one Maple session without any extra code. + +What goes wrong by default is the transcript. ADK writes prompts, replies and tool payloads into its own `gcp.vertex.agent.llm_request`, `llm_response`, `tool_call_args` and `tool_response` attributes, which Maple doesn't read, so the session shows models and tokens next to an empty conversation. Two environment variables switch ADK to the OpenTelemetry GenAI message format that Maple renders. The other trap: with a plain `Runner`, nothing exports at all until you register a tracer provider yourself. This guide covers ADK for Python 2.10 and later. ADK for Go and Kotlin emit the same span names, but their setup isn't covered here. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-google-adk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-google-adk) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Google ADK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-google-adk -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install ADK and the OTLP exporter + +```bash +pip install "google-adk>=2.10" litellm opentelemetry-exporter-otlp-proto-http +``` + +`litellm` is only needed for non-Gemini models through ADK's `LiteLlm` wrapper. ADK 2.10 pins `opentelemetry-sdk` to 1.42.1 or lower, so let pip pick the exporter version that matches instead of pinning a newer one. + +### Configure the export with environment variables + +```bash +OTEL_SERVICE_NAME=support-agent +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer%20YOUR_INGEST_KEY +# Put prompts, replies and tool calls on span attributes, in the format Maple reads +OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental +OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY +# Drop ADK's own copies of the same content, which Maple doesn't read +ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false +``` + +- The `%20` is an encoded space. OpenTelemetry header values are URL-encoded, and the Python SDK decodes it back to `Bearer YOUR_INGEST_KEY`. +- The exporter appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. If you set `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` instead, give the full `https://ingest.maple.dev/v1/traces`. +- The `proto-http` exporter always sends OTLP over HTTP with protobuf, which is what Maple ingests. Don't install the gRPC exporter. +- For an EU organization, use `https://ingest.eu.maple.dev`. + +The content variables are explained [below](#record-prompts-responses-and-tool-calls). + +### Register a tracer provider when you run ADK with a Runner + +`adk web` and `adk api_server` build a tracer provider from the `OTEL_EXPORTER_OTLP_*` variables on startup. A `Runner` in your own FastAPI app, worker or script does not: ADK's spans go to OpenTelemetry's no-op default and nothing is exported, with no warning. Register one yourself: + +```py +# telemetry.py +import json + +from google.adk.plugins.base_plugin import BasePlugin +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class SkipDuplicateToolSpans(BatchSpanProcessor): + """Drops two ADK tool spans that would count a call twice: the + `execute_tool (merged)` summary of parallel calls, and the span of a call + paused for confirmation (it runs again, in its own span, once approved).""" + + def on_end(self, span): + if span.name != "execute_tool (merged)" and not span.attributes.get("adk.awaiting_confirmation"): + super().on_end(span) + + +class ToolCallAttributes(BasePlugin): + """Records each tool call's arguments and result on its `execute_tool` span, + and marks the span of a call that is waiting for confirmation.""" + + def __init__(self): + super().__init__(name="tool_call_attributes") + + async def before_tool_callback(self, *, tool, tool_args, tool_context): + trace.get_current_span().set_attribute("gen_ai.tool.call.arguments", json.dumps(tool_args, default=str)) + + async def after_tool_callback(self, *, tool, tool_args, tool_context, result): + span = trace.get_current_span() + if tool_context.actions.requested_tool_confirmations: + span.set_attribute("adk.awaiting_confirmation", True) + span.set_attribute("gen_ai.tool.call.result", json.dumps(result, default=str)) + + +# Reads OTEL_SERVICE_NAME, OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider = TracerProvider(resource=Resource.create()) +provider.add_span_processor(SkipDuplicateToolSpans(OTLPSpanExporter())) +trace.set_tracer_provider(provider) +``` + +Import it as the first line of your entry point, before your agents and before the first `runner.run_async()`: + +```py +# main.py +import telemetry # first, so the provider exists before ADK runs + +from google.adk.agents import LlmAgent +from google.adk.models.lite_llm import LiteLlm +from google.adk.runners import Runner +from google.adk.sessions import InMemorySessionService +from google.genai import types + + +def get_weather(city: str) -> dict: + """Get the current weather for a city.""" + return {"city": city, "temperature_c": 21, "condition": "partly cloudy"} + + +agent = LlmAgent( + name="assistant", + model=LiteLlm(model="openrouter/openai/gpt-4o-mini"), + instruction="You are a concise assistant.", + tools=[get_weather], +) + +runner = Runner( + app_name="support", + agent=agent, + session_service=InMemorySessionService(), + plugins=[telemetry.ToolCallAttributes()], + auto_create_session=True, +) +``` + +If your app already sets up OpenTelemetry (Logfire, Sentry, a platform agent), don't create a second provider. Add the `SkipDuplicateToolSpans(OTLPSpanExporter())` processor to the existing one, since only the first provider set globally wins. + +Under `adk web` or `adk api_server`, skip `telemetry.py`'s provider and rely on the environment variables. Register the plugin on your `App`, which is where the CLI looks for plugins: `App(name="support", root_agent=agent, plugins=[ToolCallAttributes()])`. The `SkipDuplicateToolSpans` processor can't be added there, since ADK owns that provider, so parallel calls and approval pauses add extra tool spans. + +## Group turns into one session with the ADK session id + +ADK stamps `session.id` as `gen_ai.conversation.id` on every `invoke_agent` and `generate_content` span, and Maple groups traces by it. Each `runner.run_async()` call is one trace and one turn, so a conversation is one Maple session as long as every turn uses the same ADK session: + +```py +async def chat(conversation_id: str, user_id: str, text: str) -> str: + reply = "" + async for event in runner.run_async( + user_id=user_id, + session_id=conversation_id, # the same id on every turn of this conversation + new_message=types.Content(role="user", parts=[types.Part(text=text)]), + ): + if event.is_final_response() and event.content and event.content.parts: + reply = "".join(part.text or "" for part in event.content.parts) + return reply +``` + +Use your own thread or chat id as the ADK session id. With `auto_create_session=True`, the runner creates the session on the first turn under that id. A persistent session service (`DatabaseSessionService`, `VertexAiSessionService`) keeps it across restarts. + +The common mistake is calling `create_session()` on every request without a `session_id`. ADK then mints a fresh UUID per turn, the model forgets the conversation, and Maple shows one single-turn session per message. Use `run_async()` rather than the synchronous `runner.run()` in servers. + +## Record prompts, responses and tool calls + +ADK has two content paths, and only one of them reaches Maple: + +| Setting | What it does | Default | +| --- | --- | --- | +| `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental` | Switches `generate_content` spans to the current GenAI conventions | off | +| `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` | Puts `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions` and `gen_ai.tool.definitions` on the `generate_content` span | no content | +| `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS` | ADK's own `gcp.vertex.agent.*` JSON attributes on `call_llm` and `execute_tool` spans | on | + +With both OpenTelemetry settings, every `generate_content` span carries the full request history and the reply as `[{role, parts}]` JSON: user text, assistant text, `tool_call` parts with arguments and `tool_call_response` parts with results. That's the shape Maple's transcript renders. + +- Without the stability opt-in, content goes to OpenTelemetry log records as ``, and Maple reads neither. +- A value of `true` for the capture variable means `EVENT_ONLY` for backward compatibility: log records, not spans. Use `SPAN_ONLY`. +- `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false` only removes ADK's legacy copies, which Maple ignores. Without it, each model call sends its full history twice. + +Tool spans don't get arguments or results from ADK in any mode yet. The `ToolCallAttributes` plugin from the setup adds them as `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, so the tool pages in Maple show what each call received and returned. ADK runs tool callbacks inside the tool's span, which is why `trace.get_current_span()` is the right span there. + +Both settings are read per run, so you can also set them per request with `RunConfig(telemetry=TelemetryConfig(genai_semconv_stability_opt_in="experimental", capture_message_content=ContentCapturingMode.SPAN_ONLY))` from `google.adk.telemetry.context`. + +### Privacy + +Everything a user types and every tool result is stored in Maple once content capture is on. To keep the structure, tokens and tool names but no content, leave `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` unset and delete the two `gen_ai.tool.call.*` lines from the `ToolCallAttributes` plugin. Keep the plugin itself, since it also marks approval pauses. To redact instead, scrub values in a tool callback or in an OpenTelemetry Collector with the `redaction` or `transform` processor before the data leaves your network. + +## Tool errors, sub-agents and agents as tools + +Tool spans are `execute_tool {tool name}` with `gen_ai.tool.name`, `gen_ai.tool.description` and `gen_ai.tool.call.id` matching the model's tool call. ADK 2.7 and later mark a failed tool call with span status ERROR and `error.type` in two cases: + +- **The tool raises.** The exception is recorded on the span, and the whole run fails unless something handles it. +- **The tool returns a dict with a non-empty `"error"` key.** ADK sets `error.type=TOOL_ERROR`. This is the way to report an expected failure to the model and still see it in Maple. + +ADK's tutorials often return `{"status": "error", "error_message": "..."}`. ADK doesn't recognize that shape, so those calls show as successful. Use `{"error": "..."}`. + +If a tool raising should not end the run, let the model see the error instead with an `on_tool_error_callback`. The span keeps its ERROR status because the returned dict has an `error` key: + +```py +class ToolErrorsAsResults(BasePlugin): + def __init__(self): + super().__init__(name="tool_errors_as_results") + + async def on_tool_error_callback(self, *, tool, tool_args, tool_context, error): + return {"error": str(error)} +``` + +A tool that asks for confirmation (`FunctionTool(func, require_confirmation=True)`) gets two `execute_tool` spans with the same call id: one when ADK pauses the call, and one when the approved call runs in the next `run_async()`. The pause span isn't marked failed, but Maple would count it as a second call. The plugin marks it and `SkipDuplicateToolSpans` drops it, so each approved call counts once, in the trace where it ran. Send the approval with the same `session_id` and it stays in the same session. If the user rejects the call, the span is marked failed with ADK's "This tool call is rejected" result. + +### Sub-agents + +Every agent run is an `invoke_agent {agent name}` span with `gen_ai.agent.name`, and each model call's `generate_content` span repeats the agent name, so Maple opens one lane per agent. `SequentialAgent`, `ParallelAgent` and `LoopAgent` nest their children's `invoke_agent` spans, and `ParallelAgent` children run concurrently as sibling spans. + +To call an agent like a tool, add it to `sub_agents` with `mode="single_turn"`. ADK runs it inside the parent's session. Each delegation produces an `execute_tool {agent name}` span, with the sub-agent's reply as its result, and a sibling `invoke_agent {agent name}` span. Both sit directly under the parent agent, so Maple opens a lane for the sub-agent and counts the `execute_tool` span as one tool call. When the model delegates to several agents in one response, they run in parallel. + +Avoid wrapping agents in `AgentTool`. It runs the sub-agent in a new in-memory session with a new id, so its spans carry a second `gen_ai.conversation.id` inside your trace. Maple keeps one id per trace and picks the larger of the two, which can move that turn into a session of its own. ADK's API docs also discourage `AgentTool` in favor of `mode="single_turn"`. + +## Tokens and cost + +`generate_content` spans carry `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.reasoning.output_tokens` when the model reports them. ADK counts cached tokens inside the input and thinking tokens inside the output, which is how Maple adds them up. + +- Streamed turns (`RunConfig(streaming_mode=StreamingMode.SSE)`) report usage too. `LiteLlm` requests it with `stream_options.include_usage`. +- The `call_llm` span above each `generate_content` repeats the same usage. Maple nets a parent's usage against its children, so each call is counted once. +- ADK doesn't record cost, and Maple doesn't price tokens, so ADK sessions show as **unpriced**. LiteLLM computes a cost, but it never reaches ADK's spans. + +## Flush spans before a short-lived process exits + +`BatchSpanProcessor` sends spans every 5 seconds. A script, CLI, notebook cell or job that exits sooner loses the last turn. Flush and shut down the provider when the work ends: + +```py +# script.py +import telemetry # first + +import asyncio + +from main import chat + + +async def main(): + try: + await chat("support-4821", "user-17", "What's the weather in Berlin?") + finally: + telemetry.provider.force_flush() + telemetry.provider.shutdown() + + +asyncio.run(main()) +``` + +In a long-running server, call `provider.shutdown()` from your shutdown hook (FastAPI `lifespan`, for example). On Cloud Run or another platform that freezes the CPU between requests, call `provider.force_flush()` before the response returns, or background export stalls until the next request. + +## Check that it works + +Run one conversation of two or three turns, one of which calls a tool, then open **Agent Sessions** in Maple. Spans leave the process within 5 seconds and usually show up within a minute. You should see: + +- **One session per ADK session id**, with the framework shown as **Google ADK** and one turn per `run_async()` call. +- **A transcript** on the session page: the user messages, the assistant replies, and each tool call with its arguments and result. +- **LLM calls and tokens** for each `generate_content {model}` span, with the model from `gen_ai.request.model` (for LiteLLM, the full id such as `openrouter/openai/gpt-4o-mini`). +- **Tool calls** named after your functions, with failed ones counted as errors. +- **Cost** shown as unpriced. + +In the trace view, a turn looks like this: + +```text +invocation +└─ invoke_agent assistant + ├─ call_llm + │ └─ generate_content openrouter/openai/gpt-4o-mini + ├─ execute_tool get_weather + └─ call_llm + └─ generate_content openrouter/openai/gpt-4o-mini +``` + +## Troubleshooting + +- **No spans at all.** You run ADK through a `Runner` and never registered a tracer provider. Only `adk web` and `adk api_server` build one from the environment. Import `telemetry.py` first. +- **Sessions show tokens but an empty transcript.** `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental` or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` is missing, or the capture variable is `true`, which means log records only. Both must be in the process environment before the run starts. +- **Every message is its own session.** Each request creates a new ADK session. Pass your conversation id as `session_id` on every turn. +- **One turn lands in a different session.** An `AgentTool` ran a sub-agent under its own session id. Use `sub_agents` with `mode="single_turn"`. +- **A tool named `(merged tools)`.** ADK's summary span for parallel tool calls. Add the `SkipDuplicateToolSpans` processor. +- **A confirmation-gated tool counts twice.** ADK records the paused call and the approved call as separate spans. Register `ToolCallAttributes` and `SkipDuplicateToolSpans` together: the plugin marks the pause and the processor drops it. +- **A failing tool shows as successful.** It returned `{"status": "error", ...}` or a string. Return a dict with an `"error"` key, or raise. +- **The streamed reply appears twice in the transcript, once in pieces.** With `StreamingMode.SSE`, ADK records every streamed chunk in `gen_ai.output.messages` and then the complete reply. The tokens are right; the transcript repeats the text. +- **Every model call appears twice, with doubled tokens.** Another instrumentor wraps the same calls: `litellm.callbacks = ["otel"]`, `openinference-instrumentation-google-adk`, or an OpenAI or LiteLLM instrumentor. ADK's own spans are enough; remove the others. +- **`gen_ai.system` says `gemini` for an OpenAI model.** ADK before 2.7 hardcoded it. Upgrade. With the settings in this guide, `generate_content` spans carry no provider attribute, which doesn't affect grouping, tokens or the transcript. +- **The model name has a provider prefix.** ADK records the requested model (`openrouter/openai/gpt-4o-mini`), never the model that served the response. It also records no `gen_ai.response.id`, so Maple can't merge a call that a second instrumentor reports again. That's one more reason to keep ADK's spans as the only ones. +- **401 from the exporter.** The header must read `Authorization=Bearer%20`, with the key from **Settings → Ingestion** for the right region. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): how Maple builds sessions, turns and checks. +- [Agent tracing guides](/docs/agent-tracing): every framework. +- [ADK agent activity traces](https://google.github.io/adk-docs/observability/traces/): ADK's span reference and export setup. +- [LiteLLM](/docs/agent-tracing/litellm): tracing LiteLLM on its own, outside ADK. +- [OpenTelemetry for any agent](/docs/agent-tracing/opentelemetry): the GenAI attributes Maple reads. diff --git a/apps/landing/src/content/docs/agent-tracing/haystack.md b/apps/landing/src/content/docs/agent-tracing/haystack.md new file mode 100644 index 0000000000..dfe796add3 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/haystack.md @@ -0,0 +1,380 @@ +--- +title: "Trace Haystack agents with OpenTelemetry" +description: "Send Haystack 3 Agent and pipeline runs to Maple as one Agent Session per conversation, with the transcript, model, tokens, cost and failed tool calls." +group: "AI Agents" +order: 28 +navLabel: "Haystack" +icon: "haystack" +--- + +Haystack 3 traces its own pipelines, components, Agent steps and tool calls through the `opentelemetry-haystack` tracer. Those spans use a private `haystack.*` vocabulary: the model id and token counts exist only inside a JSON blob of the model's reply, a failed tool call ends with status `Unset`, and there is no conversation id anywhere. Sent to Maple as-is, every Haystack run shows up with the right shape but with no model calls, no tokens, no transcript, no tool failures, and one session per request. + +This guide keeps Haystack's own spans and adds a single-file tracer of about 120 lines, `maple_haystack.py`, that writes the OpenTelemetry GenAI attributes Maple reads onto them while they are still open. It covers Python 3.10+ with `haystack-ai` 3.2 and `opentelemetry-haystack` 1.0, for the `Agent` component, `AgentTool`/`PipelineTool` sub-agents and chat generators in plain pipelines. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-haystack](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-haystack) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Haystack in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-haystack -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Why not the plain Haystack tracer or OpenInference + +There are three ready-made ways to get OpenTelemetry spans out of Haystack. None of them gives Maple a complete session on its own: + +| Option | What Maple gets | What is missing | +| --- | --- | --- | +| `OpenTelemetryTracer` from `opentelemetry-haystack` | Pipeline, component, Agent, step and tool spans, detected as **Haystack** | Model, tokens, transcript and tool failures (all inside `haystack.*` blobs Maple does not read), session id | +| `openinference-instrumentation-haystack` | Model and tokens on generator spans | Tool spans (tools run inside the Agent, which it records as one opaque chain), session id (`using_session` writes `session.id`, which Maple ignores for this dialect), framework label ("Unidentified") | +| OpenLLMetry `opentelemetry-instrumentation-haystack` | `Pipeline.run`, plus `OpenAIGenerator` and `OpenAIChatGenerator` calls | Agent steps, tool spans, tokens, every other generator (OpenRouter, Anthropic, ...), content in indexed `gen_ai.prompt.N.*` keys Maple does not read, session id | + +The tracer in this guide is a subclass of the first option. It keeps every span and tag Haystack emits, so nothing you already see in Maple's trace view changes, and adds `gen_ai.*` attributes next to them. Don't run it together with the OpenInference or OpenLLMetry instrumentor: each would record every model call a second time. + +## Install and export to Maple + +```bash +pip install "haystack-ai>=3.2" "opentelemetry-haystack>=1.0" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Save this file next to your app as `maple_haystack.py`: + +```py +"""Haystack tracer that adds the OpenTelemetry GenAI attributes Maple reads.""" + +import json +import logging +from collections.abc import Iterator +from contextlib import contextmanager +from contextvars import ContextVar +from typing import Any + +from haystack.dataclasses import ChatMessage +from haystack_integrations.tracing.opentelemetry import OpenTelemetrySpan, OpenTelemetryTracer +from opentelemetry import trace +from opentelemetry.trace import StatusCode + +logger = logging.getLogger(__name__) +_conversation_id: ContextVar[str | None] = ContextVar("maple_conversation_id", default=None) + + +@contextmanager +def conversation(conversation_id: str) -> Iterator[None]: + """Every Haystack run inside this block joins the same Maple session.""" + token = _conversation_id.set(conversation_id) + try: + yield + finally: + _conversation_id.reset(token) + + +def _messages(messages: list[ChatMessage]) -> str: + out = [] + for m in messages: + parts: list[dict[str, Any]] = [{"type": "text", "content": t} for t in m.texts] + parts += [{"type": "tool_call", "id": c.id, "name": c.tool_name, "arguments": c.arguments} for c in m.tool_calls] + parts += [{"type": "tool_call_response", "id": r.origin.id, "response": r.result} for r in m.tool_call_results] + out.append({"role": m.role.value, "parts": parts}) + return json.dumps(out, default=str) + + +class MapleSpan(OpenTelemetrySpan): + def __init__(self, span: trace.Span, operation: str | None, content: bool) -> None: + super().__init__(span) + self._operation = operation + self._content = content + + def set_content_tag(self, key: str, value: Any) -> None: + try: + if self._operation == "chat": + self._chat(key, value) + elif self._operation == "execute_tool": + self._tool(key, value) + except Exception: # a tracing bug must never fail the agent run + logger.exception("maple_haystack: could not map %s", key) + if self._content: + self.set_tag(key, value) + + def _chat(self, key: str, value: Any) -> None: + if key.endswith(".input") and self._content: + messages = value["messages"] + system = [{"type": "text", "content": m.text} for m in messages if m.is_from("system")] + if system: + self._span.set_attribute("gen_ai.system_instructions", json.dumps(system)) + self._span.set_attribute("gen_ai.input.messages", _messages([m for m in messages if not m.is_from("system")])) + elif key.endswith(".output"): + replies = value["replies"] + meta = replies[0].meta + usage = meta.get("usage") or {} + prompt_details = usage.get("prompt_tokens_details") or {} + attributes = { + "gen_ai.response.model": meta.get("model"), + "gen_ai.response.finish_reasons": meta.get("finish_reason"), + "gen_ai.usage.input_tokens": usage.get("prompt_tokens"), + "gen_ai.usage.output_tokens": usage.get("completion_tokens"), + "gen_ai.usage.cache_read.input_tokens": prompt_details.get("cached_tokens"), + "gen_ai.usage.cache_write.input_tokens": prompt_details.get("cache_write_tokens"), + "gen_ai.usage.reasoning.output_tokens": (usage.get("completion_tokens_details") or {}).get("reasoning_tokens"), + "gen_ai.usage.cost": usage.get("cost"), # OpenRouter prices every call; other providers leave this out + } + if self._content: + attributes["gen_ai.output.messages"] = _messages(replies) + self._span.set_attributes({k: v for k, v in attributes.items() if v is not None}) + + def _tool(self, key: str, value: Any) -> None: + if key.endswith(".output") and isinstance(value, dict) and "error" in value: + # Haystack records a failed tool as {"error": ...} and leaves the span status unset + self._span.set_status(StatusCode.ERROR, str(value["error"])) + self._span.set_attribute("error.type", "ToolInvocationError") + if self._content: + attribute = "gen_ai.tool.call.arguments" if key.endswith(".input") else "gen_ai.tool.call.result" + self._span.set_attribute(attribute, json.dumps(value, default=str)) + + +class MapleHaystackTracer(OpenTelemetryTracer): + def __init__(self, tracer: trace.Tracer, *, content: bool = True) -> None: + super().__init__(tracer) + self._content = content + + @contextmanager + def trace(self, operation_name: str, tags: dict[str, Any] | None = None, parent_span: Any = None) -> Iterator[MapleSpan]: + tags = dict(tags or {}) + attributes: dict[str, str] = {} + operation = None + if operation_name == "haystack.agent.run": + # The Agent span has no name: use the pipeline component or the AgentTool that runs it + parent = getattr(trace.get_current_span(), "attributes", None) or {} + attributes["gen_ai.operation.name"] = "invoke_agent" + attributes["gen_ai.agent.name"] = parent.get("haystack.component.name") or parent.get("gen_ai.tool.name") or "agent" + elif operation_name == "haystack.agent.step.llm" or str(tags.get("haystack.component.type", "")).endswith("ChatGenerator"): + operation = attributes["gen_ai.operation.name"] = "chat" + elif operation_name == "haystack.agent.step.tool": + operation = attributes["gen_ai.operation.name"] = "execute_tool" + attributes["gen_ai.tool.name"] = tags["haystack.tool.name"] + if conversation_id := _conversation_id.get(): + attributes["gen_ai.conversation.id"] = conversation_id + if not self._content: + tags.pop("haystack.pipeline.input_data", None) # a plain tag, not gated by Haystack's content switch + + with self._tracer.start_as_current_span(operation_name, attributes=attributes) as raw_span: + span = MapleSpan(raw_span, operation, self._content) + span.set_tags(tags) + yield span +``` + +It maps Haystack's spans as follows: + +| Haystack span | Maple reads | +| --- | --- | +| `haystack.agent.run` | `invoke_agent`, agent name from the pipeline component or `AgentTool` that runs it | +| `haystack.agent.step.llm`, and any `*ChatGenerator` component | `chat`: model, finish reason, input/output/cache/reasoning tokens, cost, messages | +| `haystack.agent.step.tool` | `execute_tool`: tool name, arguments, result, and `Error` status when the tool failed | +| every span | `gen_ai.conversation.id` inside a `conversation()` block | + +Then configure OpenTelemetry once at startup and hand Haystack the tracer: + +```py +# telemetry.py +from haystack import tracing +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +from maple_haystack import MapleHaystackTracer + +provider = TracerProvider( + resource=Resource.create({"service.name": "support-agent", "deployment.environment.name": "production"}) +) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +tracing.enable_tracing(MapleHaystackTracer(trace.get_tracer("haystack"))) +``` + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` to the endpoint itself. + +Import `telemetry` before the first `pipeline.run()` or `agent.run()`. Unlike Haystack's content switch, the tracer has no import-order trap: Haystack looks up the active tracer on every span, so any run after `enable_tracing()` is traced. + +Two details matter here: + +- **Keep the tracer name `"haystack"`.** It becomes the instrumentation scope, which is how Maple labels the sessions as Haystack. Maple also recognizes Haystack's span names, but the scope covers spans added in newer Haystack releases too. +- **If the app already has a `TracerProvider`** (from FastAPI instrumentation or another library), skip the provider lines and pass `trace.get_tracer("haystack")` from the existing one. A second provider sends every span twice or not at all. + +## Group turns into one session + +Haystack's `Agent` is stateless: your app keeps the message history and passes it into every run, so nothing on the wire says that two runs belong to the same chat. Each `pipeline.run()` is its own trace, and without a conversation id Maple shows each one as a separate one-turn session. + +Wrap every run in `conversation()` with your own chat id: + +```py +from haystack import Pipeline +from haystack.components.agents import Agent +from haystack.dataclasses import ChatMessage +from haystack_integrations.components.generators.openrouter import OpenRouterChatGenerator + +from maple_haystack import conversation + +chat_generator = OpenRouterChatGenerator(model="openai/gpt-4o-mini") +agent = Agent(chat_generator=chat_generator, tools=[get_weather, calculate], system_prompt=SYSTEM_PROMPT) +pipeline = Pipeline() +pipeline.add_component("assistant", agent) + + +def handle_message(chat_id: str, history: list[ChatMessage], text: str) -> list[ChatMessage]: + with conversation(chat_id): + result = pipeline.run({"assistant": {"messages": [*history, ChatMessage.from_user(text)]}}) + # The Agent adds its system prompt on every run, so keep it out of the stored history + return [m for m in result["assistant"]["messages"] if not m.is_from("system")] +``` + +Use the id your app already stores for the chat, such as a database row id or a thread id from your frontend. Don't mint a new UUID per request, and don't use one id for the whole process: both break the grouping, in opposite directions. + +`conversation()` is a `ContextVar`, so it is scoped to the current request in async and threaded servers alike, and it follows the Agent into the worker threads that run tools in parallel. It writes `gen_ai.conversation.id` on every span of the run. Maple needs it on at least one span per trace to join that trace to the session. + +## Record prompts, responses and tool calls + +With the default `content=True`, the tracer writes: + +- `gen_ai.input.messages` and `gen_ai.output.messages` on every model call, as `{role, parts}` arrays with text, tool calls and tool results; +- `gen_ai.system_instructions` with the Agent's system prompt; +- `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result` on every tool span; +- Haystack's own `haystack.*.input`/`.output` tags, as before. + +Maple builds the transcript and the turn labels from the `gen_ai.*` messages. The `haystack.*` blobs only show up in the raw span attributes. + +`HAYSTACK_CONTENT_TRACING_ENABLED` has no effect with this tracer: `MapleHaystackTracer` decides on its own. Without the tracer, that variable is read once, at the first `import haystack`, and setting it any later silently records nothing. + +To keep prompts and responses out of Maple, turn content off: + +```py +tracing.enable_tracing(MapleHaystackTracer(trace.get_tracer("haystack"), content=False)) +``` + +Model, tokens, cost, finish reasons, tool names and tool failures are still recorded, because they come from the reply metadata rather than the text. The error message of a failed tool stays on the span status, and Haystack's message quotes the call's arguments (``Failed to invoke Tool `fetch_transport_data` with parameters {'city': 'Rome'}``). If arguments can carry personal data, redact them in `_tool()` before `set_status`. + +Haystack also puts the whole pipeline input on the root span as `haystack.pipeline.input_data`, a plain tag that its own content switch never gated. With `content=False` the tracer drops it, so the user's message doesn't leak through the back door. To redact rather than drop, filter the values in `_messages()` before they are written. + +## Tools, errors and sub-agents + +Each tool call is a `haystack.agent.step.tool` span with `gen_ai.tool.name` set to the tool's name. Tools of one step run in parallel threads, and their spans sit side by side under the step. + +When a tool raises, Haystack wraps the exception in `ToolInvocationError`, feeds the error text back to the model, and writes `{"error": "..."}` as the tool output. The span itself ends with status `Unset`, so without the tracer the failure is invisible. The tracer sets status `Error` with the message and `error.type=ToolInvocationError`, which is what Maple counts as a failed tool call. This happens with both values of `raise_on_tool_invocation_failure`. + +Maple opens a lane for every agent with a distinct `gen_ai.agent.name`. The tracer takes that name from what runs the Agent: + +- an Agent added to a pipeline gets its component name, `assistant` in the example above; +- an Agent wrapped in `AgentTool(agent=..., name="weather_worker", ...)` gets the tool name, and Maple shows the tool call as a delegation to that sub-agent; +- an Agent inside a `PipelineTool` gets its component name in the inner pipeline; +- an Agent run directly with `agent.run()` is called `agent`. + +A multi-agent setup with the orchestrator in a pipeline and workers as `AgentTool`s looks like this: + +```py +from haystack.tools import AgentTool + +weather_worker = AgentTool( + agent=Agent(chat_generator=chat_generator, tools=[get_weather], system_prompt="Report the weather."), + name="weather_worker", + description="Look up the current weather for a city.", +) +# budget_worker and transport_worker are built the same way +orchestrator = Agent(chat_generator=chat_generator, tools=[weather_worker, budget_worker, transport_worker]) +pipeline = Pipeline() +pipeline.add_component("orchestrator", orchestrator) + +with conversation(briefing_id): + pipeline.run({"orchestrator": {"messages": [ChatMessage.from_user("Brief me on Amsterdam.")]}}) +``` + +Approval gates (`ConfirmationHook` at `before_tool`) stay inside the same run and the same session. A call the user rejects never reaches the tool, so it has no tool span; the model sees the rejection as a tool result, which is visible in the transcript. + +## Tokens and cost + +Tokens come from the `usage` object Haystack's generators put in each reply's `meta`. The tracer reads the OpenAI field names (`prompt_tokens`, `completion_tokens`, `prompt_tokens_details.cached_tokens` and `.cache_write_tokens`, `completion_tokens_details.reasoning_tokens`), which is what `OpenAIChatGenerator`, `OpenRouterChatGenerator` and other OpenAI-compatible generators report. + +**Streaming needs one extra flag with OpenAI.** A streamed OpenAI response only carries usage when you ask for it, and Haystack doesn't: + +```py +OpenAIChatGenerator(model="gpt-4o-mini", generation_kwargs={"stream_options": {"include_usage": True}}) +``` + +OpenRouter always sends usage in the last streamed chunk, so `OpenRouterChatGenerator` needs nothing extra. + +Maple never prices tokens itself. It shows cost only when a span carries one, and the only Haystack generator that reports a price is OpenRouter's (`usage.cost`, in USD), which the tracer copies to `gen_ai.usage.cost`. With any other provider, sessions show tokens and read as **unpriced**. + +The Agent itself reports no usage, so there is nothing to double count: each model call is counted once, on its `haystack.agent.step.llm` span. + +## Short-lived processes + +`BatchSpanProcessor` exports in the background every 5 seconds. A script, CLI, notebook cell, cron job or serverless handler that exits sooner loses the last batch. Flush before the process ends: + +```py +try: + handle_message(chat_id, history, text) +finally: + provider.force_flush() + provider.shutdown() +``` + +Long-running servers only need `provider.shutdown()` in their shutdown hook. + +## Check that it works + +Run one conversation of at least two turns, one of them with a tool call. Within a minute, open **Agent Sessions** in Maple and filter by your service name. You should see: + +- **one session per conversation id**, labelled Haystack, with one turn per `pipeline.run()`; +- a transcript with the user's messages, the assistant's replies and the tool calls, each turn labelled with its user message; +- an LLM call per `haystack.agent.step.llm` span, with the model (for example `openai/gpt-4o-mini`) and input and output tokens, including the streamed turn; +- a tool call per `haystack.agent.step.tool` span, named after the tool, with a failed tool counted as an error; +- the agent name from your pipeline component or `AgentTool`, and a lane per sub-agent; +- cost on OpenRouter, or "unpriced" with other providers. + +Each turn's trace looks like this: + +```text +haystack.pipeline.run + haystack.component.run assistant + haystack.agent.run invoke_agent, agent "assistant" + haystack.agent.step + haystack.agent.step.llm chat, openai/gpt-4o-mini + haystack.agent.step.tool execute_tool, get_weather + haystack.agent.step + haystack.agent.step.llm chat +``` + +An Agent with hooks, such as a `ConfirmationHook`, also gets a `haystack.agent.hook` span before its tool calls. Like `haystack.agent.step`, it carries no model or tool attributes, so Maple doesn't count it as a call. + +## Troubleshooting + +- **Every request is its own session.** The run happened outside a `conversation()` block, or each request passes a fresh id. Wrap the `pipeline.run()` call itself, and pass the chat's stored id. +- **Sessions show the right spans but no model calls, tokens or transcript.** Haystack is still using the plain `OpenTelemetryTracer`. Check that `enable_tracing()` gets a `MapleHaystackTracer`, and that no later call replaces it. +- **No Haystack spans at all.** Haystack 3 no longer turns tracing on when `opentelemetry-sdk` is installed. Call `tracing.enable_tracing(...)` before the first run. +- **Every model call appears twice.** The OpenInference or OpenLLMetry Haystack instrumentor is also active. Remove it; this tracer already records every model call. +- **The streamed turn has no tokens.** `OpenAIChatGenerator` without `stream_options.include_usage`. Add it to `generation_kwargs`. +- **Tokens are zero with a non-OpenAI generator.** Its `meta["usage"]` doesn't use the OpenAI field names. Print `result["replies"][0].meta["usage"]` once and add its keys to `_chat()`. +- **The agent is called `agent`.** It ran through `agent.run()`, not as a pipeline component or `AgentTool`. Add it to a `Pipeline` under the name you want to see. +- **A failed tool isn't counted.** The tool caught its own exception and returned a normal value. Let it raise, or return `{"error": ...}`, which Haystack uses for failures too. +- **Sessions are split or missing turns in a script.** The process exited before the batch was exported. Call `provider.force_flush()` and `provider.shutdown()` in `finally`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Agent tracing guides](/docs/agent-tracing) +- [Trace agents with plain OpenTelemetry](/docs/agent-tracing/opentelemetry) +- [Haystack tracing docs](https://docs.haystack.deepset.ai/docs/tracing) +- [`opentelemetry-haystack` on PyPI](https://pypi.org/project/opentelemetry-haystack/) diff --git a/apps/landing/src/content/docs/agent-tracing/langchain.md b/apps/landing/src/content/docs/agent-tracing/langchain.md new file mode 100644 index 0000000000..02b5626f1e --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/langchain.md @@ -0,0 +1,311 @@ +--- +title: "Trace LangChain and LangGraph agents with OpenTelemetry" +description: "Send LangChain and LangGraph runs to Maple as Agent Sessions, one per thread, with a readable transcript, model and tool calls, tokens, failed tools and sub-agent lanes." +group: "AI Agents" +order: 13 +navLabel: "LangChain & LangGraph" +icon: "langchain" +--- + +LangChain and LangGraph report every run through their callback system: each graph node, chat model call and tool call starts and ends a run. Two libraries turn those runs into OpenTelemetry spans. OpenInference's `openinference-instrumentation-langchain` adds its own callback handler, and LangSmith's SDK can export its runs over OTLP instead of to smith.langchain.com. Both work with Maple, and this guide uses OpenInference with its GenAI dual-write, because that's the one that gives Maple a transcript it can render. + +The default that goes wrong is the conversation. Every `invoke()` of an agent or graph is a new trace, so a chat of ten messages arrives as ten traces, and nothing links them until you pass a `thread_id`. The same `thread_id` a LangGraph checkpointer already needs is the one Maple groups sessions by. This guide covers Python: LangChain 1.4 (`create_agent`) and LangGraph 1.2 (`StateGraph`) with `openinference-instrumentation-langchain` 0.1.76, on Python 3.10 or later. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-langchain](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-langchain) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for LangChain & LangGraph in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-langchain -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install the instrumentor and export to Maple + +```bash +pip install "langchain>=1.4" "langgraph>=1.2" "langchain-openai>=1.6" \ + "openinference-instrumentation-langchain>=0.1.76" "openinference-instrumentation>=0.1.66" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Pin `openinference-instrumentation` explicitly. The LangChain instrumentor accepts versions back to 0.1.61, and the GenAI dual-write this guide depends on isn't in the oldest of them. + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. `OTLPSpanExporter()` with no arguments appends `/v1/traces` to the base URL. If you pass `endpoint=` in code instead, it's used as is and has to end in `/v1/traces`. + +Then add a `tracing.py` and import it at the top of your entry point: + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.langchain import LangChainInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +# The name= you gave create_agent(), and graph nodes that act as agents +AGENT_NAMES = {"assistant"} + + +class AgentSpans(SpanProcessor): + """Marks your agents' spans as agent invocations, so Maple can name them and give each a lane.""" + + def on_start(self, span, parent_context=None): + if span.instrumentation_scope.name != "openinference.instrumentation.langchain": + return + if span.name in AGENT_NAMES: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", span.name) + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(AgentSpans()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +LangChainInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +What each part does: + +- **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.conversation.id` next to its OpenInference attributes when each span ends. Without it, Maple reads the spans as generic OpenInference: the transcript is a raw JSON blob and the `thread_id` is ignored, so every turn is its own session. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, as long as it's set before `TraceConfig` is built. +- **`AgentSpans`** fills the one thing the instrumentor leaves out. It names agent spans only by the word "agent": a `create_agent(name="support_agent")` span becomes an agent, `name="assistant"` stays a plain chain, and neither gets `gen_ai.agent.name`, which Maple needs to name agents and open a lane per sub-agent. The processor runs at span start, and the dual-write never overwrites a key that's already set. + +The instrumentor hooks LangChain's callback manager, so import order doesn't matter, as long as `instrument()` runs before the first `invoke()`. It traces every LangChain runnable in the process: agents, graphs, chains, chat models, tools and retrievers. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second one. Add `AgentSpans()` and the OTLP exporter to the existing provider and pass that provider to `instrument()`. + +LangSmith keeps working next to this. With `LANGSMITH_TRACING=true` and a LangSmith key, runs still go to smith.langchain.com over LangSmith's own API, and nothing is sent twice to Maple. Don't also set `LANGSMITH_OTEL_ENABLED`, which would add a second copy of every span to your provider (see [the LangSmith exporter](#langsmiths-opentelemetry-exporter-instead) below). + +## Group a conversation into one session with thread_id + +Maple groups traces into a session by `gen_ai.conversation.id`. The instrumentor sets it on every span of a run from the run's metadata, taking the first of `session_id`, `conversation_id` and `thread_id`. LangGraph copies `configurable` values into that metadata, so the `thread_id` you already pass for a checkpointer is enough: + +```py +import tracing # first, before the first invoke() + +from langchain.agents import create_agent +from langchain_openai import ChatOpenAI +from langgraph.checkpoint.memory import InMemorySaver + +agent = create_agent( + ChatOpenAI(model="gpt-4o-mini", stream_usage=True), + tools=[get_weather, calculate], + name="assistant", + checkpointer=InMemorySaver(), +) + + +def handle_message(conversation_id: str, text: str) -> str: + result = agent.invoke( + {"messages": [{"role": "user", "content": text}]}, + {"configurable": {"thread_id": conversation_id}}, + ) + return result["messages"][-1].content +``` + +The checkpointer isn't what sets the session. A graph compiled without one still gets `gen_ai.conversation.id` from `configurable.thread_id`, so a stateless pipeline, like a multi-agent graph run once per request, can pass a `thread_id` only for tracing. A plain LangChain chain (a `prompt | model` pipeline, no graph) doesn't copy `configurable` into the instrumentor's metadata, so pass the id as metadata instead: `chain.invoke(inputs, {"metadata": {"thread_id": conversation_id}})`. + +Use your app's own conversation id, the one it stores the chat under. A new UUID per request gives you one session per message again, and a constant puts every user in one session. + +If you skip it, every `invoke()` shows up in **Agent Sessions** as its own one-turn session named after its trace id. A human-in-the-loop resume (`Command(resume=...)`) is a new `invoke()` and a new trace too, and the shared `thread_id` is the only thing that puts it back in the same session as the turn it interrupted. + +## Record prompts, responses and tool calls + +Content capture is on by default. Every chat model span carries the full message list sent to the model and the reply, as `gen_ai.input.messages` and `gen_ai.output.messages` in `{role, parts}` form, and Maple renders them as the session transcript, with the user's message as the turn's label. The system prompt is the first input message. Tool spans carry the tool's result in `gen_ai.tool.call.result`. + +With a checkpointer, every model span repeats the thread's whole history, so a long conversation gets large. Maple has no per-attribute limit, and ingest accepts requests up to 20 MiB. + +To keep prompts and outputs out of your traces: + +```py +config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) +``` + +`hide_inputs` and `hide_outputs` drop the messages and replace `input.value` and `output.value` with `__REDACTED__`. The GenAI attributes are built from the masked values, so they're empty too. The session still shows its turns, model and tool calls, tokens and failures, with an empty transcript. Each switch has an `OPENINFERENCE_HIDE_*` environment variable, and narrower ones exist: `hide_input_text` and `hide_output_text` keep the message structure but redact the text. For pattern-based redaction, use the `redaction` processor in an OpenTelemetry Collector. + +## Tools, errors and sub-agents + +Each tool call is a span named after the tool, with `gen_ai.operation.name` `execute_tool`, `gen_ai.tool.name`, `gen_ai.tool.description` and the result. The arguments aren't on the tool span, and neither is `gen_ai.tool.call.id`, because the instrumentor doesn't record them there. The arguments are still in the transcript, in the model's tool call just before. + +A tool that raises is marked failed without extra code: its span ends with status `ERROR` and the exception as the status message, for example `RuntimeError('transport data service unavailable (503)')`, and Maple counts it on the session and on the tool's page. + +What happens to the run next is up to LangChain. `create_agent` re-raises any exception from a tool by default, so one broken tool fails the whole `invoke()`. To hand the error to the model and keep going, add a `wrap_tool_call` middleware: + +```py +from langchain.agents.middleware import wrap_tool_call +from langchain_core.messages import ToolMessage + + +@wrap_tool_call +def tool_errors_to_model(request, handler): + try: + return handler(request) + except Exception as e: + return ToolMessage(content=f"Tool error: {e}", tool_call_id=request.tool_call["id"], status="error") + + +agent = create_agent(model, tools=tools, name="assistant", middleware=[tool_errors_to_model]) +``` + +The tool span is still marked failed, because the tool's own run ends with the exception before the middleware catches it. In a `StateGraph` with LangGraph's `ToolNode`, `ToolNode(tools, handle_tool_errors=True)` does the same. + +Human-in-the-loop interrupts aren't errors. `interrupt()` and `HumanInTheLoopMiddleware` pause the graph by raising `GraphInterrupt`, and the instrumentor ends that span with status `OK`. The pause ends the turn's trace, and the resume starts a new one, so an approved action shows as two turns in the same session: the request, then the resumed tool call and reply. + +Sub-agents need the `AgentSpans` processor from the setup. Add every agent's name to `AGENT_NAMES`, and Maple shows each one in its own lane with its model and tool calls. The common LangChain pattern, a worker agent called from a tool of the orchestrator, looks like this: + +```py +AGENT_NAMES = {"orchestrator", "weather_worker"} # in tracing.py + +weather_worker = create_agent(model, tools=[get_weather], name="weather_worker") + + +@tool +def ask_weather_worker(city: str) -> str: + """Ask the weather worker for the current weather in a city.""" + result = weather_worker.invoke({"messages": [{"role": "user", "content": f"Weather in {city}?"}]}) + return result["messages"][-1].content + + +orchestrator = create_agent(model, tools=[ask_weather_worker], name="orchestrator") +``` + +The worker's run nests under the tool span and inherits the orchestrator's `thread_id`, so it lands in the same trace and session without passing anything. Maple shows the `ask_weather_worker` call as a delegation to `weather_worker`, with the tool's argument and result as the lane's input and output. In a `StateGraph` where each worker is a node, list the node names instead. + +Don't put "agent" in a tool's name. The instrumentor names a span of any kind an agent when its name contains the word, so a tool called `ask_weather_agent` loses its tool kind and Maple doesn't count it as a tool call. + +Two gaps remain in what Maple shows for LangGraph tools. LangGraph runs tools inside a node, named `tools` in `create_agent` and wherever you add a `ToolNode` under that name, and Maple currently counts that node's span as an extra, unnamed tool call, because its name contains "tool". So a turn with one tool call shows two. The real tool calls are the ones with a name. The missing `gen_ai.tool.call.id` means Maple matches tool spans to the model's calls by name, which works unless one reply calls the same tool twice. + +## Tokens and cost + +Every chat model span carries input and output tokens from LangChain's `usage_metadata`, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` next to the OpenInference `llm.token_count.*` originals. The model is the one you configured, in `gen_ai.request.model`, and the provider comes from the LangChain integration: `ChatOpenAI` is `openai`, even for an Anthropic model behind OpenRouter or another OpenAI-compatible gateway. + +Streaming is where tokens go missing. `ChatOpenAI` only asks for usage on streamed responses (`stream_options.include_usage`) when it talks to api.openai.com. With a `base_url`, or `OPENAI_BASE_URL` set, it doesn't, and every streamed call arrives with no token counts. Set `stream_usage=True` on the model, as in the example above. + +Maple shows cost only when a span carries one, and neither LangChain nor the instrumentor records cost. Sessions show as **unpriced**, with token counts. + +A `ChatPromptTemplate` in a chain produces a span named `ChatPromptTemplate`, and Maple currently counts it as a model call because the name contains "chat". It has no model or tokens, so totals are right, but the call count is one too high per template. `create_agent` and LangGraph nodes that call a model directly don't use templates. + +Don't add `openinference-instrumentation-openai`, `-anthropic` or OpenLLMetry's LangChain instrumentor next to this one. Each wraps the same model request again, and every call gets a second model span with its own tokens. + +## Flush spans before the process exits + +The instrumentor ends each span when the run's callback fires, and `BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` registers an `atexit` handler that flushes on a normal interpreter exit, which covers most scripts and CLIs. It doesn't run when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: + +```py +from tracing import provider + +try: + handle_message("conv-42", "What's the weather in Berlin?") +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. + +Context survives LangGraph's parallel nodes, `ainvoke`, and agents called from inside tools on Python 3.11 and later. On Python 3.10, asyncio doesn't carry context into tasks, so pass the node's `config` to every nested `ainvoke()` or its runs start new traces. If you run LangChain code in your own thread pool, use `ContextThreadPoolExecutor` from `langchain_core.runnables.config` instead of the standard library's. + +## LangGraph Server deployments + +On LangGraph's Agent Server (`langgraph dev` or a self-hosted server), the server imports the module that defines your graph, the one `langgraph.json` points to. Import `tracing` at the top of that module and set the `OTEL_*` variables in the server's environment (the `env` file in `langgraph.json`, or the container). The server is long-running, so `BatchSpanProcessor` exports on its own schedule and no flush is needed. + +Every run on a server thread already carries that thread's id as `configurable.thread_id`, so each LangGraph thread becomes one Maple session with no extra code. + +## LangSmith's OpenTelemetry exporter instead + +LangSmith's SDK can write its runs as OTLP spans (`LANGSMITH_OTEL_ENABLED`). Maple recognizes those spans and labels them **LangChain**, and reads the session from `langsmith.metadata.thread_id`, the same `configurable.thread_id`. It's the only path that shows the framework name, and it records `gen_ai.tool.call.id`. It gives Maple less to work with everywhere else: + +- The prompt and completion are LangChain's serialized objects (`{"lc":1,"type":"constructor",...}`) in `gen_ai.prompt` and `gen_ai.completion`, so the transcript is one raw JSON blob per model call, with no turn labels. +- An interrupt marks the interrupted node's span `ERROR`, with `GraphInterrupt(...)` as an `exception` event, so human-in-the-loop pauses read as failures. +- Middleware wrappers such as `HumanInTheLoopMiddleware.wrap_tool_call` get their own spans, and Maple counts them as extra tool calls. A failing tool's error passes through the ones inside your error handler, so one failure counts more than once. +- Prompt templates get `gen_ai.operation.name` `chat`, and `gen_ai.system` is guessed from the model name: `anthropic/claude-haiku-4.5` through OpenRouter reads as `anthropic`. +- There are no agent names. `create_agent`'s name is only in `langsmith.metadata.lc_agent_name`, which Maple doesn't read, and the `AgentSpans` trick doesn't work because LangSmith overwrites `gen_ai.operation.name` after the span starts. + +If you want it anyway, set the variables and the provider before anything imports LangChain. LangSmith looks for a global provider when its client is created, and if there isn't one it builds its own, pointed at smith.langchain.com: + +```py +import os + +os.environ["LANGSMITH_TRACING"] = "true" +os.environ["LANGSMITH_OTEL_ENABLED"] = "true" +os.environ["LANGSMITH_OTEL_ONLY"] = "true" # OTLP only, no LangSmith account or key needed + +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider() +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +# Only now import langchain, langgraph and your agents +``` + +Install `langsmith[otel]>=0.14`. Flushing takes two steps, because LangSmith converts runs to spans on a background thread and the provider has nothing to export until it's done. `provider.force_flush()` on its own exports nothing: + +```py +from langchain_core.tracers.langchain import wait_for_all_tracers + +wait_for_all_tracers() +provider.force_flush() +``` + +Don't combine it with the OpenInference instrumentor; you'd get every run twice. + +## LangChain.js and LangGraph.js + +This guide doesn't cover JavaScript yet. LangSmith's OpenTelemetry mode in JS is experimental and its `initializeOTEL()` setup is deprecated, and the JS OpenInference instrumentor has no GenAI dual-write, so Maple ignores its `session.id` and shows one session per trace. For a TypeScript agent today, see [Trace your AI agent](/docs/agent-tracing) for the frameworks that are covered. + +## Check that it works + +Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. Spans usually show up within a minute. You should see: + +- **One session** for the conversation, with one turn per `invoke()`. A second conversation with a different id is a second session. +- **Framework: Unidentified.** Maple recognizes LangChain by LangSmith's exporter, and the OpenInference spans read as generic GenAI spans. Everything else on the session page works. +- **The transcript**: your messages as turn labels, the model's replies, and its tool calls. +- **Agents**: `assistant`, plus one lane per sub-agent in `AGENT_NAMES`. Each turn's trace starts at the agent span, with `model` and `tools` node spans below it. +- **Model calls** named `ChatOpenAI` (or `ChatAnthropic`, and so on), each with a model, input and output tokens, including streamed ones. +- **Tool calls** named after your tools, like `get_weather`, with results, and a failing tool marked failed with its message. +- **Cost**: unpriced. + +If a turn is missing, check that the process flushed. + +## Troubleshooting + +- **No spans at all.** `instrument()` never ran, or ran with a different provider than the one exporting. Import `tracing` first, pass `tracer_provider=provider`, and look for `OTLPSpanExporter` errors in the logs. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. +- **One session per message.** No `thread_id` reached the run, or it changes per request. Pass `{"configurable": {"thread_id": conversation_id}}` to every `invoke()`, `stream()` and resume, or `{"metadata": {"thread_id": ...}}` for plain chains. +- **Sessions work but the session page is a JSON blob, or tokens show only in the list.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)`. +- **Streamed replies have no tokens.** `ChatOpenAI` with a custom `base_url` doesn't request streamed usage. Set `stream_usage=True`. +- **One failing tool ends the whole run.** `create_agent` re-raises tool exceptions. Add the `wrap_tool_call` middleware, or `handle_tool_errors=True` on your `ToolNode`. +- **A tool shows up as an agent, not a tool call.** Its name contains "agent". Rename it. +- **Twice as many tool calls as the agent made.** The `tools` node span is counted as a tool call, as described above. It's a known gap in how Maple classifies LangGraph node spans. +- **Every model call appears twice.** A provider instrumentor or `LANGSMITH_OTEL_ENABLED` is also active. Keep one. +- **No lanes for sub-agents.** Their names aren't in `AGENT_NAMES`, or two agents share a name. +- **Turns split into several traces on Python 3.10.** Pass `config` to nested async calls, or upgrade to Python 3.11. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. +- [openinference-instrumentation-langchain](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-langchain): the instrumentor's source. +- [Trace with OpenTelemetry](https://docs.langchain.com/langsmith/trace-with-opentelemetry): LangSmith's OpenTelemetry export. +- [OpenRouter](/docs/agent-tracing/openrouter): if your models go through OpenRouter. diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md new file mode 100644 index 0000000000..725d36efaa --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -0,0 +1,399 @@ +--- +title: "Trace LiteLLM agents and the LiteLLM Proxy with OpenTelemetry" +description: "Send LiteLLM's OpenTelemetry spans to Maple from the Python SDK or the self-hosted LiteLLM Proxy, add the agent and tool spans LiteLLM can't emit, and group each conversation into one Agent Session." +group: "AI Agents" +order: 40 +navLabel: "LiteLLM" +icon: "litellm" +--- + +LiteLLM traces model calls, and only model calls. Its OpenTelemetry logger writes one span per `acompletion()`, with the model, the provider, token counts and the prompt and reply. LiteLLM has no agent loop, no tool executor and no notion of a conversation, so the loop around those calls is your code, and the agent and tool spans have to come from your code too. This guide shows the 60 lines of Python that add them. + +Two defaults go wrong for Maple. The default (v1) logger has no conversation id at all, so every model call lands in its own one-call session, and when your code already has a span open, v1 writes its attributes onto that span after it has ended and they are dropped. LiteLLM's newer v2 logger fixes both and turns `litellm_session_id` into `gen_ai.conversation.id`, but it is off by default, only traces the async API, and in LiteLLM 1.103 it silently disables itself on OpenTelemetry 1.44 or newer. + +This guide covers the LiteLLM Python SDK and the self-hosted LiteLLM Proxy. It was tested with `litellm` 1.103.0 and the OpenTelemetry Python SDK 1.43.0 on Python 3.12. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-litellm](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-litellm) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for LiteLLM in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-litellm -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Trace in your app or at the proxy, not both + +There are two places LiteLLM can trace a model call: + +- **In your app**, when your code calls `litellm.acompletion()` directly. The next sections cover this. +- **At the LiteLLM Proxy**, when your app sends OpenAI-compatible requests to a proxy you run. See [Trace at the LiteLLM Proxy](#trace-at-the-litellm-proxy). + +In both cases your app still emits the agent and tool spans, because only your app knows about them. Pick one place for the model-call spans. If the proxy traces the calls and your app also instruments its OpenAI client, every call shows up twice in the same trace. + +## Export LiteLLM SDK spans to Maple + +Install LiteLLM with the OpenTelemetry SDK and the OTLP/HTTP exporter: + +```bash +pip install "litellm==1.103.0" "opentelemetry-sdk==1.43.0" "opentelemetry-exporter-otlp-proto-http==1.43.0" +``` + +Keep OpenTelemetry below 1.44 on LiteLLM 1.103. OpenTelemetry 1.44 removed the Events API that LiteLLM's v2 logger imports, and LiteLLM catches the import error, logs `Error initializing custom logger: No module named 'opentelemetry._events'` once, and exports nothing ([BerriAI/litellm#41990](https://github.com/BerriAI/litellm/issues/41990)). The fix is merged and ships in LiteLLM 1.104; from that release on you can drop the pin. + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +Then set up tracing once, when your process starts. Create your own `TracerProvider` and hand it to LiteLLM's v2 logger, so LiteLLM's spans and yours go through one exporter: + +```py +# tracing.py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +import litellm +from litellm.integrations.otel.logger import OpenTelemetryV2 +from litellm.integrations.otel.model.config import OpenTelemetryV2Config + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +# LiteLLM's v2 OpenTelemetry logger, writing into the same provider as your own spans. +litellm.callbacks = [ + OpenTelemetryV2( + config=OpenTelemetryV2Config(capture_message_content="span_only"), + tracer_provider=provider, + ) +] + +tracer = trace.get_tracer("support-agent") +``` + +Import `tracing` at the top of your entry point, before the first model call. Passing the logger instance means you don't need `LITELLM_OTEL_V2=true`, and LiteLLM builds no provider or exporter of its own. If your app already has a `TracerProvider`, pass that one as `tracer_provider=` instead of creating a second. + +Don't also put `"otel"` in `litellm.callbacks` or `success_callback`. That adds a second logger, and every model call is exported twice. + +### Trace the agent loop around LiteLLM + +Each `acompletion()` becomes a `chat ` span. To get turns, sub-agents and tool calls in Maple, wrap each agent run in an `invoke_agent` span and each tool call in an `execute_tool` span. LiteLLM's spans nest under whatever span is current, so they land inside the agent span in the same trace: + +```py +# agent.py +import json +from contextlib import contextmanager +from dataclasses import dataclass, field + +import litellm +from opentelemetry.trace import StatusCode + +from tracing import tracer + + +@dataclass +class Agent: + name: str + model: str + instructions: str + tools: dict = field(default_factory=dict) # tool name -> function + schemas: list = field(default_factory=list) # OpenAI-style tool definitions + + +@contextmanager +def agent_span(name: str): + with tracer.start_as_current_span(f"invoke_agent {name}") as span: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", name) + yield span + + +def run_tool(agent: Agent, call) -> str: + name = call.function.name + with tracer.start_as_current_span(f"execute_tool {name}") as span: + span.set_attribute("gen_ai.operation.name", "execute_tool") + span.set_attribute("gen_ai.tool.name", name) + span.set_attribute("gen_ai.tool.call.id", call.id) + span.set_attribute("gen_ai.tool.call.arguments", call.function.arguments) + try: + result = agent.tools[name](**json.loads(call.function.arguments or "{}")) + except Exception as exc: + # The model gets the error as the tool result; the span is marked failed. + span.set_status(StatusCode.ERROR, str(exc)) + span.set_attribute("error.type", type(exc).__name__) + result = {"error": str(exc)} + output = json.dumps(result) + span.set_attribute("gen_ai.tool.call.result", output) + return output + + +async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: + with agent_span(agent.name): + while True: + response = await litellm.acompletion( + model=agent.model, + messages=[{"role": "system", "content": agent.instructions}, *messages], + tools=agent.schemas or None, + litellm_session_id=conversation_id, + ) + message = response.choices[0].message + messages.append(message.model_dump(exclude_none=True)) + if not message.tool_calls: + return message.content or "" + for call in message.tool_calls: + messages.append({"role": "tool", "tool_call_id": call.id, "content": run_tool(agent, call)}) +``` + +The v2 logger only traces the async API. `litellm.completion()` and the other sync calls produce no span at all, because the logger closes spans in its async success callback. Use `acompletion()`. If your code is sync throughout, see the troubleshooting entry on sync code. + +## Group every turn of a conversation into one session + +Maple groups LiteLLM traces into sessions by `gen_ai.conversation.id`. The v2 logger writes it on every `chat` span from `litellm_session_id=`, or from `metadata={"session_id": ...}` if you already pass metadata. Pass it on every call, from the chat or thread id your app already has: + +```py +from agent import Agent, run_agent + +assistant = Agent("assistant", "openrouter/openai/gpt-4o-mini", "You are a concise assistant.") +history: dict[str, list] = {} + + +async def handle_message(chat_id: str, text: str) -> str: + messages = history.setdefault(chat_id, []) + messages.append({"role": "user", "content": text}) + return await run_agent(assistant, chat_id, messages) +``` + +The id must be stable for the whole conversation and different between conversations. A per-process constant merges every user into one session. + +Without it, each turn is its own session named `trace:`, and with the v1 logger there is no way to set it on LiteLLM's spans at all. + +Leave `gen_ai.conversation.id` off your own `invoke_agent` span. Maple labels a session with the framework of its earliest span that carries the session id, so putting it on your span labels the session **Unidentified** instead of **LiteLLM**. The id on LiteLLM's `chat` spans is enough, because Maple groups whole traces: one span with the id pulls in every span of its trace. + +## Record prompts, responses and tool calls + +The v2 logger records no content by default. `capture_message_content="span_only"` in `tracing.py` turns it on, and each `chat` span then carries `gen_ai.input.messages` and `gen_ai.output.messages` as JSON in the OpenAI chat format: the system prompt, the history, tool calls and tool results. Maple builds the transcript from those attributes. The environment variable `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only` does the same when LiteLLM builds the config itself, as the proxy does. + +Don't use `event_only`. It moves content to OpenTelemetry log events, which Maple doesn't read, and the transcript is empty. + +To keep prompts out of Maple, set `capture_message_content="no_content"` and remove the `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result` lines from `run_tool`. Sessions keep their turns, models, tool names, tokens and errors, and the transcript is empty. + +`litellm.turn_off_message_logging = True` is the in-between option. It applies to every LiteLLM logging callback, and the messages keep their roles and tool calls, but each text becomes `redacted-by-litellm`. For pattern-based redaction of message content, run an OpenTelemetry Collector between your app and Maple. + +## Tools, errors and sub-agents + +Each tool call is an `execute_tool ` span with the tool name, the model's call id, the arguments and the result. Maple matches it to the model reply that requested it by `gen_ai.tool.call.id`. + +`run_tool` catches the exception, gives the model an `{"error": ...}` result so it can recover, and marks the span failed with `error.type` and an ERROR status. Maple counts the call as failed. If your loop instead returns error strings from tools without touching the span, Maple counts them as successes. + +A failed model call needs nothing from you. LiteLLM marks its `chat` span ERROR and sets `error.type` to the exception class, for example `RateLimitError`. + +For several agents, give each its own `name`. It becomes `gen_ai.agent.name` on its `invoke_agent` span, and Maple draws one lane per agent name. Run the workers inside an outer agent span so the whole run is one trace, and pass the same conversation id to every call: + +```py +import asyncio + +from agent import Agent, agent_span, run_agent + +# weather_worker, transport_worker and summary are Agent(...) instances with their own tools and models + + +async def briefing(conversation_id: str, city: str) -> str: + with agent_span("orchestrator"): + weather, transport = await asyncio.gather( + run_agent(weather_worker, conversation_id, [{"role": "user", "content": f"Weather in {city}?"}]), + run_agent(transport_worker, conversation_id, [{"role": "user", "content": f"Transport in {city}?"}]), + ) + notes = f"Weather: {weather}\nTransport: {transport}" + return await run_agent(summary, conversation_id, [{"role": "user", "content": notes}]) +``` + +`asyncio.gather` keeps the OpenTelemetry context, so the parallel workers are siblings under `invoke_agent orchestrator` and overlap in time. If a worker delegates through a tool call instead, call `run_agent` inside that tool's function. An `execute_tool` span whose only child is an `invoke_agent` span shows as a delegation, with the tool's arguments and result as the lane's input and output. + +## Tokens and cost + +Each `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. Your own spans carry no usage, so nothing is counted twice. + +For streaming, pass `stream_options={"include_usage": True}` and consume the stream inside the agent span. LiteLLM closes the `chat` span when the stream ends, with the token counts and `gen_ai.response.time_to_first_chunk`: + +```py +async def stream_agent(agent: Agent, conversation_id: str, messages: list): + with agent_span(agent.name): + stream = await litellm.acompletion( + model=agent.model, + messages=[{"role": "system", "content": agent.instructions}, *messages], + stream=True, + stream_options={"include_usage": True}, + litellm_session_id=conversation_id, + ) + text = "" + async for chunk in stream: + delta = chunk.choices[0].delta.content if chunk.choices else None + if delta: + text += delta + yield delta + messages.append({"role": "assistant", "content": text}) +``` + +Cost shows as unpriced by default. LiteLLM prices every call, but the v2 logger writes the price to `litellm.cost.total` (and v1 buries it in the `hidden_params` JSON), and Maple reads cost only from `gen_ai.usage.cost` and never prices tokens itself. To get cost into Maple, add up LiteLLM's price per agent run and put it on the `invoke_agent` span: + +```py +async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: + with agent_span(agent.name) as span: + cost = 0.0 + while True: + response = await litellm.acompletion( + model=agent.model, + messages=[{"role": "system", "content": agent.instructions}, *messages], + tools=agent.schemas or None, + litellm_session_id=conversation_id, + ) + cost += response._hidden_params.get("response_cost") or 0.0 + message = response.choices[0].message + messages.append(message.model_dump(exclude_none=True)) + if not message.tool_calls: + span.set_attribute("gen_ai.usage.cost", cost) + return message.content or "" + for call in message.tool_calls: + messages.append({"role": "tool", "tool_call_id": call.id, "content": run_tool(agent, call)}) +``` + +The session total is then correct, but the cost sits on the agent, not on each model call, so the per-model breakdown stays unpriced. For a stream, the price is on the last chunk's `usage.cost`. + +## Trace at the LiteLLM Proxy + +When your apps call a LiteLLM Proxy you run, trace the model calls there. Every app behind the proxy gets model-call spans without code changes, and content and retention settings live in one place. Enable the v2 logger in the proxy's `config.yaml`: + +```yaml +model_list: + - model_name: gpt-4o-mini + litellm_params: + model: openrouter/openai/gpt-4o-mini + api_key: os.environ/OPENROUTER_API_KEY + +litellm_settings: + callbacks: ["otel"] +``` + +and in the proxy's environment: + +```bash +LITELLM_OTEL_V2=true +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +OTEL_SERVICE_NAME=litellm-proxy +OTEL_ENVIRONMENT_NAME=production +OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only +``` + +The official `ghcr.io/berriai/litellm` image ships OpenTelemetry 1.28 and the FastAPI instrumentation, so the 1.44 problem above doesn't apply to it. A pip-installed proxy needs `opentelemetry-sdk`, `opentelemetry-exporter-otlp-proto-http` and `opentelemetry-instrumentation-fastapi` (1.43.0 and 0.64b0 with LiteLLM 1.103). + +In your app, keep `agent_span` and `run_tool` from above, drop the LiteLLM logger from `tracing.py`, and send two headers with every request: `traceparent`, so the proxy's spans join the app's trace under the agent span, and `x-litellm-session-id`, which the proxy turns into `gen_ai.conversation.id`: + +```py +import os + +from openai import AsyncOpenAI +from opentelemetry import propagate + +client = AsyncOpenAI(base_url="http://localhost:4000", api_key=os.environ["LITELLM_API_KEY"]) + + +async def call_model(conversation_id: str, messages: list, tools: list | None): + headers = {"x-litellm-session-id": conversation_id} + propagate.inject(headers) # adds traceparent for the current span + return await client.chat.completions.create( + model="gpt-4o-mini", messages=messages, tools=tools, extra_headers=headers + ) +``` + +`metadata: {"session_id": ...}` in the request body works as well. A W3C `baggage` header with `session.id=...` is picked up only when neither is present. + +Don't add an OpenAI client instrumentor (OpenInference, OpenLLMetry, `opentelemetry-instrumentation-openai-v2`) to the app on top of this. The app's span and the proxy's span describe the same call, and Maple can't always tell them apart, so LLM call counts and tokens double. If you can't configure the proxy, do the opposite: leave its OpenTelemetry off and instrument the client in the app. + +The proxy also emits its own housekeeping spans (database, Redis, guardrails) in the same trace. They carry no model or tokens, and appear in the trace view next to the `chat` span. + +## Short-lived processes + +LiteLLM creates its `chat` span after the call returns, from a background logging queue. A script that exits right after its last call loses that call's span: the queue is drained at interpreter exit, after the `TracerProvider` has already shut down. Drain the queue, then flush, before the event loop ends: + +```py +import asyncio + +from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER + +from tracing import provider + + +async def flush_tracing() -> None: + await asyncio.sleep(0) # LiteLLM queues its log event on the next loop tick + await GLOBAL_LOGGING_WORKER.flush() + provider.force_flush() + + +async def main() -> None: + try: + print(await handle_message("chat-42", "What's the weather in Berlin?")) + finally: + await flush_tracing() + + +asyncio.run(main()) +provider.shutdown() +``` + +Call `flush_tracing()` at the end of every Lambda or Cloud Run job invocation, after each notebook cell that calls a model, and in your web framework's shutdown hook. A long-running server needs it only at shutdown. + +## Check that it works + +Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: + +- one session per conversation, with the id you passed as `litellm_session_id`, and framework **LiteLLM**; +- one turn per `invoke_agent` span, labeled with the user's message, and a transcript with the prompts, replies and tool calls; +- `invoke_agent ` spans from your code, `chat ` spans from LiteLLM inside them, and `execute_tool ` spans for tool calls; +- a lane per agent name for multi-agent runs, all in the caller's session; +- input and output tokens on every `chat` span, including streamed ones; +- failed tool calls marked as failed, with the error as the result; +- cost shown as unpriced, or the per-run total if you added `gen_ai.usage.cost`. + +## Troubleshooting + +- **No LiteLLM spans at all, and the log shows `No module named 'opentelemetry._events'`.** LiteLLM 1.103 with OpenTelemetry 1.44 or newer. Pin `opentelemetry-sdk` and the exporter to 1.43.0, or upgrade to LiteLLM 1.104. +- **Your spans arrive but no `chat` spans.** The code calls the sync `litellm.completion()`, which the v2 logger doesn't trace. Switch to `acompletion()`. +- **Spans named `litellm_request` and `raw_gen_ai_request`, and each call is its own session.** That is the v1 logger: `litellm.callbacks = ["otel"]` without the v2 instance. Use the `OpenTelemetryV2` setup above. +- **Model, tokens and prompts missing, and the SDK logs `Setting attribute on ended span`.** Also v1: with a span already open it writes onto that span instead of creating its own. If you must stay on v1 for sync code, set `USE_OTEL_LITELLM_REQUEST_SPAN=true` and `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental`, and put `gen_ai.conversation.id` on your `invoke_agent` span, since v1 can't carry it. The framework then shows as **Unidentified**. +- **Every call is its own session.** No `litellm_session_id=` (SDK) or `x-litellm-session-id` header (proxy) was sent. Pass it on every call. +- **The framework shows as Unidentified.** Your own span carries `gen_ai.conversation.id` or `maple_ai.session.id`. Remove it and let LiteLLM's spans carry the id. +- **Spans arrive with no prompts or replies.** Content capture is off, which is the v2 default. Set `capture_message_content="span_only"` or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only`. +- **The last call of a script or Lambda is missing.** The process ended before LiteLLM's logging queue ran. Await `flush_tracing()` before the loop ends. +- **Every model call appears twice.** Two loggers (the v2 instance plus `"otel"` in `litellm.callbacks`), or the proxy and an in-app OpenAI instrumentor both tracing the same call. Keep one. +- **Proxy spans land in their own traces, apart from the app's agent span.** The request carried no `traceparent`. Inject it with `propagate.inject(headers)` inside the agent span, and don't set `OTEL_IGNORE_CONTEXT_PROPAGATION` on the proxy. +- **Nothing arrives from the proxy.** It is still on the default `console` exporter because no endpoint reached it. Check that `OTEL_EXPORTER_OTLP_ENDPOINT` is set in the proxy's environment, not only in your app's. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [All agent tracing guides](/docs/agent-tracing) +- [Trace OpenRouter calls with Broadcast](/docs/agent-tracing/openrouter) +- [LiteLLM: OpenTelemetry v2](https://docs.litellm.ai/docs/observability/opentelemetry_v2) +- [LiteLLM: OpenTelemetry (v1)](https://docs.litellm.ai/docs/observability/opentelemetry_integration) +- [Instrument a Python application](/docs/guides/instrumentation-python) diff --git a/apps/landing/src/content/docs/agent-tracing/llamaindex.md b/apps/landing/src/content/docs/agent-tracing/llamaindex.md new file mode 100644 index 0000000000..5bc3fc98ad --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/llamaindex.md @@ -0,0 +1,324 @@ +--- +title: "Trace LlamaIndex agents with OpenTelemetry" +description: "Send LlamaIndex FunctionAgent, AgentWorkflow and Workflow runs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools and sub-agent lanes." +group: "AI Agents" +order: 23 +navLabel: "LlamaIndex" +icon: "llamaindex" +--- + +LlamaIndex reports what it does through its own instrumentation dispatcher: every agent run, workflow step, model call and tool call opens a dispatcher span and fires events. Two packages turn those into OpenTelemetry spans. LlamaIndex's own `llama-index-observability-otel` copies the span tree but puts the payload in span events, and it loses the model's reply and token usage on every streamed call. OpenInference's `openinference-instrumentation-llama-index` writes model, tokens, messages and tool results as span attributes, which is what Maple reads. Use the OpenInference one. + +It needs four adjustments before a chat looks right in Maple. The instrumentor writes OpenInference attribute names unless you turn on its GenAI output, it never sets a conversation id, it doesn't record agent names, and it opens two or three "LLM" spans for every model call, so Maple would count each call two or three times. This guide covers all four for llama-index-core 0.14.25 with `openinference-instrumentation-llama-index` 4.5.2 on Python 3.10 or later, using `FunctionAgent`, `AgentWorkflow` and custom `Workflow` classes. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-llamaindex](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-llamaindex) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for LlamaIndex in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-llamaindex -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Why not llama-index-observability-otel + +LlamaIndex's docs point to `LlamaIndexOpenTelemetry` from `llama-index-observability-otel` (0.7.0). It's a bridge: it mirrors dispatcher spans into OpenTelemetry spans and dispatcher events into span events. In Maple that gives you structure and nothing else: + +- **No model, tokens or transcript.** Model settings and the prompt sit inside an `LLMChatStartEvent` span event, and Maple never reads span events. The matching end event, with the reply and the usage, is dropped because the streamed `astream_chat` span closes before the stream is consumed. +- **Every call appears twice.** Each model call has two nested `OpenRouter.astream_chat` spans (or `OpenAI.astream_chat`, after your model class). +- **It ignores the standard exporter variables.** `OTEL_EXPORTER_OTLP_ENDPOINT` does nothing; the default exporter is `ConsoleSpanExporter`, so without an explicit `span_exporter=` everything goes to stdout. + +If you already run it and only want sessions, `instrument_tags({"gen_ai.conversation.id": conversation_id})` around `agent.run()` groups the traces; dotted tag keys become span attributes verbatim. For transcripts and tokens, switch to OpenInference. Don't run both, or every span exists twice. + +## Install the instrumentor and export to Maple + +```bash +pip install "llama-index-core>=0.14.25" "openinference-instrumentation-llama-index>=4.5.2" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Add your model package (`llama-index-llms-openai`, `llama-index-llms-anthropic`, `llama-index-llms-openrouter`, ...) as usual. The instrumentor requires llama-index-core 0.14.19 or later. On anything older it logs a dependency conflict and instruments nothing. + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL: `OTLPSpanExporter()` with no arguments appends `/v1/traces`. If you pass `endpoint=` in code instead, it's used as is and has to end in `/v1/traces`. + +Then add a `tracing.py` and import it at the top of your entry point, before the first `agent.run()`: + +```py +# tracing.py +from llama_index.core.instrumentation.dispatcher import active_instrument_tags +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.llama_index import LlamaIndexInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +LLM_METHODS = (".chat", ".achat", ".stream_chat", ".astream_chat", + ".complete", ".acomplete", ".stream_complete", ".astream_complete") + + +class LlamaIndexForMaple(SpanProcessor): + """Sits in front of the exporter: one span per model call, agent names, no false HITL failures.""" + + def __init__(self, exporter_processor: SpanProcessor): + self._next = exporter_processor + self._open_llm_spans = {} + + def on_start(self, span, parent_context=None): + # instrument_tags({"gen_ai.agent.name": ...}) becomes an attribute, so sub-agents get lanes + agent_name = active_instrument_tags.get().get("gen_ai.agent.name") + if agent_name: + span.set_attribute("gen_ai.agent.name", agent_name) + if span.name.endswith(LLM_METHODS): + self._open_llm_spans[span.context.span_id] = span + self._next.on_start(span, parent_context) + + def on_end(self, span): + self._open_llm_spans.pop(span.context.span_id, None) + if span.name.endswith("._prepare_chat_with_tools"): + return # builds the request; never calls the model + if (span.status.description or "").startswith("WaitingForEvent"): + return # ctx.wait_for_event() suspends the tool and replays it later; not a failure + outer = self._open_llm_spans.get(span.parent.span_id) if span.parent else None + if outer is not None and outer.name == span.name: + outer.set_attributes(span.attributes) # the inner twin holds the messages and usage + return + self._next.on_end(span) + + def shutdown(self): + self._next.shutdown() + + def force_flush(self, timeout_millis=30000): + return self._next.force_flush(timeout_millis) + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))) +trace.set_tracer_provider(provider) + +LlamaIndexInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +What each part does: + +- **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.conversation.id` next to its OpenInference attributes when each span ends. Without it, Maple can count tokens but the session page has no transcript. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, but only if it's set before `TraceConfig` is built. +- **`LlamaIndexForMaple`** wraps the exporting processor and fixes what the instrumentor gets wrong for LlamaIndex. The next sections explain each fix. +- **One `TracerProvider`.** If the app already has one (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second. Add `LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))` to the existing provider and pass that provider to `instrument()`. + +### Why each model call opens three spans + +LlamaIndex's dispatcher wraps every method that implements an abstract base method, and the instrumentor marks every span on an LLM object as kind `LLM`. A single `FunctionAgent` step with an OpenAI-compatible model (`OpenRouter`, `OpenAILike` and the others built on it) produces: + +```text +BaseWorkflowAgent.run_agent_step +├── OpenRouter._prepare_chat_with_tools kind LLM, builds the request, ~1 ms +└── OpenRouter.astream_chat kind LLM, OpenAILike's override + └── OpenRouter.astream_chat kind LLM, OpenAI's method: messages, usage +``` + +Maple counts every inference span without usage from a reporting ancestor as a model call, so this one call would count three times. A class that implements the call itself, such as `OpenAI`, produces two spans, the helper and the call. Tokens are not doubled, since only the innermost span carries usage. + +`LlamaIndexForMaple` drops the `_prepare_chat_with_tools` helper, copies the inner span's attributes onto its same-named parent and drops the inner span, leaving one `OpenRouter.astream_chat` span per call with the messages and usage. It only merges spans with identical names that nest directly, so a chat engine's `CondensePlusContextChatEngine.chat` calling `OpenAI.chat` is left alone. + +## Group a conversation into one session + +LlamaIndex keeps a conversation in a workflow `Context` (`Context(agent)`, or the chat memory you pass), and every `agent.run()` starts a new root span and a new trace. Nothing in those spans says which conversation they belong to, so without a conversation id every message becomes its own one-turn session in Maple, named after its trace id. + +Wrap each `agent.run()` call in OpenInference's `using_session` with your app's conversation id: + +```py +from llama_index.core.agent.workflow import AgentStream, FunctionAgent +from llama_index.core.instrumentation.dispatcher import instrument_tags +from llama_index.core.workflow import Context +from openinference.instrumentation import using_session + +agent = FunctionAgent(name="assistant", llm=llm, tools=[get_weather, calculate], + system_prompt="You are a helpful assistant.") +contexts: dict[str, Context] = {} + + +async def handle_message(conversation_id: str, text: str): + if conversation_id not in contexts: + contexts[conversation_id] = Context(agent) + ctx = contexts[conversation_id] + with using_session(conversation_id), instrument_tags({"gen_ai.agent.name": agent.name}): + handler = agent.run(user_msg=text, ctx=ctx) + async for event in handler.stream_events(): + if isinstance(event, AgentStream): + yield event.delta + await handler +``` + +The instrumentor copies the id onto every span it creates as `session.id`, and the GenAI output repeats it as `gen_ai.conversation.id`, the key Maple reads for these spans. `using_session` sets a contextvar, and `agent.run()` starts the workflow's tasks immediately, so the id reaches every step, tool and model call of the run, including parallel workflow steps and the replay after a human-in-the-loop approval. Only the `agent.run()` call has to be inside the `with`; consuming the stream outside it is fine, which keeps the context manager out of your streaming generator. + +Use the id your app already stores the chat under. A new UUID per request gives you one session per message again, and a constant puts every user in one session. Keep one `Context` per conversation too: a shared `Context` shares the chat memory, and the traces would look like one long conversation. + +`llama_index.core.workflow.Context` is not a session id: it never reaches a span. Neither is the per-run `llamaindex.run_id` of the native package. + +## Record prompts, responses and tool calls + +Content capture is on by default. Every model span carries the full message list sent to the model as `gen_ai.input.messages` (system prompt, chat history, tool results) and the reply as `gen_ai.output.messages`, including tool call parts with their ids and arguments. Maple renders these as the session transcript, with the last user message as the turn label. + +The input list grows with the conversation: with a persistent `Context`, every model span repeats the whole chat so far. Maple has no per-attribute limit, and ingest accepts requests up to 20 MiB. + +To keep prompts and outputs out of your traces: + +```py +config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) +``` + +`hide_inputs` drops the input messages and replaces `input.value` with `__REDACTED__`; `hide_outputs` does the same for outputs and tool results. The session still shows turns, model and tool calls, tokens and failures, with an empty transcript. `hide_input_text` and `hide_output_text` keep the message structure and redact only the text. Each switch also has an `OPENINFERENCE_HIDE_*` environment variable. For pattern-based redaction (emails, card numbers), use the `redaction` processor in an OpenTelemetry Collector. + +## Tools, errors and sub-agents + +Each tool call is a `FunctionTool.acall` span of kind `TOOL`, with `gen_ai.operation.name` `execute_tool`, the tool's name in `gen_ai.tool.name`, its description, and its return value (LlamaIndex's `ToolOutput`, with `raw_input` and `raw_output`) in `gen_ai.tool.call.result`. It sits under a `BaseWorkflowAgent.call_tool` step span. + +A tool that raises is marked failed without extra code. `FunctionTool.acall` ends with status `ERROR` and the exception as the status message, for example `RuntimeError: transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. The `call_tool` step stays `OK`, because `FunctionAgent` catches the error and hands it to the model as the tool result. A tool that returns an error string instead of raising looks like a success. + +Two gaps remain on tool spans. `gen_ai.tool.call.arguments` holds the tool's parameter schema, not the call's arguments, because the GenAI output copies OpenInference's `tool.parameters` there; the real arguments are in the model's tool call part in the transcript and in `raw_input` inside the result. And tool spans carry no `gen_ai.tool.call.id`, so Maple can't link a tool span to the exact call in the model's reply. + +Human-in-the-loop tools that call `ctx.wait_for_event(...)` run twice: the first call raises LlamaIndex's `WaitingForEvent` to suspend the step, and the step replays once the `HumanResponseEvent` arrives. The instrumentor records the first attempt as a failed tool call. `LlamaIndexForMaple` drops it, so an approved call shows as one successful tool call. + +### Name your agents + +The instrumentor doesn't record which agent a span belongs to, and without `gen_ai.agent.name` Maple has no agent facet and no lanes. `LlamaIndexForMaple` copies a `gen_ai.agent.name` tag from LlamaIndex's `instrument_tags` onto every span started inside it. Put the tag around each agent's `run()`: + +```py +from llama_index.core.workflow import Context, Event, StartEvent, StopEvent, Workflow, step + + +class WeatherTask(Event): + city: str + + +class TransportTask(Event): + city: str + + +class WorkerDone(Event): + text: str + + +async def run_agent(agent: FunctionAgent, message: str) -> str: + with instrument_tags({"gen_ai.agent.name": agent.name}): + handler = agent.run(user_msg=message) + return str(await handler) + + +class Briefing(Workflow): + @step + async def plan(self, ctx: Context, ev: StartEvent) -> WeatherTask | TransportTask | None: + ctx.send_event(WeatherTask(city=ev.city)) + ctx.send_event(TransportTask(city=ev.city)) + + @step(num_workers=2) + async def weather(self, ev: WeatherTask) -> WorkerDone: + return WorkerDone(text=await run_agent(weather_worker, f"Weather in {ev.city}?")) + + @step(num_workers=2) + async def transport(self, ev: TransportTask) -> WorkerDone: + return WorkerDone(text=await run_agent(transport_worker, f"Transport in {ev.city}?")) + + @step + async def summarize(self, ctx: Context, ev: WorkerDone) -> StopEvent | None: + done = ctx.collect_events(ev, [WorkerDone, WorkerDone]) + if done is None: + return None + return StopEvent(result=await run_agent(summary_agent, "\n\n".join(d.text for d in done))) +``` + +Run the whole workflow inside `using_session(conversation_id)`. The workflow is one trace: `Briefing.run` at the root, one span per step, and each worker's `FunctionAgent.run` under its step with its own `gen_ai.agent.name`. Maple opens a lane for every agent span whose name differs from its caller's. Parallel steps (`num_workers`) stay in the same trace and keep the session and agent tags. + +The same works for agents called as tools: put `instrument_tags` inside the tool function around the sub-agent's `run()`. The tool span with one `FunctionAgent.run` child then shows as a delegation, with the tool's arguments and result as the lane's input and output. + +`AgentWorkflow` handoffs are different. The whole multi-agent run is one `AgentWorkflow.run` span, and LlamaIndex switches the active agent inside it without a span per agent, so there is nothing to tag. Handoffs show as one agent in Maple. If you need lanes, run each agent as its own `FunctionAgent.run()` from a workflow step or a tool, as above. + +## Tokens and cost + +The model span carries input and output tokens from the provider's reply as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus cached input tokens when the provider reports them, next to the OpenInference `llm.token_count.*` originals. The model is the one you configured (`gen_ai.request.model`, for example `openai/gpt-4o-mini`); the instrumentor doesn't record the model name the provider returns, or a response id. + +`FunctionAgent` streams every model call by default, even when you never read `AgentStream` events, and usage on a stream arrives only in the last chunk. OpenRouter always sends it. OpenAI's API sends it only when asked, so pass `stream_options` on OpenAI-compatible models: + +```py +from llama_index.llms.openai import OpenAI + +llm = OpenAI(model="gpt-4o-mini", additional_kwargs={"stream_options": {"include_usage": True}}) +``` + +LlamaIndex strips the option from non-streaming requests, so it's safe to set once. + +The provider comes from the model class: `OpenRouter` and every `OpenAILike` model report `openai`, even for an Anthropic model behind OpenRouter. + +Maple shows cost only when a span carries one, and neither LlamaIndex nor the instrumentor records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call, and Maple joins them to the same session. + +Don't add `openinference-instrumentation-openai` (or another provider instrumentor) next to the LlamaIndex one. It wraps the same HTTP call and gives every model call a second span with its own usage. + +## Flush spans before the process exits + +`BatchSpanProcessor` exports every 5 seconds, and the `TracerProvider` flushes on a normal interpreter exit. That doesn't happen when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: + +```py +from tracing import provider + +try: + result = await workflow.run(city="Amsterdam") +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. + +## Check that it works + +Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: + +- **One session** for the conversation, with one turn per `agent.run()`. Each turn's trace starts at `FunctionAgent.run` (or `AgentWorkflow.run`, or your workflow class's `.run`). +- **Framework: Unidentified.** Maple doesn't recognise the OpenInference LlamaIndex scope as LlamaIndex yet, so the session is filed under the generic GenAI bucket. Everything else on this list still works. +- **The transcript**: your messages, the model's replies and the tool calls it made. +- **Model calls** named after your model class and method, for example `OpenRouter.astream_chat` or `OpenAI.achat`, one per call, each with a model and input and output tokens. +- **Tool calls** named `FunctionTool.acall`, with the tool's name and result. +- **Agents**: `assistant`, plus one lane per tagged sub-agent. +- **Cost**: unpriced. + +A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. + +## Troubleshooting + +- **No spans at all.** `instrument()` never ran, or llama-index-core is older than 0.14.19 and the instrumentor skipped itself (look for `DependencyConflict` in the logs). Import `tracing` first in the entry point and look for an `OTLPSpanExporter` error in the logs. +- **Spans print to the console instead of reaching Maple.** You're running `LlamaIndexOpenTelemetry` without `span_exporter=`. Switch to the OpenInference setup above. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. +- **Tokens in the list, empty session page.** The GenAI output is off. Pass `TraceConfig(enable_genai_semconv=True)`. +- **One session per message.** `agent.run()` isn't inside `using_session(...)`, or the id changes per request. `Context` and `llamaindex.run_id` are not session ids. +- **Each model call counted two or three times.** `LlamaIndexForMaple` isn't in front of the exporter. Add the exporter through it, not directly with `add_span_processor(BatchSpanProcessor(...))`. +- **Model calls take 1 ms.** Streamed model spans end when LlamaIndex hands back the stream, not when the last token arrives, so their duration isn't the model's latency. The enclosing `BaseWorkflowAgent.run_agent_step` span has the real time. If you don't stream tokens to users, `FunctionAgent(..., streaming=False)` records model spans with their full duration. +- **No tokens on streamed calls.** The provider didn't send usage on the stream. Add `stream_options={"include_usage": True}` through `additional_kwargs`. +- **Tool arguments show the tool's schema.** A known gap in the instrumentor's GenAI output; see [Tools, errors and sub-agents](#tools-errors-and-sub-agents). +- **An approved tool call shows as failed first.** `wait_for_event()` suspends the tool by raising `WaitingForEvent`. `LlamaIndexForMaple` drops that attempt; check it's installed. +- **No lanes, no agent names.** Wrap each agent's `run()` in `instrument_tags({"gen_ai.agent.name": agent.name})`, with `LlamaIndexForMaple` installed. `AgentWorkflow` handoffs can't be split into lanes. +- **Tool calling silently doesn't happen.** `OpenAILike` and `OpenRouter` default to `is_function_calling_model=False`, so the agent falls back to prose and emits no tool spans. Pass `is_function_calling_model=True`. +- **Streaming query engine spans land in separate traces.** llama-index-core 0.14.25 defers the model call of a `StreamingResponse` until you consume it, after the query span has ended. OpenInference is fixing this in [PR #3841](https://github.com/Arize-ai/openinference/pull/3841); until it ships, pin `llama-index-core<0.14.25` if you trace streaming query engines. Agents are not affected. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. +- [LlamaIndex observability](https://developers.llamaindex.ai/python/framework/module_guides/observability/): LlamaIndex's own page on tracing. +- [openinference-instrumentation-llama-index](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-llama-index): the instrumentor's source and `TraceConfig` options. +- [OpenRouter](/docs/agent-tracing/openrouter): cost per call if your models go through OpenRouter. diff --git a/apps/landing/src/content/docs/agent-tracing/mastra.md b/apps/landing/src/content/docs/agent-tracing/mastra.md new file mode 100644 index 0000000000..31e46d1d02 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/mastra.md @@ -0,0 +1,325 @@ +--- +title: "Trace Mastra agents and workflows with OpenTelemetry" +description: "Export Mastra's built-in GenAI spans to Maple with @mastra/otel-exporter so each conversation is one Agent Session with its transcript, tool calls, sub-agents and tokens." +group: "AI Agents" +order: 11 +navLabel: "Mastra" +icon: "mastra" +--- + +Mastra traces itself. Every `agent.generate()` or `agent.stream()` produces an `invoke_agent` span, a `chat` span per model call with the prompt, the reply and the token counts, and an `execute_tool` span per tool call with its arguments and result. The `@mastra/otel-exporter` package turns those spans into OpenTelemetry GenAI spans (semantic conventions v1.38) and sends them to any OTLP endpoint, including Maple. You don't need an OpenTelemetry SDK or an instrumentation package. + +Two things go wrong by default. The exporter ignores the standard `OTEL_EXPORTER_OTLP_*` variables and falls back to OTLP/JSON, so a setup copied from another guide exports nothing and logs a single warning. And the conversation id Maple groups by is Mastra's memory thread id: an agent called without `memory: { thread, resource }` sends no id at all, so every message becomes its own session. + +This guide covers Mastra 1.x on Node.js 22.13 or newer. It was written against `@mastra/core` 1.71, `@mastra/observability` 1.18 and `@mastra/otel-exporter` 1.4. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-mastra](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-mastra) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Mastra in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-mastra -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Export Mastra spans to Maple + +Install the observability packages next to `@mastra/core`, plus the OTLP/protobuf trace exporter: + +```bash +npm install @mastra/observability@latest @mastra/otel-exporter@latest @opentelemetry/exporter-trace-otlp-proto +``` + +Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week. The exporter decides which span is the model call from features both packages report, and a stale `@mastra/observability` can leave you with spans but no model calls. + +Then configure observability on your `Mastra` instance: + +```ts +// src/mastra/index.ts +import { Mastra } from "@mastra/core/mastra" +import { SpanType } from "@mastra/core/observability" +import { Observability } from "@mastra/observability" +import { OtelExporter } from "@mastra/otel-exporter" +import { supportAgent } from "./agents/support" + +export const mapleExporter = new OtelExporter({ + provider: { + custom: { + endpoint: "https://ingest.maple.dev", + protocol: "http/protobuf", + headers: { Authorization: `Bearer ${process.env.MAPLE_INGEST_KEY}` }, + }, + }, + resourceAttributes: { "deployment.environment.name": "production" }, +}) + +export const mastra = new Mastra({ + agents: { supportAgent }, + observability: new Observability({ + configs: { + maple: { + serviceName: "support-agent", + exporters: [mapleExporter], + // One span per streamed chunk adds nothing Maple uses + excludeSpanTypes: [SpanType.MODEL_CHUNK], + }, + }, + }), +}) +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself, and strips it if you include it, so both forms work. + +Three details in this block are load-bearing: + +- `observability` must be an `Observability` instance. A plain `{ configs: ... }` object logs a warning and installs a no-op, so nothing is exported. +- `protocol: "http/protobuf"` must be explicit. The `custom` provider defaults to `http/json`, and each protocol loads a different exporter package; if that package is missing, tracing is disabled with one error line at startup. +- The endpoint and key go in code. `OTEL_EXPORTER_OTLP_ENDPOINT` and `OTEL_EXPORTER_OTLP_HEADERS` are not read by this exporter. The `headers` object is sent as is, so there is no `%20` encoding to get wrong. + +Only agents and workflows registered on this `Mastra` instance, or called with a `tracingContext` from one, are traced. An `Agent` you construct and call on its own, outside `mastra.getAgent()`, has no observability attached. + +### The exporter also sends logs + +`OtelExporter` exports two signals by default: traces to `/v1/traces` and Mastra's own log records (warnings and errors, such as a tool that threw) to `/v1/logs`. Maple accepts both. The logs carry the trace and span ids of the run that wrote them. Agent Sessions reads only the spans. To send traces only: + +```ts +new OtelExporter({ + provider: { custom: { /* as above */ } }, + signals: { logs: false }, +}) +``` + +### An app that already has OpenTelemetry + +The exporter runs its own `BatchSpanProcessor` and doesn't touch a global `TracerProvider`, so it coexists with an existing Node SDK setup. The Mastra spans then form their own traces, separate from your HTTP spans. If you want Mastra's spans nested under your request spans, use Mastra's [OpenTelemetry bridge](https://mastra.ai/reference/observability/tracing/bridges/otel) instead of `OtelExporter`, and point your existing exporter at Maple. Use one or the other, not both, or every span arrives twice. + +## Group every turn of a conversation into one session + +Maple groups traces into sessions by `gen_ai.conversation.id`. Mastra writes it on every span from the memory thread id of the run, and only from there. Each `generate()` or `stream()` call is its own trace, so a chat that sends one request per user message needs the same thread id on every call: + +```ts +const agent = mastra.getAgent("supportAgent") + +export async function handleMessage(chatId: string, userId: string, text: string) { + const result = await agent.generate(text, { + memory: { thread: chatId, resource: userId }, + }) + return result.text +} +``` + +The thread id must be stable for the whole conversation and different between conversations. Use the chat or conversation id your app already has; a per-process constant merges every user into one session. + +This is the same `memory` option that makes the agent remember earlier messages, so an agent with `memory: new Memory(...)` that already passes it needs nothing more. If you skip it, each message shows up in **Agent Sessions** as its own one-turn session named after the trace id. Mastra's server routes (`mastra dev`, `@mastra/server`) pass the thread the client sends, or the one your middleware sets as `mastra__threadId` in the request context. + +Streaming works the same way. Consume the whole stream: the `invoke_agent` span ends, and gets exported, when the stream finishes. + +```ts +const stream = await agent.stream(text, { memory: { thread: chatId, resource: userId } }) +for await (const chunk of stream.textStream) { + res.write(chunk) +} +``` + +### Workflows and agents without memory + +A workflow run has no thread, and neither does an agent you call without `memory`. Put the id in the root span's metadata instead. Mastra copies root metadata to every span in the trace, and the exporter turns `metadata.threadId` into `gen_ai.conversation.id`: + +```ts +const run = await mastra.getWorkflow("briefingWorkflow").createRun() +const result = await run.start({ + inputData: { request }, + tracingOptions: { metadata: { threadId: conversationId } }, +}) +``` + +The same `tracingOptions` works on `agent.generate()` for an agent that has no memory configured. + +## Record prompts, responses and tool calls + +Content capture is on by default. Each `chat` span carries `gen_ai.input.messages` and `gen_ai.output.messages` as JSON in the GenAI message format, the `invoke_agent` span carries the agent's instructions as `gen_ai.system_instructions`, and each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Maple builds the transcript from those attributes. + +Mastra bounds what it serializes into a span. The defaults are 128 KiB per string, 50 items per array, 50 keys per object and 8 levels deep. A cut string ends in `…[truncated]`, which Maple shows as truncated. A message list longer than 50 entries loses the newest messages, so a turn with a long history can be missing its latest user message. Raise the array limit if your agents keep long histories: + +```ts +new Observability({ + configs: { + maple: { + serviceName: "support-agent", + exporters: [mapleExporter], + serializationOptions: { maxArrayLength: 200 }, + }, + }, +}) +``` + +To keep content out of Maple for one request, hide it for the whole trace: + +```ts +await agent.generate(text, { + memory: { thread: chatId, resource: userId }, + tracingOptions: { hideInput: true, hideOutput: true }, +}) +``` + +Sessions keep their turns, models, tool names, tokens and errors, but the transcript is empty and tool calls have no arguments or results. + +Mastra also applies a `SensitiveDataFilter` to every span by default. It redacts values under keys like `password`, `token`, `apiKey`, `authorization` and `secret`, including inside JSON strings, and replaces them with `[REDACTED]`. It matches key names, not free text, so a user who types a password into the chat still sends it to Maple. If you need pattern-based redaction of message text, run it in an OpenTelemetry Collector between your app and Maple. + +## Tools, errors and sub-agents + +Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the model's `gen_ai.tool.call.id`, the tool description, and the arguments and result. Maple matches each call to the model reply that requested it by that id. + +A tool that throws is marked failed: status ERROR, `error.type` set to Mastra's error id (`TOOL_EXECUTION_FAILED`), and the exception message as the status message. The agent keeps running and the model sees the error. A tool that returns an error value instead of throwing stays green, and Maple counts the call as a success, so throw for failures you want to see: + +```ts +import { createTool } from "@mastra/core/tools" +import { z } from "zod" + +export const fetchTransportData = createTool({ + id: "fetch_transport_data", + description: "Fetch public transport options for a city.", + inputSchema: z.object({ city: z.string() }), + execute: async () => { + throw new Error("transport data service unavailable (503)") + }, +}) +``` + +Tools with `requireApproval: true` suspend the run before the tool executes. The resumed run (`approveToolCallGenerate()` or `declineToolCallGenerate()`) continues the same trace; pass the same `memory` to it. A declined call produces no `execute_tool` span. + +### Sub-agents: keep one conversation id + +Mastra's multi-agent idiom is a supervisor: an agent with an `agents` property calls each sub-agent through a tool named `agent-`. Each delegation runs the sub-agent inside the supervisor's trace, under that tool span. Give every agent a `name`; it becomes `gen_ai.agent.name`, and Maple draws one lane per agent name. An `execute_tool agent-weather_worker` span whose only child is `invoke_agent weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. + +The catch is memory. When the supervisor runs with a thread, Mastra gives each delegation its own new thread id, so the sub-agent's spans carry a different `gen_ai.conversation.id` from the conversation they belong to. One trace then has several ids, Maple picks one of them for the whole trace, not necessarily yours, and it can split the turn into one turn per id. Add a span processor that copies the root span's thread id to every span of the trace: + +```ts +// src/mastra/conversation-id.ts +import type { SpanOutputProcessor } from "@mastra/core/observability" + +export const conversationIdFromRoot: SpanOutputProcessor = { + name: "conversation-id-from-root", + process(span) { + let root = span + while (root?.parent) root = root.parent + const threadId = root?.metadata?.threadId + if (span && threadId) span.metadata = { ...span.metadata, threadId } + return span + }, + async shutdown() {}, +} +``` + +```ts +new Observability({ + configs: { + maple: { + serviceName: "support-agent", + exporters: [mapleExporter], + spanOutputProcessors: [conversationIdFromRoot], + }, + }, +}) +``` + +The processor also covers workflows whose steps call agents with their own `memory`. Pass `tracingContext` from the step's `execute` arguments to `agent.generate()`, so the agent's spans join the workflow's trace instead of starting a new one: + +```ts +const weatherStep = createStep({ + id: "weather_worker", + inputSchema: z.object({ city: z.string() }), + outputSchema: z.object({ findings: z.string() }), + execute: async ({ inputData, tracingContext }) => { + const res = await weatherWorker.generate(`Report the weather in ${inputData.city}.`, { tracingContext }) + return { findings: res.text } + }, +}) +``` + +Steps in `.parallel([...])` run concurrently, and their lanes overlap in time in Maple. A workflow run's root span is `invoke_workflow `; Maple treats it as the turn. + +## Tokens and cost + +The `chat` span of each model call carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. Only that span carries usage, so the enclosing agent, step and generation spans add nothing to the total. + +Reasoning tokens are exported as `gen_ai.usage.reasoning_tokens`, a key Maple doesn't read. They're still inside the output token count, so totals are right; only the reasoning breakdown is missing. + +Streamed calls report usage too. Time to first token is exported only as `mastra.completion_start_time`, a timestamp Maple doesn't read, so sessions show no time to first token. + +Cost shows as unpriced. Mastra doesn't put a cost attribute on its spans, and Maple never prices tokens itself. Tokens, models and call counts are complete. + +## Short-lived processes + +The exporter batches spans and sends them every 5 seconds. A script, CLI, test run or serverless function that exits sooner loses the batch. In a script, shut Mastra down before exiting; that flushes every exporter: + +```ts +try { + await main() +} finally { + await mastra.shutdown() +} +``` + +In a serverless handler, flush at the end of each request instead, and keep the instance for the next one: + +```ts +export async function POST(req: Request) { + const { chatId, userId, text } = await req.json() + try { + const result = await mastra.getAgent("supportAgent").generate(text, { + memory: { thread: chatId, resource: userId }, + }) + return Response.json({ text: result.text }) + } finally { + await mastra.observability.flush() + } +} +``` + +For a streamed response, flush after the stream has finished, for example in Next.js `after()` or Cloudflare's `ctx.waitUntil()`. Flushing when the handler returns the stream is too early: the spans end when the last chunk is sent. + +## Check that it works + +Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive, longer if the process is still running and hasn't hit the 5-second export interval. You should see: + +- one session per conversation, with your thread id as the session id, and framework **Mastra**; +- one turn per `generate()` or `stream()` call, labeled with the user's message, and a transcript with the prompts, replies and tool calls; +- `invoke_agent ` spans for runs, `chat ` spans for model calls, and `execute_tool ` spans for tool calls, with `agent_step` and `model_generation` spans in between; +- `invoke_workflow ` as the turn for workflow runs; +- a lane per sub-agent, named after its `name`; +- token counts on every model call, including streamed ones; +- failed tool calls marked as failed, with the thrown message; +- cost shown as unpriced. + +To see what the exporter does locally, set `logLevel: "debug"` on `OtelExporter`. It prints every queued span and a line per batch, `Export completed: N spans sent successfully` or `Export FAILED` with the reason. + +## Troubleshooting + +- **Nothing arrives and there is no error.** `observability` is a plain object instead of `new Observability(...)`, or the agent isn't registered on the `Mastra` instance. Check the startup log for a no-op observability warning. +- **`Traces http/json exporter is not installed` or `http/protobuf exporter is not installed` at startup.** The protocol's exporter package is missing. Set `protocol: "http/protobuf"` and install `@opentelemetry/exporter-trace-otlp-proto`. +- **`Custom configuration requires endpoint. Tracing will be disabled.`** The exporter got no `endpoint`. It doesn't read `OTEL_EXPORTER_OTLP_ENDPOINT`; pass the endpoint in code. +- **`Export FAILED` with 401 or 403 in debug output.** The `Authorization` header is missing or the key is wrong. The header value is `Bearer ` followed by the ingest key. +- **Every message is its own session.** The call has no `memory: { thread, resource }`, or the thread id changes per request. Pass the conversation's id on every call; for workflows, use `tracingOptions.metadata.threadId`. +- **A supervisor run shows several turns, or lands in another session.** Delegations got their own thread ids. Add the `conversationIdFromRoot` processor. +- **Nothing arrives from a script or serverless function.** The process ended before the batch was exported. Call `mastra.shutdown()` in a script, or `mastra.observability.flush()` at the end of each request. +- **Spans arrive, but no model calls, models or tokens.** `@mastra/observability` and `@mastra/otel-exporter` are from different releases. Update `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` together. +- **Thousands of tiny `model_chunk` spans per streamed reply.** Add `excludeSpanTypes: [SpanType.MODEL_CHUNK]`. +- **Hundreds of `workflow_step` spans per turn.** `includeInternalSpans: true` is set. Mastra runs its agent loop as internal workflows; the flag exports all of them, around 5 times the spans and 25 times the bytes per turn. Leave it off. +- **A workflow's agents show up as separate traces.** The step called `agent.generate()` without `tracingContext`. Pass it from the step's `execute` arguments. +- **The latest user message is missing from a long conversation's transcript.** The message list hit `maxArrayLength` (50). Raise it in `serializationOptions`. +- **A failed tool shows as successful.** The tool returned an error value instead of throwing. Throw an `Error`. +- **Spans show up twice.** Two exporters send the same spans to Maple, for example `OtelExporter` plus the OpenTelemetry bridge, or `OtelExporter` plus an OpenLLMetry or OpenInference instrumentation of the AI SDK underneath. Keep `OtelExporter` and remove the other. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [All agent tracing guides](/docs/agent-tracing) +- [Mastra: OpenTelemetry exporter](https://mastra.ai/docs/observability/tracing/exporters/otel) +- [Mastra: tracing overview](https://mastra.ai/docs/observability/tracing/overview) +- [Trace Vercel AI SDK agents](/docs/agent-tracing/vercel-ai-sdk) diff --git a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md new file mode 100644 index 0000000000..43eea0cf07 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md @@ -0,0 +1,394 @@ +--- +title: "Trace Microsoft Agent Framework and Semantic Kernel agents with OpenTelemetry" +description: "Send Microsoft Agent Framework and Semantic Kernel traces to Maple so each conversation becomes one agent session with its transcript, tool calls, tokens and failures, in Python and .NET." +group: "AI Agents" +order: 30 +navLabel: "Microsoft Agent Framework" +icon: "dotnet" +--- + +Microsoft Agent Framework (MAF) ships its own OpenTelemetry instrumentation in Python and .NET. Every `agent.run()` produces an `invoke_agent` span, a `chat` span per model call and an `execute_tool` span per tool call, using the current GenAI semantic conventions, with tokens (including cache and reasoning buckets) and, once you switch content capture on, the full prompts and replies as span attributes. Workflows add `workflow.run`, `executor.process` and `edge_group.process` spans. + +What it doesn't emit is a conversation id. MAF only sets `gen_ai.conversation.id` when the model provider stores the conversation server side, so with Chat Completions, OpenRouter or any local chat history, every turn arrives as its own one-turn session. This guide adds the id with a 15-line span processor. It covers `agent-framework` 1.19 (Python), `Microsoft.Agents.AI` 1.22 (.NET), and Semantic Kernel 1.44 (Python), the framework MAF replaces, which has its own section below. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-microsoft-agent-framework](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-microsoft-agent-framework) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Microsoft Agent Framework in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-microsoft-agent-framework -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Export Agent Framework traces to Maple (Python) + +MAF depends on the OpenTelemetry API and SDK but installs no exporter. Add the HTTP one: + +```bash +pip install "agent-framework-core>=1.19.0" "agent-framework-openai>=1.14.4" opentelemetry-exporter-otlp-proto-http +``` + +The `agent-framework` meta-package works too; it pulls in every connector (Azure, Anthropic, Bedrock and more), so install the core and the connectors you use instead. + +Call `configure_otel_providers()` once at startup, before you create agents: + +```py +# telemetry.py +import os + +from agent_framework.observability import configure_otel_providers +from opentelemetry import trace + +from maple_tracing import ConversationIdProcessor + +configure_otel_providers( + service_name="support-agent", + resource_attributes={"deployment.environment.name": "production"}, + otlp_endpoint="https://ingest.maple.dev", # EU: https://ingest.eu.maple.dev + otlp_protocol="http/protobuf", + otlp_headers={"Authorization": f"Bearer {os.environ['MAPLE_INGEST_KEY']}"}, + enable_sensitive_data=True, # prompts, replies, tool arguments and results + enable_message_events=False, # skip the duplicate copy of the content in OTLP logs +) +trace.get_tracer_provider().add_span_processor(ConversationIdProcessor()) +``` + +`ConversationIdProcessor` comes from the next section. Instrumentation itself is on by default, so this is the whole setup: `configure_otel_providers()` builds the tracer, meter and logger providers and an OTLP exporter for each. It appends `/v1/traces`, `/v1/metrics` and `/v1/logs` to the endpoint for you. + +**Set the protocol explicitly.** MAF defaults to gRPC, unlike the OpenTelemetry spec's `http/protobuf`. Without `otlp_protocol` (or `OTEL_EXPORTER_OTLP_PROTOCOL`), the exporter fails with an import error if the gRPC package isn't installed, or silently retries a gRPC connection that never succeeds if it is. + +The same configuration through environment variables, with a bare `configure_otel_providers()` call: + +```bash +export OTEL_SERVICE_NAME="support-agent" +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export ENABLE_SENSITIVE_DATA="true" +export ENABLE_MESSAGE_EVENTS="false" +``` + +If your app already configures OpenTelemetry (Azure Monitor, Logfire, your own `TracerProvider`), don't call `configure_otel_providers()`. Add Maple's exporter and `ConversationIdProcessor` to the provider you have, and call `enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)` from `agent_framework.observability`. + +## Export Agent Framework traces to Maple (.NET) + +```bash +dotnet add package Microsoft.Agents.AI --version 1.22.0 +dotnet add package Microsoft.Agents.AI.OpenAI --version 1.22.0 +dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol --version 1.19.1 +``` + +Instrument the agent with `UseOpenTelemetry()`. In 1.22 it also instruments the agent's chat client, so you get the `chat` and `execute_tool` spans without wrapping the `IChatClient` yourself: + +```csharp +using System.ClientModel; +using Microsoft.Agents.AI; +using Microsoft.Extensions.AI; +using OpenAI; +using OpenTelemetry; +using OpenTelemetry.Exporter; +using OpenTelemetry.Resources; +using OpenTelemetry.Trace; + +using var tracerProvider = Sdk.CreateTracerProviderBuilder() + .ConfigureResource(r => r.AddService("support-agent")) + .AddSource("*Microsoft.Agents.AI*") // agent, chat and workflow spans + .AddSource("*Microsoft.Extensions.AI") // chat clients you instrument yourself + .AddProcessor(new ConversationIdProcessor()) + .AddOtlpExporter(o => + { + o.Endpoint = new Uri("https://ingest.maple.dev/v1/traces"); // EU: ingest.eu.maple.dev + o.Protocol = OtlpExportProtocol.HttpProtobuf; + o.Headers = "Authorization=Bearer YOUR_INGEST_KEY"; + }) + .Build(); + +var openAi = new OpenAIClient( + new ApiKeyCredential(Environment.GetEnvironmentVariable("OPENAI_API_KEY")!)); + +AIAgent agent = openAi.GetChatClient("gpt-4o-mini").AsIChatClient() + .AsAIAgent( + instructions: "You are a helpful assistant.", + name: "support_agent", + tools: [AIFunctionFactory.Create(GetWeather, name: "get_weather")]) + .AsBuilder() + .UseOpenTelemetry(configure: a => a.EnableSensitiveData = true) + .Build(); +``` + +Pass each tool a `name`. `AIFunctionFactory.Create` otherwise uses the method name, and a local function in a top-level `Program.cs` compiles to something like `_Main_g_GetWeather_0_3`, which is what Maple then shows as the tool. + +The wildcards matter. With no `sourceName`, the agent emits on `Experimental.Microsoft.Agents.AI`, and `Microsoft.Extensions.AI` chat clients on `Experimental.Microsoft.Extensions.AI`. A plain `AddSource("Microsoft.Agents.AI.*")`, as some samples use, matches neither, and you get zero spans. Workflows emit on `Microsoft.Agents.AI.Workflows` once you call `.WithOpenTelemetry()` on the `WorkflowBuilder`. + +When you set `Endpoint` in code, it must include `/v1/traces`, and the .NET exporter also defaults to gRPC, so keep the `HttpProtobuf` line. .NET agent spans show up under the **Unidentified** framework in Maple (the Python instrumentation scope is what identifies MAF); sessions, transcripts, tokens and tools work the same. The agent span is named `invoke_agent support_agent()` in .NET, while `gen_ai.agent.name` stays `support_agent`. + +## Group every turn of a conversation into one session + +Maple groups traces into a session by `gen_ai.conversation.id`. On an ordinary `agent.run()`, MAF writes that attribute only from the provider's own conversation id (`AgentSession.service_session_id`), which exists when the Responses API stores history (`store=True`) or a Foundry agent owns the thread. With Chat Completions, OpenRouter, Ollama, or anything that keeps history in an `AgentSession` in your process, it's never set. The session's own `session_id` isn't exported either. + +Skip this and every `agent.run()` is its own session in Maple, named after its trace id: a ten-message chat becomes ten one-turn sessions, and a human-in-the-loop approval splits a turn in two. + +Two things look like fixes and aren't: + +- **`gen_ai.agent.id`** is a random id per `Agent` object. A server that builds one agent at startup stamps the same value on every user's conversation. +- **The `conversation_id` chat option** (`Agent(..., conversation_id=...)` or `options={"conversation_id": ...}`) does reach the `invoke_agent` span, but it's a provider setting, not a label. It turns off the in-memory history MAF adds to a session, and the Responses API client sends it as `previous_response_id`. + +Instead, add this processor. It stamps the id on every span started inside a `conversation()` block, including the `chat`, `execute_tool` and workflow spans, and it follows `asyncio` tasks the block creates: + +```py +# maple_tracing.py +from contextlib import contextmanager +from contextvars import ContextVar + +from opentelemetry.sdk.trace import SpanProcessor + +_conversation_id: ContextVar[str | None] = ContextVar("conversation_id", default=None) + + +class ConversationIdProcessor(SpanProcessor): + """Puts gen_ai.conversation.id on every span started inside `conversation()`.""" + + def on_start(self, span, parent_context=None): + if (conversation_id := _conversation_id.get()) is not None: + span.set_attribute("gen_ai.conversation.id", conversation_id) + + +@contextmanager +def conversation(conversation_id: str): + token = _conversation_id.set(conversation_id) + try: + yield + finally: + _conversation_id.reset(token) +``` + +Wrap each request in it. The session's `session_id` is a good id: it's stable for the life of the conversation and survives `to_dict()`/`from_dict()` if you persist sessions. + +```py +from agent_framework import Agent, AgentSession +from agent_framework.openai import OpenAIChatCompletionClient + +from maple_tracing import conversation + +agent = Agent( + OpenAIChatCompletionClient(model="gpt-4o-mini"), + "You are a helpful assistant.", + name="support_agent", + tools=[get_weather, calculate], +) + + +async def handle_message(session: AgentSession, text: str) -> str: + with conversation(session.session_id): + response = await agent.run(text, session=session) + return response.text +``` + +Create the session with your own chat id if you have one: `agent.create_session(session_id=chat_id)`. For streaming, keep the whole `async for` inside the block, because the `chat` span starts on the first pull: + +```py +with conversation(session.session_id): + async for update in agent.run(text, session=session, stream=True): + print(update.text, end="") +``` + +In .NET, the same idea is an `Activity` processor with an `AsyncLocal`: + +```csharp +using System.Diagnostics; +using OpenTelemetry; + +sealed class ConversationIdProcessor : BaseProcessor +{ + public static readonly AsyncLocal Current = new(); + + public override void OnStart(Activity activity) + { + if (Current.Value is { } id) activity.SetTag("gen_ai.conversation.id", id); + } +} +``` + +```csharp +AgentSession session = await agent.CreateSessionAsync(); +ConversationIdProcessor.Current.Value = chatId; // once per request, before RunAsync +var response = await agent.RunAsync(userMessage, session); +``` + +## Record prompts, responses and tool calls + +Content is off by default in both languages. Without it, Maple shows the model, tokens and timing of each call, but the transcript is empty and tool calls have no arguments or results. + +- **Python:** `enable_sensitive_data=True` or `ENABLE_SENSITIVE_DATA=true`. Content lands on the spans as `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, which is the shape Maple reads. +- **.NET:** `EnableSensitiveData = true` in the `UseOpenTelemetry` callback, or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=true`. + +Two Python settings change where content goes: + +- `OTEL_SEMCONV_STABILITY_OPT_IN`: MAF uses the current GenAI conventions only while this variable is unset or includes `gen_ai_latest_experimental`. If another library in your app sets it to something like `http`, MAF drops back to the v1.36 conventions: the message attributes and tool arguments disappear from the spans and the content goes to log events, which Maple doesn't read for sessions. Use `OTEL_SEMCONV_STABILITY_OPT_IN=http,gen_ai_latest_experimental`. +- `ENABLE_MESSAGE_EVENTS` defaults to `true` and sends a second copy of every message as OTLP log records. Maple builds the transcript from span attributes, so turn it off unless another tool reads those logs. + +With content on, everything the user types and every tool result is stored in Maple. To keep a service's content out, leave sensitive data off there and keep the spans; sessions, tokens, tool names and failures still work. Tool definitions (`gen_ai.tool.definitions`, the JSON schema of each tool) are sent on every `invoke_agent` span even with sensitive data off. + +## Tools, errors and sub-agents + +Each function tool call is an `execute_tool ` span with `gen_ai.tool.name`, `gen_ai.tool.call.id`, the arguments and the result. When a tool raises, MAF marks that span ERROR with `error.type` (the exception class) and the message in the status, then hands the model an error string as the tool result so the run continues. Maple counts the failure on the tool, and the agent and `chat` spans above it stay green, which is correct: the turn itself succeeded. .NET does the same (`error.type=System.InvalidOperationException`, for example). + +Python MAF also logs each tool failure as an ERROR and a WARN record. `configure_otel_providers()` exports those to Maple as logs, next to the span. + +Give every agent a distinct `name`. Maple opens a sub-agent lane when an `invoke_agent` span's `gen_ai.agent.name` differs from its parent's, and an agent with no name gets its UUID as the name. + +The simplest multi-agent pattern is agents as tools. The orchestrator's `execute_tool weather_worker` span has the worker's `invoke_agent weather_worker` span as its child, which Maple shows as a delegation with the task and the answer: + +```py +weather_worker = Agent(client, "Report current weather.", name="weather_worker", tools=[get_weather]) +budget_worker = Agent(client, "Estimate trip costs.", name="budget_worker", tools=[calculate]) + +orchestrator = Agent( + client, + "Delegate each part of the briefing to the right worker, then summarize.", + name="orchestrator", + tools=[weather_worker.as_tool(), budget_worker.as_tool()], +) + +with conversation(session.session_id): + result = await orchestrator.run(brief, session=session) +``` + +Workflows (`WorkflowBuilder`, or the `SequentialBuilder`, `ConcurrentBuilder`, `HandoffBuilder` and `MagenticBuilder` orchestrations) trace each executor as `executor.process ` with the agent's `invoke_agent` span inside it. Fan-in is recorded as span links, not parent-child, so parallel workers appear as siblings under `workflow.run`. The executors run as `asyncio` tasks and inherit the conversation id from the block. + +Build the workflow inside the `conversation()` block too. `WorkflowBuilder.build()` emits its own one-span `workflow.build` trace. Inside the block it joins the session as a short extra trace with no model calls; built once at import time, it becomes a one-span session of its own. + +Tools that need approval (`@tool(approval_mode="always_require")`) end the run with a pending request. The resume is a new `agent.run()` and a new trace; with the processor in place, it joins the same session. A rejected call emits no `execute_tool` span at all; the rejection only appears in the next `chat` span's input messages. + +## Tokens and cost + +Every `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `cache_read.input_tokens`, `cache_creation.input_tokens` and `reasoning.output_tokens` when the provider returns them. Streaming works without extra settings: the OpenAI chat completions client requests `include_usage` itself. + +The `invoke_agent` span repeats the total of that agent's own `chat` calls. A sub-agent called as a tool reports its calls on its own `invoke_agent` span, not the orchestrator's. Maple subtracts the child `chat` spans, so the session total counts each call once. + +In .NET, streamed `chat` spans also carry `gen_ai.response.time_to_first_chunk`, which Maple shows as time to first token. Python MAF doesn't record it. + +MAF emits no cost attribute, and Maple doesn't price tokens, so sessions show as unpriced. `gen_ai.provider.name` on `chat` spans is the client type (`openai` for the OpenAI client, even when it points at OpenRouter or Ollama), and `server.address` holds the real base URL. + +## Flush before a short-lived process exits + +`configure_otel_providers()` uses batch processors, and MAF has no flush helper. A script, CLI, notebook cell or serverless handler that exits without flushing loses its last turns. Shut the three providers down in a `finally`: + +```py +from opentelemetry import _logs, metrics, trace + + +def shutdown_telemetry() -> None: + for provider in (trace.get_tracer_provider(), metrics.get_meter_provider(), _logs.get_logger_provider()): + provider.shutdown() # exports whatever is still buffered + + +try: + asyncio.run(main()) +finally: + shutdown_telemetry() +``` + +For a long-running server, don't shut down per request. In a serverless handler that is frozen between invocations, call `trace.get_tracer_provider().force_flush()` before returning. + +In .NET, `using var tracerProvider` flushes on dispose at the end of `Main`. In a hosted app, `AddOpenTelemetry()` flushes on graceful shutdown. + +## Semantic Kernel + +Semantic Kernel (SK) 1.44 is in maintenance while Microsoft moves agent work to MAF, and its telemetry differs in four ways. + +**It's gated by environment variables read at import time.** Set them before the first `import semantic_kernel`, or SK emits no GenAI spans and logs no warning. SK brings no exporter or provider, so set up your own: + +```bash +pip install "semantic-kernel>=1.44.1" opentelemetry-sdk opentelemetry-exporter-otlp-proto-http +``` + +`semantic-kernel` 1.44 depends on a pre-release `azure-ai-agents`, so `uv` needs `--prerelease=allow`; pip resolves it as is. + +```py +# telemetry.py: import this before anything that imports semantic_kernel +import os + +os.environ["SEMANTICKERNEL_EXPERIMENTAL_GENAI_ENABLE_OTEL_DIAGNOSTICS"] = "true" +os.environ["SEMANTICKERNEL_EXPERIMENTAL_GENAI_ENABLE_OTEL_DIAGNOSTICS_SENSITIVE"] = "true" + +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +from maple_tracing import ConversationIdProcessor + +provider = TracerProvider(resource=Resource.create({"service.name": "support-agent"})) +provider.add_span_processor(ConversationIdProcessor()) +provider.add_span_processor( + BatchSpanProcessor( + OTLPSpanExporter( + endpoint="https://ingest.maple.dev/v1/traces", # EU: https://ingest.eu.maple.dev/v1/traces + headers={"Authorization": f"Bearer {os.environ['MAPLE_INGEST_KEY']}"}, + ) + ) +) +trace.set_tracer_provider(provider) +``` + +**Model-call content goes to Python logging, not spans.** The `chat ` spans carry model, tokens and finish reason; the prompts and replies are log records Maple doesn't read. The transcript comes from the agent instead: `ChatCompletionAgent` puts the messages you pass in and the reply on its `invoke_agent` span as `gen_ai.input.messages` and `gen_ai.output.messages`. Pass the messages positionally, `await agent.get_response(text, thread=thread)`. SK reads them from the second positional argument, and `get_response(messages=text, ...)` records an empty input. Code that calls the kernel directly (`kernel.invoke_prompt`, a chat service without an agent) has no transcript in Maple. + +**There's no conversation id either.** `ChatHistoryAgentThread.id` isn't exported, so use the same `ConversationIdProcessor` and wrap each request in `with conversation(thread_id):`. + +**The agent runtime splits orchestrations across traces.** `ConcurrentOrchestration`, `SequentialOrchestration` and the other orchestrations run on `InProcessRuntime`, which starts each message on a new trace. One run became 10 traces in our test. Open your own span around the run and start the runtime inside the `conversation()` block, so the runtime's tasks inherit both: + +```py +tracer = trace.get_tracer("support-agent") + +with conversation(conversation_id), tracer.start_as_current_span("briefing"): + runtime = InProcessRuntime() + runtime.start() + result = await orchestration.invoke(task=brief, runtime=runtime) + output = await result.get() + await runtime.stop_when_idle() +``` + +Smaller differences: tool spans are named `execute_tool -`, failing tools get ERROR status and `error.type` but no result attribute, finish reasons are Python enum names (`FinishReason.STOP`) so Maple's reply-length and refusal checks can't read them, and a `temperature` of `0` is left off the span. Flush with `provider.shutdown()` as above. + +In .NET, SK reads the same variables, or the `AppContext` switches `Microsoft.SemanticKernel.Experimental.GenAI.EnableOTelDiagnostics` and `...EnableOTelDiagnosticsSensitive`. Add `AddSource("Microsoft.SemanticKernel*")` to the tracer provider. SK .NET spans show up under the **Unidentified** framework in Maple. + +## Check that it works + +Run one conversation of two or three turns, one of which calls a tool. Sessions appear in **Agent Sessions** within about a minute. You should see: + +- **One session per conversation**, labeled **Microsoft Agent Framework** (or **Semantic Kernel**) for Python, whose id is your conversation id, not `trace:…`. A second conversation is a second session. +- **One turn per `agent.run()`**, each rooted at `invoke_agent support_agent` with `chat gpt-4o-mini` and `execute_tool get_weather` spans under it. +- **A transcript** with the system instructions, your messages and the replies, labeled by the first line of each user message. +- **Tool calls** with arguments and results, and failed tools counted under **Tool errors**. +- **Tokens** on every model call, including streamed ones. Cost shows as unpriced. +- For multi-agent runs, a lane per agent name. + +## Troubleshooting + +- **Nothing arrives, no error.** The protocol is still gRPC. Set `otlp_protocol="http/protobuf"` or `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf`. If you set `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`, it's used as is and needs the `/v1/traces` path. +- **`ImportError: opentelemetry-exporter-otlp-proto-grpc is required`.** Same cause: the protocol defaulted to gRPC. +- **Every turn is its own session.** No `gen_ai.conversation.id`. Register `ConversationIdProcessor` and wrap the call, including the whole `async for` when streaming, in `conversation()`. +- **Every user shares one session.** The id is a process-wide constant: `gen_ai.agent.id`, a module-level string, or `conversation()` called once at startup. Use a per-conversation id. +- **An extra session with a single `workflow.build` span.** The workflow was built outside `conversation()`. Build it per request inside the block. +- **Spans but an empty transcript (Python).** Sensitive data is off, or `OTEL_SEMCONV_STABILITY_OPT_IN` is set without `gen_ai_latest_experimental`. +- **The agent forgets earlier turns after adding tracing.** You passed `conversation_id` as a chat option, which disables the in-memory history. Remove it and use the processor. +- **Every message part is one character.** `Message("user", text)` iterates a bare string; pass a list: `Message("user", [text])`. +- **OpenRouter rejects the second turn with a `previous_response_id` error.** `OpenAIChatClient` is the Responses API client. Use `OpenAIChatCompletionClient` for OpenRouter and other Chat Completions endpoints. +- **.NET: no spans at all.** `AddSource` doesn't match the default `Experimental.` prefix. Use `AddSource("*Microsoft.Agents.AI*")`. +- **Semantic Kernel: `AutoFunctionInvocationLoop` spans but no `chat` or `invoke_agent`.** The `SEMANTICKERNEL_EXPERIMENTAL_GENAI_*` variables were set after `semantic_kernel` was imported. +- **Semantic Kernel: an orchestration shows as many small sessions or traces.** Wrap the run in your own span and `conversation()` and start the runtime inside it. + +## Related + +- [Agent Sessions](/docs/agent-sessions/overview): what a session, turn, model call and tool call are in Maple. +- [Trace your AI agent](/docs/agent-tracing): guides for other frameworks. +- [Python instrumentation](/docs/guides/instrumentation-python) and [.NET instrumentation](/docs/guides/instrumentation-csharp): tracing the rest of the service. +- [Agent Framework observability](https://learn.microsoft.com/en-us/agent-framework/agents/observability): Microsoft's reference for the settings above. +- [Semantic Kernel telemetry](https://learn.microsoft.com/en-us/semantic-kernel/concepts/enterprise-readiness/observability/): the SK diagnostics switches. diff --git a/apps/landing/src/content/docs/agent-tracing/openai-agents.md b/apps/landing/src/content/docs/agent-tracing/openai-agents.md new file mode 100644 index 0000000000..0fa2fbe359 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/openai-agents.md @@ -0,0 +1,318 @@ +--- +title: "Trace OpenAI Agents SDK runs with OpenTelemetry" +description: "Send OpenAI Agents SDK runs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools, handoffs and agents as tools." +group: "AI Agents" +order: 12 +navLabel: "OpenAI Agents SDK" +icon: "openai" +--- + +The OpenAI Agents SDK traces every run out of the box, but not with OpenTelemetry. Its tracing pipeline builds its own traces and spans (agent, generation, function, handoff, guardrail) and uploads them to the OpenAI dashboard. To get them into Maple you swap that uploader for OpenInference's `openinference-instrumentation-openai-agents`, which turns each SDK span into an OpenTelemetry span as it ends. + +The SDK's own way to tie the traces of one chat together, `group_id`, never reaches OpenTelemetry. The bridge drops it, so a ten-message conversation shows up in Maple as ten one-turn sessions until you wrap each run in OpenInference's `using_session`. A few more defaults need changing: the bridge writes OpenInference attributes that Maple's session page doesn't decode for this framework, and streamed calls to any provider other than OpenAI lose their tokens and their reply. This guide covers `openai-agents` 0.22 with `openinference-instrumentation-openai-agents` 2.5 on Python 3.10 to 3.14. TypeScript (`@openai/agents`) works with less detail; see [TypeScript](#typescript-openaiagents). + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-openai-agents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for OpenAI Agents SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-openai-agents -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install the bridge and export to Maple + +```bash +pip install "openai-agents>=0.22" "openinference-instrumentation-openai-agents>=2.5" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. `OTLPSpanExporter()` with no arguments appends `/v1/traces` to that base URL and sends `http/protobuf`. If you pass `endpoint=` in code instead, it's used as is, so it has to end in `/v1/traces`. + +Then add a `tracing.py` and import it at the top of your entry point: + +```py +# tracing.py +from agents import set_trace_processors +from agents.tracing import TracingProcessor +from agents.tracing.span_data import FunctionSpanData, GenerationSpanData, HandoffSpanData +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.openai_agents import OpenAIAgentsInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +def _chat_message(response: dict) -> dict: + """The assistant message inside a Responses-shaped dict, as a Chat Completions message.""" + text, calls = "", [] + for item in response.get("output") or []: + if item.get("type") == "message": + text += "".join(c.get("text", "") for c in item.get("content") or [] if c.get("type") == "output_text") + elif item.get("type") == "function_call": + calls.append({"id": item["call_id"], "type": "function", + "function": {"name": item["name"], "arguments": item["arguments"]}}) + return {"role": "assistant", "content": text or None, "tool_calls": calls or None} + + +class MapleSpanFixes(TracingProcessor): + """Fills three gaps in what OpenInference exports. Must run before the OpenInference processor.""" + + def on_span_end(self, span): + data = span.span_data + current = trace.get_current_span() # the matching OpenTelemetry span, still open here + if isinstance(data, FunctionSpanData) and data.input and getattr(current, "name", None) == data.name: + # Real arguments; OpenInference would copy the tool's JSON schema. + current.set_attribute("gen_ai.tool.call.arguments", data.input) + elif isinstance(data, HandoffSpanData) and data.to_agent: + # Handoff spans carry no tool name. + current.set_attribute("gen_ai.tool.name", f"transfer_to_{data.to_agent}") + elif isinstance(data, GenerationSpanData) and data.output and data.output[0].get("object") == "response": + # Streamed Chat Completions calls record a Responses object OpenInference can't read. + data.output = [_chat_message(data.output[0])] + + def on_trace_start(self, t): pass + def on_trace_end(self, t): pass + def on_span_start(self, span): pass + def shutdown(self): pass + def force_flush(self): pass + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +# Replaces the SDK's default processor (which uploads to OpenAI) with MapleSpanFixes, +# then appends the OpenInference processor after it. +set_trace_processors([MapleSpanFixes()]) +OpenAIAgentsInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), + exclusive_processor=False, +) +``` + +What each part does: + +- **`enable_genai_semconv=True`** makes OpenInference write the OpenTelemetry GenAI attributes (`gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.agent.name`, `gen_ai.tool.*`) next to its own `llm.*` attributes when each span ends. Without it, the Agent Sessions list shows token counts but the session page has no transcript. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, but only if it's set before `TraceConfig` is built; passing it in code avoids that trap. +- **`MapleSpanFixes`** fills three gaps in what the bridge exports. The GenAI dual-write fills `gen_ai.tool.call.arguments` from `tool.parameters`, which is the tool's JSON schema, so every tool call would show `{"properties": {"city": ...}}` instead of `{"city": "Berlin"}`; the dual-write never overwrites a key that's already set, so the processor sets the real arguments first. Handoff spans get no tool name, so it names them `transfer_to_`, the tool the model called. And a streamed Chat Completions call records its reply as a Responses API object the bridge can't parse, so without the fix the streamed turn has no reply in the transcript; the processor rewrites it into a Chat Completions message before the bridge reads it. +- **`set_trace_processors([...])` plus `exclusive_processor=False`** puts `MapleSpanFixes` ahead of the OpenInference processor, which the ordering requires. It also removes the SDK's default processor, so nothing is uploaded to the OpenAI dashboard and you don't need an OpenAI key for tracing. To keep that upload too, pass `set_trace_processors([MapleSpanFixes(), default_processor()])` with `default_processor` from `agents.tracing.processors`. + +The bridge hooks into the SDK's processor list rather than patching imports, so import order only matters in one way: `tracing.py` has to run before the first `Runner.run`. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), don't create a second one. Add the OTLP exporter to the existing provider and pass that provider to `instrument()`. + +Don't call `set_tracing_disabled(True)`, set `OPENAI_AGENTS_DISABLE_TRACING=1` or pass `RunConfig(tracing_disabled=True)` to stop the upload to OpenAI. Those switch off the SDK's whole tracing pipeline, which is where the bridge gets its data, and you get zero spans. + +### Models from other providers + +With an `OPENAI_API_KEY`, agents use OpenAI's Responses API and need nothing extra. To use another provider through an OpenAI-compatible endpoint (OpenRouter, LiteLLM, vLLM, Ollama), give the agent a `OpenAIChatCompletionsModel` and turn on `include_usage`: + +```py +import os + +from agents import Agent, ModelSettings, OpenAIChatCompletionsModel +from openai import AsyncOpenAI + +client = AsyncOpenAI(base_url="https://openrouter.ai/api/v1", api_key=os.environ["OPENROUTER_API_KEY"]) + +agent = Agent( + name="assistant", + instructions="You are a helpful assistant. Be brief.", + model=OpenAIChatCompletionsModel(model="openai/gpt-4o-mini", openai_client=client), + model_settings=ModelSettings(include_usage=True), +) +``` + +The SDK only asks for token usage on streamed Chat Completions calls when the client points at `api.openai.com`. For any other base URL it sends no `stream_options`, the provider returns no usage chunk, and every `Runner.run_streamed` turn has zero tokens. `include_usage=True` sends `stream_options={"include_usage": true}`; non-streamed calls report usage either way. + +## Group a conversation into one session + +Each `Runner.run` is its own trace. The SDK has two ids that look like the answer, and neither reaches OpenTelemetry: `RunConfig(group_id=...)` and the `session_id` of a `SQLiteSession` stay inside the SDK. Maple groups this framework's traces by the `session.id` attribute, and the bridge only sets it inside OpenInference's `using_session`: + +```py +from agents import RunConfig, Runner, SQLiteSession +from openinference.instrumentation import using_session + + +async def handle_message(conversation_id: str, text: str) -> str: + with using_session(conversation_id): + result = await Runner.run( + agent, + text, + session=SQLiteSession(conversation_id, "chats.db"), + run_config=RunConfig(workflow_name="support workflow"), + ) + return result.final_output +``` + +Use your app's own conversation id, the one it already stores the chat under, and give the SDK session the same one. A new UUID per request gives you one session per message again, and a constant gives every user one shared session. + +`using_session` stores the id in a Python contextvar. The bridge copies it onto every span it creates inside the block, including tool calls that run concurrently and agents called as tools, because asyncio tasks inherit the context. For streaming, call `Runner.run_streamed` inside the block; the SDK starts its background task there, so iterating `stream_events()` afterwards is fine: + +```py +with using_session(conversation_id): + result = Runner.run_streamed(agent, text, session=SQLiteSession(conversation_id, "chats.db")) + async for event in result.stream_events(): + ... +``` + +`workflow_name` names the trace's root span. The default is `Agent workflow`, which is the same for every run in your app. Keep `chat`, `completion` and `tool` out of the name. The bridge also gives each run a bookkeeping span with that name and no operation, and Maple classifies such spans by name: `support chat` would add one phantom LLM call per run, and `tool` in the name a phantom tool call. A name ending in `workflow` or `agent` is safe. + +If you skip `using_session`, every run shows up in **Agent Sessions** as its own one-turn session named after its trace id. Setting `gen_ai.conversation.id` yourself doesn't help either: Maple reads `session.id` first for this framework, and the dual-write already copies it to `gen_ai.conversation.id`. + +## Record prompts, responses and tool calls + +Content capture is on by default, in two places. The SDK records model inputs and outputs and tool arguments and results on its spans (`trace_include_sensitive_data`, default `True`), and OpenInference copies them onto the OpenTelemetry span. With the dual-write on, each model span carries the content three times: as flattened `llm.input_messages.N.*` keys, as the `input.value` JSON, and as `gen_ai.input.messages`, which is the one Maple renders. Every model span also repeats the whole conversation so far. Maple has no per-attribute limit and accepts requests up to 20 MiB, so this costs bandwidth rather than data. + +To keep prompts and outputs out of your traces, turn capture off at the source: + +```bash +export OPENAI_AGENTS_TRACE_INCLUDE_SENSITIVE_DATA=false +``` + +or per run with `RunConfig(trace_include_sensitive_data=False)`. The session still shows turns, models, tool names, tokens and failures, with an empty transcript. Tool error messages are replaced by `Tool execution failed. Error details are redacted.` + +OpenInference's own switches (`TraceConfig(hide_inputs=True, hide_outputs=True)` or `OPENINFERENCE_HIDE_INPUTS` / `OPENINFERENCE_HIDE_OUTPUTS`) redact at the bridge instead, and also empty the transcript, because the GenAI messages are derived from the attributes they hide. For partial redaction, such as masking emails, drop or rewrite attributes in an OpenTelemetry Collector with the `transform` or `redaction` processor. + +## Tools, errors, handoffs and agents as tools + +Each function tool call is a span named after the tool, with `gen_ai.operation.name` `execute_tool`, `gen_ai.tool.name`, the tool's description, its arguments (with `MapleSpanFixes`) and its result in `gen_ai.tool.call.result`. + +A tool that raises is marked failed without extra code. The SDK catches the exception, sends the model `An error occurred while running the tool. Please try again. Error: ...` as the tool result, and records the error on its span. The bridge turns that into status `ERROR` with a message like `Error running tool (non-fatal): {'tool_name': 'fetch_transport_data', 'error': 'transport data service unavailable (503)'}`, and Maple counts the call as failed on the session and on the tool's page. A tool that returns an error string instead of raising counts as a success. + +Every agent the run enters gets a span named after it, with `gen_ai.agent.name`, and Maple opens a lane for each agent whose name differs from its caller's. Give every `Agent` a distinct `name`. The spans in between are the bridge's bookkeeping: a `CHAIN` span named after the workflow per `Runner.run`, and a `turn` span per step of the agent loop. + +The two multi-agent idioms look different in a trace: + +- **Agents as tools** (`agent.as_tool(tool_name=..., tool_description=...)`): the calling agent's tool span contains the whole nested run, with the sub-agent's model and tool calls inside. The tool's arguments and result are the sub-agent's input and output. +- **Handoffs** (`handoffs=[...]`): the model calls a `transfer_to_` tool, which shows up as a `handoff to ` span with the source and target agent names as input and output. The target agent's span is a sibling of the source agent's, not a child, and the conversation continues there. Maple counts each handoff as a tool call, named `transfer_to_` by `MapleSpanFixes`. + +Parallel sub-agents work if every run is inside one SDK trace and one `using_session` block, so they share a trace id: + +```py +import asyncio + +from agents import trace + +with using_session(conversation_id), trace("amsterdam briefing"): + weather, budget = await asyncio.gather( + Runner.run(weather_worker, "Weather in Amsterdam?"), + Runner.run(budget_worker, "3-day budget for Amsterdam?"), + ) +``` + +Without the `trace()` block, each `Runner.run` starts its own trace, and the briefing shows up as several turns of the session instead of one. + +Tools that need approval (`@function_tool(needs_approval=True)`) show up twice for one approved call. The run that pauses records a tool span with arguments but no result, and the resumed `Runner.run(agent, state)` records the real execution. The resume is its own trace, so the session shows two turns for the approval: the pause and the resumed run, both labelled with the user's message. Wrap the resume in the same `using_session` id or it becomes a separate session. + +Two gaps remain. Tool spans have no `gen_ai.tool.call.id`, because the SDK's function span doesn't carry the model's call id, so Maple can't tie a tool span to the exact call in the model's reply. And on the Chat Completions path, model spans have no `gen_ai.response.id`; the Responses API path has one. + +## Tokens and cost + +Every model span carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals. On the Responses API, cached input tokens are recorded too, and reasoning tokens as `llm.token_count.completion_details.reasoning`. Streamed Chat Completions calls need `include_usage=True`, as above. + +Model spans are named `generation` on the Chat Completions path and `response` on the Responses path. On the Chat Completions path the model is the id you configured (`openai/gpt-4o-mini`). On the Responses path it's the name OpenAI returns, which is usually a dated snapshot such as `gpt-4o-mini-2024-07-18`. The provider is `openai` for both, even for an Anthropic model behind OpenRouter, because the SDK only speaks the OpenAI API. Maple uses the OpenAI token convention for it, where cached tokens are part of the input total, which is also what OpenRouter returns. + +The SDK also totals usage per run and per turn, but the bridge doesn't export those totals, so nothing is counted twice. + +Maple shows cost only when a span carries one, and neither the SDK nor the bridge records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call. + +Don't add `openinference-instrumentation-openai` next to the Agents bridge. It patches the `openai` client the SDK calls, so every model call gets a second model span under the `generation` span. The same goes for Logfire's `instrument_openai_agents()`, Langfuse's or Traceloop's Agents instrumentation, and the OpenTelemetry project's `opentelemetry-instrumentation-genai-openai-agents`: pick one. + +## Flush spans before the process exits + +`BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` registers an `atexit` handler that flushes on a normal interpreter exit, which covers most scripts and CLIs. It doesn't run when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: + +```py +from tracing import provider + +try: + asyncio.run(handle_message("conv-42", "What's the weather in Berlin?")) +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +Call `provider.shutdown()` instead when the process is about to exit. The SDK's own `flush_traces()` doesn't help here: the bridge's `force_flush()` is a no-op, and the spans wait in the OpenTelemetry batch processor. + +## Check that it works + +Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: + +- **One session** for the conversation, framework **OpenAI Agents SDK**, with one turn per `Runner.run`. Each turn's trace starts at a span named after your `workflow_name`. +- **The transcript**: the agent's instructions as the system message, your messages, the model's replies (streamed ones included) and its tool calls. +- **Model calls** named `generation` (Chat Completions) or `response` (Responses API), as many as the app made, each with a model and input and output tokens, streamed turns included. +- **Tool calls** named after your tools, such as `get_weather`, with arguments and results. A tool that raised is marked failed. +- **Agents**: one lane per agent name, such as `assistant`, plus sub-agents and handoff targets. +- **Cost**: unpriced. + +A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. + +## TypeScript (`@openai/agents`) + +The same bridge exists for the TypeScript SDK as `@arizeai/openinference-instrumentation-openai-agents`. We ran it with `@openai/agents` 0.18 and bridge 0.2.15, and it works with Maple with three differences. Maple doesn't identify it as the OpenAI Agents SDK, so the framework shows as **Unidentified**. The TypeScript bridge has no GenAI dual-write: the transcript is built from the OpenInference `input.value` and `output.value` JSON, so model replies show as the raw API response, tool calls have no separate arguments and result fields, and there's no `gen_ai.agent.name` for lanes. And the session id has to be `gen_ai.conversation.id`, because Maple doesn't read `session.id` for unidentified OpenInference spans: + +```ts +import * as agents from "@openai/agents" +import { context } from "@opentelemetry/api" +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { detectResources, envDetector } from "@opentelemetry/resources" +import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace-node" +import { setAttributes } from "@arizeai/openinference-core" +import { OpenAIAgentsInstrumentation } from "@arizeai/openinference-instrumentation-openai-agents" + +export const provider = new NodeTracerProvider({ + resource: detectResources({ detectors: [envDetector] }), // OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES + spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())], // OTEL_EXPORTER_OTLP_* variables +}) +provider.register() +new OpenAIAgentsInstrumentation({ tracerProvider: provider }).manuallyInstrument(agents) + +export function handleMessage(conversationId: string, text: string) { + const ctx = setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }) + return context.with(ctx, () => agents.run(agent, text, { session })) +} +``` + +The `envDetector` line matters: the 2.x `NodeTracerProvider` doesn't read `OTEL_SERVICE_NAME` on its own, and every span arrives as `unknown_service:node`. The same environment variables as above configure the exporter. In a script, `await provider.forceFlush()` before exiting. + +Models, tokens and tool names come through as in Python. If you need the full session page in TypeScript today, the [provider SDKs guide](/docs/agent-tracing/provider-sdks) shows how to emit the GenAI attributes yourself. + +## Troubleshooting + +- **No spans at all.** `set_tracing_disabled(True)`, `OPENAI_AGENTS_DISABLE_TRACING=1` or `RunConfig(tracing_disabled=True)` is set somewhere, or `tracing.py` ran after the first `Runner.run`. Remove the switch and import `tracing` first. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. +- **Tokens in the list, empty session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)`, or set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` before `instrument()` runs. Also check the bridge is 2.5 or later: older versions don't record agent names. +- **One session per message.** The run isn't inside `using_session(...)`, or the id changes per request. `group_id` and `SQLiteSession` ids don't reach Maple. +- **Streamed turns have zero tokens.** The model goes through a non-OpenAI base URL without `ModelSettings(include_usage=True)`. +- **Tool arguments show the tool's JSON schema.** `MapleSpanFixes` isn't registered, or it runs after the OpenInference processor. Call `set_trace_processors([MapleSpanFixes()])` before `instrument(..., exclusive_processor=False)`. +- **A streamed turn has tokens but no reply in the transcript.** Same cause: `MapleSpanFixes` isn't first. The streamed Chat Completions reply is otherwise only in `output.value` as a Responses object. +- **More LLM calls than the app made.** `workflow_name` contains `chat` or `completion`, so Maple counts each run's bookkeeping span as a model call. Rename it, for example to `support workflow`. +- **`UserError: Unknown prefix: anthropic`, or OpenRouter gets `gpt-4o-mini` without its prefix.** `Agent(model="vendor/model")` strings go through the SDK's provider prefixes. Pass `OpenAIChatCompletionsModel(model=..., openai_client=...)` instead. +- **`Tracing client error 401` in the logs.** The SDK's default processor is still uploading to OpenAI without a valid key. `set_trace_processors` in `tracing.py` removes it; make sure nothing calls `add_trace_processor` or re-adds `default_processor()` without a key. +- **Every model call appears twice.** `openinference-instrumentation-openai`, Logfire or another Agents instrumentation is also active. Keep one. +- **A multi-agent run is split over several turns.** Parallel `Runner.run` calls outside an SDK `trace()` block each start a trace. Wrap them in one `with trace("...")`. +- **Tool errors show as redacted.** `OPENAI_AGENTS_TRACE_INCLUDE_SENSITIVE_DATA=false` also redacts error details. That's the privacy switch working. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. +- [Tracing in the OpenAI Agents SDK](https://openai.github.io/openai-agents-python/tracing/): the SDK's span types, `trace()`, and the sensitive-data switch. +- [openinference-instrumentation-openai-agents](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-openai-agents): the bridge's source. +- [OpenRouter](/docs/agent-tracing/openrouter) and [LiteLLM](/docs/agent-tracing/litellm): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/openrouter.md b/apps/landing/src/content/docs/agent-tracing/openrouter.md new file mode 100644 index 0000000000..69cf5935c8 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/openrouter.md @@ -0,0 +1,277 @@ +--- +title: "Trace OpenRouter calls in Maple with Broadcast" +description: "Send every OpenRouter model call to Maple as an OpenTelemetry trace with tokens and real cost, grouped into one Agent Session per conversation, and join it to your app's own traces." +group: "AI Agents" +order: 41 +navLabel: "OpenRouter" +icon: "openrouter" +--- + +OpenRouter Broadcast exports a trace for every request that goes through your OpenRouter account. You configure it once in the OpenRouter dashboard, with no SDK and no code in your app. Each trace carries the model, the provider that served it, input, output, cached and reasoning tokens, the cost OpenRouter charged you, and the prompt and completion. Maple reads all of it and shows these traces under **Agent Sessions** as vendor **OpenRouter**. + +Out of the box, though, every model call is its own trace and its own session. Broadcast only knows what is in the request body. Unless your code sends a `session_id`, a ten-turn conversation shows up as thirty one-call "sessions". This guide covers the dashboard setup, the two request fields that fix the grouping (`session_id` and `trace`), and what Broadcast can't see: your tools and your agent structure. Code samples use the `openai` SDK (npm 7.23, PyPI 3.20), `@openrouter/ai-sdk-provider` 3.1 for the Vercel AI SDK, and `@openrouter/sdk` 1.3. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-openrouter](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openrouter) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for OpenRouter in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-openrouter -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +The agent can change your code, but not your OpenRouter dashboard. It finishes by telling you the exact values to paste into the Broadcast destination below. + +## Point Broadcast at Maple + +1. In OpenRouter, open [Settings → Observability](https://openrouter.ai/settings/observability) and turn on **Enable Broadcast**. In an organization account, only an organization admin can edit this. +2. Click the edit icon next to **OpenTelemetry Collector**. +3. Set **Endpoint** to the full traces URL. OpenRouter posts to it verbatim and does not append `/v1/traces`: + + ```text + https://ingest.maple.dev/v1/traces + ``` + + EU organizations use `https://ingest.eu.maple.dev/v1/traces`. + +4. Set **Headers** to a JSON object with your Maple ingest key: + + ```json + { "Authorization": "Bearer maple_pk_your_key" } + ``` + +5. Leave the sampling rate at 1.0 and Privacy Mode off for now (see [privacy](#prompts-completions-and-privacy-mode) below). +6. Click **Test Connection**. OpenRouter only saves the destination if the test passes. + +OpenRouter sends OTLP over HTTP with JSON encoding only. Maple's ingest accepts JSON on `/v1/traces`, so no collector is needed in between. + +Three destination settings silently decide what reaches Maple: + +- **API key filter.** If the destination lists API keys, only requests made with those keys are exported. Excluded keys always win. Leave it empty to export every key. +- **Sampling rate.** Sampling is per `session_id`, so a session is either complete or absent. Anything below 1.0 drops whole conversations. +- **Data regions.** A destination only receives requests served in its regions. If your app calls `eu.openrouter.ai`, the destination needs the Europe region (the default is global). + +Destinations can also be created through OpenRouter's [observability API](https://openrouter.ai/docs/api/api-reference/observability/create-an-observability-destination) (`type: "otel-collector"`) with a management key, if you manage OpenRouter as code. + +## Group each conversation into one session + +Maple builds a session from the `session.id` attribute on OpenRouter's spans. OpenRouter copies it from the `session_id` field of your request (up to 256 characters) or from the `x-session-id` header. The body wins if you send both. + +Send the same id on every request of a conversation: your chat thread id, conversation id or agent run id. Use a new one for each conversation. A process-wide constant merges every user into one session. + +With the `openai` SDK in TypeScript, `session_id` isn't in the types, so it needs a `@ts-expect-error`. The SDK sends unknown fields as-is: + +```ts +import OpenAI from "openai" + +const client = new OpenAI({ + baseURL: "https://openrouter.ai/api/v1", + apiKey: process.env.OPENROUTER_API_KEY, +}) + +const completion = await client.chat.completions.create({ + model: "openai/gpt-4o-mini", + messages, + // @ts-expect-error OpenRouter-only field + session_id: conversationId, +}) +``` + +Or skip the type workaround and send the header, which the SDK types allow per request: + +```ts +const completion = await client.chat.completions.create( + { model: "openai/gpt-4o-mini", messages }, + { headers: { "x-session-id": conversationId } }, +) +``` + +In Python, use `extra_body`: + +```py +import os + +from openai import OpenAI + +client = OpenAI(base_url="https://openrouter.ai/api/v1", api_key=os.environ["OPENROUTER_API_KEY"]) + +completion = client.chat.completions.create( + model="openai/gpt-4o-mini", + messages=messages, + extra_body={"session_id": conversation_id}, +) +``` + +With the Vercel AI SDK and `@openrouter/ai-sdk-provider`, everything under `providerOptions.openrouter` is merged into the request body: + +```ts +import { createOpenRouter } from "@openrouter/ai-sdk-provider" +import { streamText } from "ai" + +const openrouter = createOpenRouter({ apiKey: process.env.OPENROUTER_API_KEY }) + +const result = streamText({ + model: openrouter("openai/gpt-4o-mini"), + messages, + providerOptions: { openrouter: { session_id: conversationId } }, +}) +``` + +OpenRouter's own SDKs have typed fields: `sessionId` in `@openrouter/sdk` (`openRouter.chat.send({ model, messages, sessionId })`) and `session_id=` in the `openrouter` Python package. + +`session_id` also makes OpenRouter route a session's requests to the same provider, so prompt caches hit more often. + +If you skip it, each call lands in its own session, named `trace:` in Maple, with one turn and one model call. + +## Join Broadcast to your own traces + +By default, each OpenRouter request becomes a separate trace with an `LLM Generation` root span. That has two consequences in Maple: + +- **Turns.** Maple counts one turn per trace, so a user message that makes three model calls (one per tool round trip) shows as three turns. +- **Duplicates.** If your app already sends its own traces to Maple (the Vercel AI SDK, OpenAI Agents SDK, LangChain and so on), every model call is recorded twice: once by your app and once by Broadcast. + +The `trace` request field fixes both. OpenRouter uses `trace.trace_id` and `trace.parent_span_id` verbatim as the OTLP trace id and parent span id, so the Broadcast spans land inside your trace. Use W3C ids: 32 lowercase hex characters for the trace id and 16 for the span id. + +In TypeScript, the least invasive place is a `fetch` wrapper. It reads the span that is active when the SDK sends the HTTP request, which is your framework's model-call span if it has one: + +```ts +import { trace } from "@opentelemetry/api" + +// Nests each OpenRouter Broadcast trace under the span that made the request. +export const openRouterFetch: typeof fetch = (input, init) => { + const span = trace.getActiveSpan()?.spanContext() + if (span && typeof init?.body === "string") { + const body = JSON.parse(init.body) + body.trace = { ...body.trace, trace_id: span.traceId, parent_span_id: span.spanId } + init = { ...init, body: JSON.stringify(body) } + } + return fetch(input, init) +} +``` + +Pass it to the client: `new OpenAI({ baseURL, apiKey, fetch: openRouterFetch })` or `createOpenRouter({ apiKey, fetch: openRouterFetch })`. Both SDKs send the body as a JSON string, which is what the wrapper expects. + +In Python, add the current span's ids next to `session_id`: + +```py +from opentelemetry import trace + + +def openrouter_extra_body(conversation_id: str) -> dict: + body = {"session_id": conversation_id} + ctx = trace.get_current_span().get_span_context() + if ctx.is_valid: + body["trace"] = { + "trace_id": format(ctx.trace_id, "032x"), + "parent_span_id": format(ctx.span_id, "016x"), + } + return body + + +client.chat.completions.create(model=model, messages=messages, extra_body=openrouter_extra_body(conversation_id)) +``` + +This parents the Broadcast spans to whatever span is current at the call site, usually your turn or agent span. + +Once the spans share a trace, Maple counts each model call once: + +- If the Broadcast `LLM Generation` span is a descendant of your app's model-call span, Maple counts the usage at the deepest span that reports it. +- If they are siblings, or in different traces, Maple matches them by `gen_ai.response.id`. OpenRouter's is the `gen-…` id in the response. Instrumentations that record the response id (the Vercel AI SDK, OpenTelemetry's `openai-v2` instrumentation) match; ones that don't are counted twice. + +Use the same value for `session_id` as for your framework's conversation id. A trace that carries two different session ids is assigned to one of them, the lexically larger, without a warning. + +## Prompts, completions and Privacy Mode + +Broadcast includes content by default. The `LLM Generation` span carries the request messages in `gen_ai.prompt` and the reply in `gen_ai.completion`, both as JSON strings. + +They are wrapped objects, not message arrays: `{"messages": [...]}` for the prompt and `{"completion": "...", "reasoning": "..."}` for the reply. Maple shows each one as a single raw JSON block on the span, so the transcript is readable but not rendered as chat bubbles, and turns have no label from the user's message. If your app also emits `gen_ai.input.messages` on its own spans, that transcript renders normally. + +In our captures, the completion object also echoes the request body, including your tool definitions and the `user`, `session_id` and `trace` fields. + +To keep content out of Maple, turn on **Privacy Mode** on the destination. OpenRouter strips prompts and completions; tokens, cost, timing, model and metadata still arrive, so sessions, counts and cost still work. Privacy Mode does not remove `user`, `session_id` or custom `trace` metadata, so don't put emails or names in them. + +OpenRouter shortens any single value over 10,000,000 characters and adds a `.truncated` attribute. Maple's ingest accepts request bodies up to 20 MiB. + +## Tools, errors and provider fallbacks + +Broadcast sees HTTP requests to OpenRouter, not your agent. Your tools run in your process, so a Broadcast-only setup has **no tool spans**: tool calls appear inside the completion JSON, but Maple's tool counts, tool pages and tool failure groups stay empty, and a tool that throws is invisible. There is no agent name either, so sub-agents don't get lanes. + +For tools and agent structure, instrument the app with its framework guide from [Agent tracing](/docs/agent-tracing) and nest Broadcast under it as shown above. Broadcast then adds what most frameworks lack: cost and the provider routing. + +Model failures do show up. Each request's trace has an `LLM Generation` root and a `provider attempt N: ` child per upstream provider OpenRouter tried: + +- A failed attempt that OpenRouter recovered from by falling back to another provider is a retry. Maple does not count it as a failure. +- If every attempt fails, `LLM Generation` has status Error with the message `Provider returned error`, and Maple counts it as a `provider_error` in the session. + +On that error path, OpenRouter currently drops the `trace` object, so the failed call lands in its own trace. It keeps `session.id`, so it still joins the right session. + +## Tokens and cost + +Every `LLM Generation` span carries: + +| Attribute | What Maple does with it | +| --- | --- | +| `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` | Input and output tokens | +| `gen_ai.usage.input_tokens.cached` | Cache-read tokens (included in input) | +| `gen_ai.usage.output_tokens.reasoning` | Reasoning tokens (included in output) | +| `gen_ai.usage.total_cost` | Cost in USD, OpenRouter's actual charge | +| `gen_ai.request.model`, `gen_ai.response.model` | Model, as the OpenRouter slug (`openai/gpt-4o-mini`) | +| `gen_ai.response.finish_reasons` | Truncation and refusal checks | + +Cost is the main reason to use Broadcast even when your app is already instrumented. Maple never prices tokens itself, and most frameworks export no cost, so their sessions read "unpriced". When the app's span and the Broadcast span describe the same call, Maple keeps the larger cost of the two, which is OpenRouter's. + +Streaming doesn't matter here: OpenRouter accounts usage server-side, so streamed calls carry tokens and cost without `stream_options.include_usage`. + +Not read by Maple: + +- `gen_ai.usage.input_tokens.cache_write` (cache writes). Totals are unaffected, but the cache-write column stays empty. +- `trace.metadata.openrouter.first_token_ms` (time to first token). Maple reads TTFT only from `gen_ai.response.time_to_first_chunk`. +- The **Cost** option under **Additional generation metadata**, which adds `span.metadata.openrouter_generation.*` attributes. You don't need it for Maple. + +`gen_ai.provider.name` is the model's author (`openai`, `anthropic`), not the provider that served the request. That one is in `trace.metadata.openrouter.provider_name` (for example `Amazon Bedrock`). + +## Short-lived processes + +Broadcast needs no flush. OpenRouter sends traces from its own servers after each request completes, so a script, a serverless function or a CLI that exits right after the response loses nothing. + +Expect about a minute between the response and the trace in Maple. + +If your app also exports its own spans, those still need the usual flush on exit (`sdk.shutdown()` in Node, `provider.force_flush()` in Python). Otherwise Maple shows the Broadcast spans with a missing parent. + +## Check that it works + +Run one conversation of three or more turns, with a tool call, sending the same `session_id` on every request. Wait about a minute, then open **Agent Sessions** in Maple and filter by service `openrouter`. + +- **One session for the conversation**, named by your `session_id`, with vendor **OpenRouter**. A second conversation is a second session. +- **Model calls.** Each call has an `LLM Generation` span with one or more `provider attempt N: ` children, and sometimes `generation` or `moderation` children. Only `LLM Generation` counts as a model call. +- **Tokens and cost** on the session and per model, with cost in USD. +- **Transcript**: each call's prompt and completion as a JSON block, unless Privacy Mode is on. +- **Turns**: one per trace. Broadcast-only, that's one per model call. Nested under your own traces, it's one per turn of your agent. +- **Tools**: none from Broadcast. Tool spans come from your app's instrumentation. + +The service name on Broadcast spans is always `openrouter`, and there is no environment attribute. Custom keys in the `trace` object arrive as `trace.metadata.` attributes, which you can search in [Traces](/docs/explore/traces) but which don't set Maple's environment or service. + +## Troubleshooting + +- **Test Connection fails.** The endpoint must be the full `https://ingest.maple.dev/v1/traces` URL and the headers valid JSON with `"Authorization": "Bearer "`. EU organizations use `ingest.eu.maple.dev`. +- **Test Connection passes but nothing arrives.** Check the destination's API key filter and data regions against the key and endpoint your app actually uses, and that **Enable Broadcast** is on for the account or organization your app's key belongs to. A placeholder key such as `MAPLE_TEST` passes the test, but Maple discards everything it sends. +- **Every call is its own session, named `trace:`.** The request has no `session_id`. Check the outgoing body, not just your code: a wrapper or framework may drop unknown fields. +- **A session named `trace:00000000000000000000000000000001` with one span.** That's the `openrouter-connection-test` span from **Test Connection**. Ignore it. +- **A session with an `openai/gpt-4-turbo` call you never made, trace name `Test Trace - OpenRouter Observability`.** OpenRouter's sample trace from the destination settings, with sample tokens and cost. +- **Tokens or LLM calls are about double what you expect.** Your app's instrumentation and Broadcast both report the call, in different traces and without a shared response id. Nest Broadcast with `trace.trace_id` and `trace.parent_span_id`, or filter that service's API key out of the destination. +- **Broadcast spans show up as a separate trace despite `trace.trace_id`.** The id isn't W3C hex, or the call failed: on the all-providers-failed path OpenRouter drops the `trace` object. +- **Turns have no label and the transcript is raw JSON.** Expected for Broadcast content (see [above](#prompts-completions-and-privacy-mode)). +- **Sessions are missing entirely at random.** The destination's sampling rate is below 1.0. +- **Tool counts are zero.** Expected for Broadcast-only. Add your framework's instrumentation for tool spans. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Agent tracing guides](/docs/agent-tracing) +- [OpenRouter Broadcast](https://openrouter.ai/docs/guides/features/broadcast) and its [OpenTelemetry Collector destination](https://openrouter.ai/docs/guides/features/broadcast/otel-collector) +- [Vercel AI SDK](/docs/agent-tracing/vercel-ai-sdk), if you call OpenRouter through `@openrouter/ai-sdk-provider` diff --git a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md new file mode 100644 index 0000000000..fcbf5f0310 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md @@ -0,0 +1,883 @@ +--- +title: "Trace any AI agent with the OpenTelemetry GenAI conventions" +description: "Write the three OpenTelemetry spans Maple needs by hand, in any language, so a hand-rolled agent loop shows up in Agent Sessions with its transcript, tool calls, tokens and cost." +group: "AI Agents" +order: 51 +navLabel: "Any language (OTel GenAI)" +icon: "opentelemetry" +--- + +If your agent is a loop you wrote yourself, runs on a framework without OpenTelemetry support, or lives in Go, Rust, Ruby or Elixir, nothing emits agent spans for you. You write them. Maple reads the OpenTelemetry GenAI semantic conventions, so three kinds of span are enough: one `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. + +What goes wrong is the detail. The GenAI conventions are still in Development status and have renamed attributes several times, and a span that looks right can still arrive with an empty transcript: messages sent as plain text, content put in span events, a JSON string cut off by an attribute length limit, or a fresh conversation id on every request. + +This guide lists the exact keys and value formats Maple reads, with full examples in TypeScript and Python and a shorter one in Go. It follows the conventions in [`semantic-conventions-genai`](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. The code was written against the OpenTelemetry JS SDK 2.11, Python SDK 1.45 and Go SDK 1.46, with the OpenAI SDK 7.23 for JavaScript and 3.20 for Python. + +If you use a framework, check the [framework guides](/docs/agent-tracing) first. Most of them emit these spans for you. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-opentelemetry](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-opentelemetry) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for my hand-rolled agent in this project, using the OpenTelemetry GenAI conventions. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-opentelemetry -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## The three spans Maple needs + +One user message produces one trace: + +```text +invoke_agent support gen_ai.conversation.id = chat_42 +├── chat openai/gpt-4o-mini model call: asks for get_weather +├── execute_tool get_weather tool call +└── chat openai/gpt-4o-mini model call: final answer +``` + +Maple classifies a span by `gen_ai.operation.name`, not by its name. A span without that attribute is ignored by Agent Sessions, even if it carries a model or token counts. The span names above follow the spec (`invoke_agent {agent}`, `chat {model}`, `execute_tool {tool}`) and are what you'll see in the trace view. + +**`invoke_agent`** (span kind `INTERNAL`), one per agent run: + +| Attribute | Value | What Maple does with it | +| --- | --- | --- | +| `gen_ai.operation.name` | `invoke_agent` | Marks the span as an agent run. A root agent span starts a turn. | +| `gen_ai.agent.name` | `support` | Agent facet, and a separate lane for each sub-agent. | +| `gen_ai.conversation.id` | your chat or thread id | Groups the trace into a session. | +| `gen_ai.input.messages`, `gen_ai.output.messages` | JSON string, see below | Optional. The turn's prompt and answer, and a sub-agent lane's input and output. | + +**`chat`** (span kind `CLIENT`), one per model call. `generate_content` and `text_completion` work too. + +| Attribute | Value | What Maple does with it | +| --- | --- | --- | +| `gen_ai.operation.name` | `chat` | Counts the span as an LLM call. | +| `gen_ai.provider.name` | `openai`, `anthropic`, `gcp.gemini`, `openrouter`... | Decides how token counts are read (see [Tokens and cost](#tokens-and-cost)). | +| `gen_ai.request.model`, `gen_ai.response.model` | model ids | Model facet. The response model wins when both are set. | +| `gen_ai.response.id` | the provider's response id | Counts two spans for the same response as one call. | +| `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` | int | Token totals. | +| `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` | int | Cache and reasoning breakdown. | +| `gen_ai.usage.cost` | double, USD | Session cost. Not part of the OpenTelemetry spec. | +| `gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages` | JSON string | The transcript. | +| `gen_ai.response.finish_reasons` | string array, e.g. `["stop"]` | Refusal (`content_filter`) and truncation (`length`) checks. | +| `gen_ai.response.time_to_first_chunk` | double, **seconds** | Time to first token on streamed calls. | + +**`execute_tool`** (span kind `INTERNAL`), one per tool call: + +| Attribute | Value | What Maple does with it | +| --- | --- | --- | +| `gen_ai.operation.name` | `execute_tool` | Counts the span as a tool call. | +| `gen_ai.tool.name` | `get_weather` | Tool pages and facet. | +| `gen_ai.tool.call.id` | the model's tool call id | Matches the call to the model message that requested it. | +| `gen_ai.tool.call.arguments` | JSON string of an object | Arguments in the transcript. | +| `gen_ai.tool.call.result` | JSON string of an object or array | Result in the transcript. A bare string is dropped. | + +A failed span of any kind gets status `ERROR` with the error message and an `error.type` attribute. Maple counts a span as failed when either is present. + +Older spellings still work, so spans from an emitter written against an earlier version of the conventions don't need a rewrite: `gen_ai.system` (renamed to `gen_ai.provider.name` in 1.37), `gen_ai.usage.prompt_tokens` and `completion_tokens`, and whole-value `gen_ai.prompt` and `gen_ai.completion`. When both old and new keys are set, the new one wins. Use the current names in new code. + +### The message format + +`gen_ai.input.messages` and `gen_ai.output.messages` are JSON arrays of `{role, parts}` messages, serialized to a string: + +```json +[ + { "role": "user", "parts": [{ "type": "text", "content": "What's the weather in Berlin?" }] }, + { + "role": "assistant", + "parts": [{ "type": "tool_call", "id": "call_1", "name": "get_weather", "arguments": { "city": "Berlin" } }] + }, + { "role": "tool", "parts": [{ "type": "tool_call_response", "id": "call_1", "response": "{\"temperature_c\":21}" }] } +] +``` + +Output messages add a `finish_reason` to each message. `gen_ai.system_instructions` is an array of parts without a role: `[{"type":"text","content":"You are a concise assistant."}]`. Maple also renders `reasoning` parts, and accepts `content` (a string or a part array) in place of `parts`. + +Always set these as a **string** holding JSON. Maple drops a plain-text value like `"What's the weather?"`. The spec also allows structured attribute values, but Maple receives those as a list of strings and renders them as raw text. + +## Export spans to Maple + +Point the OTLP exporter at Maple with the standard variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporters append `/v1/traces` themselves. + +TypeScript (Node.js 20 or newer): + +```bash +npm install @opentelemetry/api @opentelemetry/sdk-trace-node @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources openai +``` + +```ts +// tracing.ts: import this first in every entry point +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { resourceFromAttributes } from "@opentelemetry/resources" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { NodeTracerProvider } from "@opentelemetry/sdk-trace-node" + +export const provider = new NodeTracerProvider({ + resource: resourceFromAttributes({ + "service.name": "support-agent", + "deployment.environment.name": "production", + }), + // Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS + spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())], +}) +// Registers the global provider and the async context manager, so spans nest across awaits +provider.register() +``` + +Python (3.10 or newer): + +```bash +pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" openai +``` + +```py +# tracing.py: import this first in every entry point +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) +``` + +Import `tracing` first in every entry point: the web server, each worker and each script. If your app already has a `TracerProvider` (from Sentry, Datadog, `opentelemetry-instrument` or `NodeSDK`), don't create a second one. Add the `BatchSpanProcessor` to the existing provider instead. + +Name the tracer after your app. Maple recognizes some frameworks by their instrumentation scope name, and a tracer named `openrouter` or `langsmith`, for example, makes Maple read the session from that framework's key and ignore `gen_ai.conversation.id`. + +## Instrument the agent loop in TypeScript + +The examples call an OpenAI-compatible Chat Completions API through OpenRouter. Any provider works; change `baseURL`, the model ids and `PROVIDER`. Each model call streams, so the same code records time to first chunk and works for streamed chat replies. + +```ts +// agent.ts +import { type Span, SpanKind, SpanStatusCode, trace } from "@opentelemetry/api" +import OpenAI from "openai" +import type { + ChatCompletionAssistantMessageParam, + ChatCompletionMessageFunctionToolCall, + ChatCompletionMessageParam, + ChatCompletionTool, +} from "openai/resources/chat/completions" + +const tracer = trace.getTracer("support-agent") +const client = new OpenAI({ baseURL: "https://openrouter.ai/api/v1", apiKey: process.env.OPENROUTER_API_KEY }) +const PROVIDER = "openrouter" // gen_ai.provider.name: who you send the request to + +type Message = ChatCompletionMessageParam +type ToolCall = ChatCompletionMessageFunctionToolCall +type Tool = { definition: ChatCompletionTool; run: (args: Record) => unknown } +export type Agent = { name: string; model: string; instructions: string; tools: Record } + +const json = (value: unknown) => JSON.stringify(value) + +// OpenAI message -> GenAI semconv message: { role, parts: [...] } +function toSemconv(message: Message) { + if (message.role === "tool") { + return { role: "tool", parts: [{ type: "tool_call_response", id: message.tool_call_id, response: message.content }] } + } + const parts: object[] = typeof message.content === "string" && message.content ? [{ type: "text", content: message.content }] : [] + if (message.role === "assistant") { + for (const call of message.tool_calls ?? []) { + if (call.type !== "function") continue + parts.push({ type: "tool_call", id: call.id, name: call.function.name, arguments: JSON.parse(call.function.arguments || "{}") }) + } + } + return { role: message.role, parts } +} + +function markFailed(span: Span, error: unknown) { + const err = error instanceof Error ? error : new Error(String(error)) + span.setStatus({ code: SpanStatusCode.ERROR, message: err.message }) + span.setAttribute("error.type", err.name) +} + +// One model call = one `chat` span. Streams, so time to first chunk is recorded too. +async function chat(agent: Agent, messages: Message[], onText?: (delta: string) => void) { + return tracer.startActiveSpan( + `chat ${agent.model}`, + { + kind: SpanKind.CLIENT, + attributes: { + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": PROVIDER, + "gen_ai.request.model": agent.model, + "gen_ai.system_instructions": json([{ type: "text", content: agent.instructions }]), + "gen_ai.input.messages": json(messages.map(toSemconv)), + }, + }, + async (span) => { + try { + const started = performance.now() + const stream = await client.chat.completions.create({ + model: agent.model, + messages: [{ role: "system", content: agent.instructions }, ...messages], + tools: Object.values(agent.tools).map((tool) => tool.definition), + stream: true, + stream_options: { include_usage: true }, // without it, streamed calls report no tokens + }) + let id = "" + let model = agent.model + let text = "" + let finishReason = "stop" + let usage: (OpenAI.CompletionUsage & { cost?: number }) | undefined + const calls: ToolCall[] = [] + for await (const chunk of stream) { + if (!id) span.setAttribute("gen_ai.response.time_to_first_chunk", (performance.now() - started) / 1000) + id = chunk.id + model = chunk.model + if (chunk.usage) usage = chunk.usage + const choice = chunk.choices[0] + if (!choice) continue + if (choice.finish_reason) finishReason = choice.finish_reason + if (choice.delta.content) { + text += choice.delta.content + onText?.(choice.delta.content) + } + for (const delta of choice.delta.tool_calls ?? []) { + const call = (calls[delta.index] ??= { id: "", type: "function", function: { name: "", arguments: "" } }) + if (delta.id) call.id = delta.id + call.function.name += delta.function?.name ?? "" + call.function.arguments += delta.function?.arguments ?? "" + } + } + const reply: ChatCompletionAssistantMessageParam = { role: "assistant", content: text, ...(calls.length ? { tool_calls: calls } : {}) } + span.setAttributes({ + "gen_ai.response.id": id, + "gen_ai.response.model": model, + "gen_ai.response.finish_reasons": [finishReason], + "gen_ai.output.messages": json([{ ...toSemconv(reply), finish_reason: finishReason }]), + }) + if (usage) { + span.setAttributes({ + "gen_ai.usage.input_tokens": usage.prompt_tokens, + "gen_ai.usage.output_tokens": usage.completion_tokens, + "gen_ai.usage.cache_read.input_tokens": usage.prompt_tokens_details?.cached_tokens ?? 0, + "gen_ai.usage.reasoning.output_tokens": usage.completion_tokens_details?.reasoning_tokens ?? 0, + }) + if (usage.cost !== undefined) span.setAttribute("gen_ai.usage.cost", usage.cost) // OpenRouter returns USD cost + } + return reply + } catch (error) { + markFailed(span, error) + throw error + } finally { + span.end() + } + }, + ) +} + +// One tool call = one `execute_tool` span. A failure is marked on the span and returned to the model. +async function runTool(agent: Agent, call: ToolCall) { + const name = call.function.name + return tracer.startActiveSpan( + `execute_tool ${name}`, + { + kind: SpanKind.INTERNAL, + attributes: { + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.type": "function", + "gen_ai.tool.call.id": call.id, + "gen_ai.tool.call.arguments": call.function.arguments || "{}", + }, + }, + async (span) => { + try { + const result = await agent.tools[name]!.run(JSON.parse(call.function.arguments || "{}")) + // Maple reads a JSON object or array here; a bare string is dropped + const output = json(typeof result === "object" && result !== null ? result : { result }) + span.setAttribute("gen_ai.tool.call.result", output) + return output + } catch (error) { + markFailed(span, error) + return json({ error: error instanceof Error ? error.message : String(error) }) + } finally { + span.end() + } + }, + ) +} + +// One agent run = one `invoke_agent` span. For a user turn it is the root of the trace. +export async function runAgent( + agent: Agent, + messages: Message[], + options: { conversationId?: string; onText?: (delta: string) => void } = {}, +): Promise { + const input = messages.at(-1) + return tracer.startActiveSpan( + `invoke_agent ${agent.name}`, + { + kind: SpanKind.INTERNAL, + attributes: { + "gen_ai.operation.name": "invoke_agent", + "gen_ai.agent.name": agent.name, + ...(options.conversationId ? { "gen_ai.conversation.id": options.conversationId } : {}), + ...(input ? { "gen_ai.input.messages": json([toSemconv(input)]) } : {}), + }, + }, + async (span) => { + try { + for (let step = 0; step < 10; step++) { + const reply = await chat(agent, messages, options.onText) + messages.push(reply) + if (!reply.tool_calls?.length) { + span.setAttribute("gen_ai.output.messages", json([toSemconv(reply)])) + return typeof reply.content === "string" ? reply.content : "" + } + for (const call of reply.tool_calls) { + if (call.type !== "function") continue + messages.push({ role: "tool", tool_call_id: call.id, content: await runTool(agent, call) }) + } + } + throw new Error("agent exceeded 10 steps") + } catch (error) { + markFailed(span, error) + throw error + } finally { + span.end() + } + }, + ) +} +``` + +A chat backend keeps one message list per conversation and calls `runAgent` once per user message: + +```ts +// main.ts +import { provider } from "./tracing" // first import: sets up the provider +import type { ChatCompletionMessageParam } from "openai/resources/chat/completions" +import { type Agent, runAgent } from "./agent" + +const assistant: Agent = { + name: "support", + model: "openai/gpt-4o-mini", + instructions: "You are a concise assistant.", + tools: { + get_weather: { + definition: { + type: "function", + function: { + name: "get_weather", + description: "Current weather for a city", + parameters: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, + }, + }, + run: ({ city }) => ({ city, temperature_c: 21, condition: "partly cloudy" }), + }, + }, +} + +// One history per conversation. Store it in your database in a real backend. +const histories = new Map() + +export async function handleMessage(chatId: string, text: string, onText?: (delta: string) => void) { + const history = histories.get(chatId) ?? [] + histories.set(chatId, history) + history.push({ role: "user", content: text }) + return runAgent(assistant, history, { conversationId: chatId, onText }) +} + +// In a script: two messages of one conversation, then flush +try { + await handleMessage("chat_42", "Hi! Briefly introduce yourself.") + await handleMessage("chat_42", "What's the weather in Berlin?", (delta) => process.stdout.write(delta)) +} finally { + await provider.shutdown() +} +``` + +## Instrument the agent loop in Python + +The same loop with the OpenAI Python SDK: + +```py +# agent.py +import json +import os +import time +from dataclasses import dataclass, field +from typing import Any, Callable + +from openai import OpenAI, omit +from opentelemetry import trace +from opentelemetry.trace import SpanKind, Status, StatusCode + +tracer = trace.get_tracer("support-agent") +client = OpenAI(base_url="https://openrouter.ai/api/v1", api_key=os.environ["OPENROUTER_API_KEY"]) +PROVIDER = "openrouter" # gen_ai.provider.name: who you send the request to + + +@dataclass +class Tool: + definition: dict[str, Any] + run: Callable[..., Any] + + +@dataclass +class Agent: + name: str + model: str + instructions: str + tools: dict[str, Tool] = field(default_factory=dict) + + +def to_semconv(message: dict[str, Any]) -> dict[str, Any]: + """OpenAI message -> GenAI semconv message: {role, parts: [...]}.""" + if message["role"] == "tool": + part = {"type": "tool_call_response", "id": message["tool_call_id"], "response": message["content"]} + return {"role": "tool", "parts": [part]} + parts: list[dict[str, Any]] = [] + if message.get("content"): + parts.append({"type": "text", "content": message["content"]}) + for call in message.get("tool_calls") or []: + fn = call["function"] + parts.append({"type": "tool_call", "id": call["id"], "name": fn["name"], "arguments": json.loads(fn["arguments"] or "{}")}) + return {"role": message["role"], "parts": parts} + + +def mark_failed(span: trace.Span, error: Exception) -> None: + span.set_status(Status(StatusCode.ERROR, str(error))) + span.set_attribute("error.type", type(error).__qualname__) + + +def chat(agent: Agent, messages: list[dict[str, Any]], on_text: Callable[[str], None] | None = None) -> dict[str, Any]: + """One model call = one `chat` span. Streams, so time to first chunk is recorded too.""" + attributes = { + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": PROVIDER, + "gen_ai.request.model": agent.model, + "gen_ai.system_instructions": json.dumps([{"type": "text", "content": agent.instructions}]), + "gen_ai.input.messages": json.dumps([to_semconv(m) for m in messages]), + } + with tracer.start_as_current_span(f"chat {agent.model}", kind=SpanKind.CLIENT, attributes=attributes) as span: + try: + started = time.perf_counter() + stream = client.chat.completions.create( + model=agent.model, + messages=[{"role": "system", "content": agent.instructions}, *messages], + tools=[tool.definition for tool in agent.tools.values()] or omit, + stream=True, + stream_options={"include_usage": True}, # without it, streamed calls report no tokens + ) + response_id, model, text, finish_reason, usage = "", agent.model, "", "stop", None + calls: dict[int, dict[str, Any]] = {} + for chunk in stream: + if not response_id: + span.set_attribute("gen_ai.response.time_to_first_chunk", time.perf_counter() - started) + response_id, model = chunk.id, chunk.model + usage = chunk.usage or usage + if not chunk.choices: + continue + choice = chunk.choices[0] + finish_reason = choice.finish_reason or finish_reason + if choice.delta.content: + text += choice.delta.content + if on_text: + on_text(choice.delta.content) + for delta in choice.delta.tool_calls or []: + call = calls.setdefault(delta.index, {"id": "", "type": "function", "function": {"name": "", "arguments": ""}}) + call["id"] = delta.id or call["id"] + if delta.function: + call["function"]["name"] += delta.function.name or "" + call["function"]["arguments"] += delta.function.arguments or "" + reply: dict[str, Any] = {"role": "assistant", "content": text} + if calls: + reply["tool_calls"] = [calls[i] for i in sorted(calls)] + span.set_attributes({ + "gen_ai.response.id": response_id, + "gen_ai.response.model": model, + "gen_ai.response.finish_reasons": [finish_reason], + "gen_ai.output.messages": json.dumps([{**to_semconv(reply), "finish_reason": finish_reason}]), + }) + if usage: + span.set_attributes({ + "gen_ai.usage.input_tokens": usage.prompt_tokens, + "gen_ai.usage.output_tokens": usage.completion_tokens, + "gen_ai.usage.cache_read.input_tokens": getattr(usage.prompt_tokens_details, "cached_tokens", None) or 0, + "gen_ai.usage.reasoning.output_tokens": getattr(usage.completion_tokens_details, "reasoning_tokens", None) or 0, + }) + cost = getattr(usage, "cost", None) # OpenRouter returns USD cost + if cost is not None: + span.set_attribute("gen_ai.usage.cost", cost) + return reply + except Exception as error: + mark_failed(span, error) + raise + + +def run_tool(agent: Agent, call: dict[str, Any]) -> str: + """One tool call = one `execute_tool` span. A failure is marked on the span and returned to the model.""" + name, arguments = call["function"]["name"], call["function"]["arguments"] or "{}" + attributes = { + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.type": "function", + "gen_ai.tool.call.id": call["id"], + "gen_ai.tool.call.arguments": arguments, + } + with tracer.start_as_current_span(f"execute_tool {name}", kind=SpanKind.INTERNAL, attributes=attributes) as span: + try: + result = agent.tools[name].run(**json.loads(arguments)) + # Maple reads a JSON object or array here; a bare string is dropped + output = json.dumps(result if isinstance(result, (dict, list)) else {"result": result}) + span.set_attribute("gen_ai.tool.call.result", output) + return output + except Exception as error: + mark_failed(span, error) + return json.dumps({"error": str(error)}) + + +def run_agent( + agent: Agent, + messages: list[dict[str, Any]], + conversation_id: str | None = None, + on_text: Callable[[str], None] | None = None, +) -> str: + """One agent run = one `invoke_agent` span. For a user turn it is the root of the trace.""" + attributes = {"gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agent.name} + if conversation_id: + attributes["gen_ai.conversation.id"] = conversation_id + if messages: + attributes["gen_ai.input.messages"] = json.dumps([to_semconv(messages[-1])]) + with tracer.start_as_current_span(f"invoke_agent {agent.name}", kind=SpanKind.INTERNAL, attributes=attributes) as span: + try: + for _ in range(10): + reply = chat(agent, messages, on_text) + messages.append(reply) + if not reply.get("tool_calls"): + span.set_attribute("gen_ai.output.messages", json.dumps([to_semconv(reply)])) + return reply["content"] + for call in reply["tool_calls"]: + messages.append({"role": "tool", "tool_call_id": call["id"], "content": run_tool(agent, call)}) + raise RuntimeError("agent exceeded 10 steps") + except Exception as error: + mark_failed(span, error) + raise +``` + +```py +# main.py +from tracing import provider # first import: sets up the provider + +from agent import Agent, Tool, run_agent + +get_weather = Tool( + definition={ + "type": "function", + "function": { + "name": "get_weather", + "description": "Current weather for a city", + "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}, + }, + }, + run=lambda city: {"city": city, "temperature_c": 21, "condition": "partly cloudy"}, +) +assistant = Agent("support", "openai/gpt-4o-mini", "You are a concise assistant.", {"get_weather": get_weather}) + +# One history per conversation. Store it in your database in a real backend. +histories: dict[str, list[dict]] = {} + + +def handle_message(chat_id: str, text: str, on_text=None) -> str: + history = histories.setdefault(chat_id, []) + history.append({"role": "user", "content": text}) + return run_agent(assistant, history, conversation_id=chat_id, on_text=on_text) + + +if __name__ == "__main__": + # In a script: two messages of one conversation, then flush + try: + handle_message("chat_42", "Hi! Briefly introduce yourself.") + handle_message("chat_42", "What's the weather in Berlin?", on_text=lambda d: print(d, end="", flush=True)) + finally: + provider.shutdown() +``` + +`start_as_current_span` keeps the span in a context variable, so the `chat` and `execute_tool` spans nest under `invoke_agent` in the same trace. That holds across `await` in asyncio code. It doesn't hold in a new thread: code submitted to a `ThreadPoolExecutor` starts with an empty context and becomes a separate trace. Pass the context along with `contextvars.copy_context().run(...)` if your tools run in threads. + +## Go and other languages + +Any OpenTelemetry SDK works, as long as it sets the same attributes. In Go, start the turn span and wrap your existing client call: + +```go +// genai.go +package agent + +import ( + "context" + "encoding/json" + "fmt" + + "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/codes" + "go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp" + "go.opentelemetry.io/otel/sdk/resource" + sdktrace "go.opentelemetry.io/otel/sdk/trace" + "go.opentelemetry.io/otel/trace" +) + +var tracer = otel.Tracer("support-agent") + +// SetupTracing reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. +// Call Shutdown on the returned provider before the process exits. +func SetupTracing(ctx context.Context) (*sdktrace.TracerProvider, error) { + exporter, err := otlptracehttp.New(ctx) + if err != nil { + return nil, err + } + tp := sdktrace.NewTracerProvider( + sdktrace.WithBatcher(exporter), + sdktrace.WithResource(resource.NewSchemaless( + attribute.String("service.name", "support-agent"), + attribute.String("deployment.environment.name", "production"), + )), + ) + otel.SetTracerProvider(tp) + return tp, nil +} + +// Message is the GenAI semconv shape: {role, parts}. +type Message struct { + Role string `json:"role"` + Parts []map[string]any `json:"parts"` + FinishReason string `json:"finish_reason,omitempty"` +} + +// ChatResult holds what your provider returned, copied verbatim. +type ChatResult struct { + ID, Model, FinishReason string + Output Message + InputTokens, OutputTokens int64 + CostUSD float64 // 0 when the provider doesn't return a cost +} + +func jsonAttr(key string, value any) attribute.KeyValue { + b, _ := json.Marshal(value) + return attribute.String(key, string(b)) +} + +func fail(span trace.Span, err error) { + span.SetStatus(codes.Error, err.Error()) + span.SetAttributes(attribute.String("error.type", fmt.Sprintf("%T", err))) +} + +// StartTurn opens the invoke_agent span for one user message. End it when the turn is done. +func StartTurn(ctx context.Context, agentName, conversationID string) (context.Context, trace.Span) { + return tracer.Start(ctx, "invoke_agent "+agentName, trace.WithAttributes( + attribute.String("gen_ai.operation.name", "invoke_agent"), + attribute.String("gen_ai.agent.name", agentName), + attribute.String("gen_ai.conversation.id", conversationID), + )) +} + +// Chat wraps one model call (your existing client code goes in call). +func Chat(ctx context.Context, model string, input []Message, call func(context.Context) (ChatResult, error)) (ChatResult, error) { + ctx, span := tracer.Start(ctx, "chat "+model, trace.WithSpanKind(trace.SpanKindClient), trace.WithAttributes( + attribute.String("gen_ai.operation.name", "chat"), + attribute.String("gen_ai.provider.name", "openai"), + attribute.String("gen_ai.request.model", model), + jsonAttr("gen_ai.input.messages", input), + )) + defer span.End() + res, err := call(ctx) + if err != nil { + fail(span, err) + return res, err + } + res.Output.FinishReason = res.FinishReason + span.SetAttributes( + attribute.String("gen_ai.response.id", res.ID), + attribute.String("gen_ai.response.model", res.Model), + attribute.StringSlice("gen_ai.response.finish_reasons", []string{res.FinishReason}), + attribute.Int64("gen_ai.usage.input_tokens", res.InputTokens), + attribute.Int64("gen_ai.usage.output_tokens", res.OutputTokens), + jsonAttr("gen_ai.output.messages", []Message{res.Output}), + ) + if res.CostUSD > 0 { + span.SetAttributes(attribute.Float64("gen_ai.usage.cost", res.CostUSD)) + } + return res, nil +} +``` + +An `execute_tool` span follows the same pattern with the attributes from the table. Rust (`opentelemetry` crate), Ruby (`opentelemetry-sdk`), Elixir (`opentelemetry_api`), Java and .NET follow the same pattern: set the attributes from the tables above as strings, ints, doubles and string arrays, and serialize every message and tool payload to a JSON string first. + +## Group turns into one session + +Maple files a trace under the session id it finds in `gen_ai.conversation.id`. Setting it on the `invoke_agent` span of each turn is enough: every span of that trace joins the session, including HTTP and database spans. Use the id your app already has for the conversation, like a chat id or thread id, and pass the same value for every message of that conversation. + +Without it, each trace is its own one-turn session, named `trace:`. A chat backend that handles one message per request then shows one session per message. + +A few things break grouping: + +- **A new id per request.** A UUID generated in the request handler, or the trace id, is different for every message. The GenAI spec says the same: when there's no real conversation id, leave the attribute out rather than inventing one. +- **Different ids in one trace.** If a sub-agent sets its own id, the trace carries two, and Maple silently keeps the one that sorts last. Let sub-agents inherit the session from the trace, as the examples do, or give them the same id. +- **An id only on a span without `gen_ai.operation.name`.** Maple only reads the id from spans it classifies as AI spans. + +### When a framework's session key isn't read: `maple_ai.session.id` + +Some frameworks write their session id under a key Maple doesn't read for that framework, for example OpenInference's `session.id` from LangChain or LlamaIndex instrumentation, or the Vercel AI SDK's `runtimeContext`. The fix is a wrapper span of your own around each turn, carrying Maple's own session key: + +```ts +// chatId comes from your request; frameworkAgent is the framework's agent +await tracer.startActiveSpan( + "invoke_agent support", + { + attributes: { + "gen_ai.operation.name": "invoke_agent", + "gen_ai.agent.name": "support", + "maple_ai.session.id": chatId, + }, + }, + async (span) => { + try { + return await frameworkAgent.run(message) // the framework's spans nest under this one + } finally { + span.end() + } + }, +) +``` + +`maple_ai.session.id` works on any span and overrides framework detection for that span, so the session shows the framework as **Maple**. Put it only on your own wrapper span, never on the framework's spans: a span carrying it loses the framework's attribute decoding, and Maple reads its token counts as if they include cache. Use the same value the framework would use, because with two different session ids in one trace, the one that sorts last wins. You don't need it for spans written by hand with `gen_ai.conversation.id`. + +## Record prompts, responses and tool calls + +The transcript comes from five attributes: `gen_ai.system_instructions`, `gen_ai.input.messages` and `gen_ai.output.messages` on `chat` spans, and `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result` on `execute_tool` spans. Each has to be a JSON string on the span itself. + +- **Span events and logs aren't read.** Older versions of the conventions put messages in events like `gen_ai.user.message`, and some SDKs send them as log records. Agent Sessions only reads span attributes. +- **Indexed keys aren't read.** `gen_ai.prompt.0.content` and `llm.input_messages.0.message.content` are separate dialects. Maple doesn't reassemble them. +- **Attribute length limits break the JSON.** The SDKs don't limit attribute length by default. If `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` or `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` is set (some platforms and distributions set one), a long history is cut mid-string, no longer parses, and Maple drops the whole attribute. Leave them unset. To cap size yourself, drop the oldest messages whole and serialize what's left. +- **Batches over 20 MiB are rejected.** Maple's ingest endpoint returns 413 for larger requests. Base64 images in every message's history add up fast; replace them with a placeholder part before serializing. + +The spec treats all of this as opt-in, because prompts and tool results often contain personal data. To keep metadata only, skip those five attributes; models, tokens, timings, tool names and errors still work, and the transcript stays empty. To keep content but remove specific values, redact them in the `toSemconv` / `to_semconv` function before serializing, or with an OpenTelemetry Collector `redaction` or `transform` processor. Keep API keys out of tool arguments and results. + +## Tools, errors and sub-agents + +When a tool throws, the examples mark its span failed and return the error to the model as the tool result, so the agent can recover: + +- status `ERROR`, with the error message as the status description; +- `error.type`, the exception class (`RuntimeError`, `TypeError`) or an error code; +- no `gen_ai.tool.call.result`, because the tool didn't return one. + +Maple counts a span as failed if it has `ERROR` status, a non-empty `error.type`, or `gen_ai.response.status` set to `failed`. A tool that catches its own error and returns `{"error": "..."}` without marking the span shows as successful. The tool pages group failures by their message, with ids and numbers masked, so the status description should say what failed. + +Set `gen_ai.tool.call.id` to the id the model gave the call. Maple matches each result to its call by id, which keeps parallel calls from the same model reply apart. + +For a sub-agent, run it inside a tool call: + +```ts +// weatherWorker is an Agent like `assistant` above, with get_weather +const orchestrator: Agent = { + name: "orchestrator", + model: "openai/gpt-4o-mini", + instructions: "Delegate weather questions to the weather worker.", + tools: { + ask_weather_worker: { + definition: { + type: "function", + function: { + name: "ask_weather_worker", + description: "Ask the weather worker a question", + parameters: { type: "object", properties: { question: { type: "string" } }, required: ["question"] }, + }, + }, + // No conversationId: the sub-agent's spans are in this trace, so it inherits the session + run: ({ question }) => runAgent(weatherWorker, [{ role: "user", content: String(question) }]), + }, + }, +} +``` + +Maple shows an `execute_tool` span whose only child is an `invoke_agent` span as a delegation, in a lane named after the sub-agent's `gen_ai.agent.name`, with the tool's arguments and result as the lane's input and output. Give every agent a distinct name: two agents called `agent` share one lane, and an agent span without a name gets no lane at all. + +## Tokens and cost + +Put usage on `chat` spans only. Maple sums it per model call and nets a parent's usage against its children, but an agent span that reports a running total for the whole conversation is counted on top. + +Maple reads five buckets: `input_tokens`, `output_tokens`, `cache_read.input_tokens`, `cache_write.input_tokens` and `reasoning.output_tokens`, all under `gen_ai.usage.`. It also accepts `gen_ai.usage.cache_creation.input_tokens`, the spelling before `cache_write`. It doesn't read `total_tokens`, `gen_ai.usage.reasoning_tokens` or `gen_ai.usage.cache_read_input_tokens`. + +Providers disagree on whether input includes cached tokens, so Maple interprets the numbers by `gen_ai.provider.name`. Copy the provider's raw numbers and set the provider to the API you called: + +| `gen_ai.provider.name` | Input tokens | Output tokens | +| --- | --- | --- | +| `anthropic` | Anthropic's `input_tokens`, **excluding** cache | includes thinking | +| `gcp.gemini`, `gcp.vertex_ai` | `promptTokenCount`, including cache | `candidatesTokenCount`, excluding thoughts | +| `openai`, `openrouter`, anything else | `prompt_tokens`, including cache | includes reasoning | + +The Anthropic row differs from the spec, which asks for the inclusive total on every provider. If you send Anthropic's input as `input_tokens + cache_read + cache_write`, Maple counts cached tokens twice. A call to Claude through OpenRouter's OpenAI-compatible API uses `openrouter`, and OpenAI-shaped numbers. + +Streaming needs `stream_options: { include_usage: true }` on OpenAI's API, or the stream carries no usage. The usage arrives in the last chunk, which has no choices. OpenRouter always sends usage. + +Maple never prices tokens. A session shows cost only when `chat` spans carry `gen_ai.usage.cost` in USD (`gen_ai.usage.total_cost` also works); otherwise it's shown as unpriced. OpenRouter returns the cost in `usage.cost`, which the examples copy. OpenAI and Anthropic don't return a cost, so compute it from your own price table or leave it out. + +Set `gen_ai.response.id`. If the same call is also exported by a gateway, for example [OpenRouter Broadcast](/docs/agent-tracing/openrouter), both spans carry the same response id and Maple counts the call once. + +## Short-lived processes + +`BatchSpanProcessor` exports every few seconds. A script, CLI, Lambda or notebook can exit before the last batch is sent. Flush explicitly: + +- **TypeScript:** `await provider.shutdown()` at the end of a script, or `await provider.forceFlush()` before a serverless handler returns (inside `waitUntil` or `after()` on platforms that have one). +- **Python:** `provider.shutdown()` at the end of a script, or `provider.force_flush()` in a `finally` in a Lambda handler or after each notebook cell. +- **Go:** `defer tp.Shutdown(context.Background())` in `main`. + +## Check that it works + +Run one conversation with two or more messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: + +- one session per conversation, with your conversation id as the session id, and the framework shown as **Unidentified** (or **Maple** with `maple_ai.session.id`); +- one turn per user message, labeled with the message, and a transcript with the system instructions, prompts, replies and tool calls; +- `invoke_agent `, `chat ` and `execute_tool ` spans, with every `chat` and `execute_tool` span inside its turn's `invoke_agent` span; +- input and output tokens on every model call, including streamed ones; +- the tool that failed marked as failed, with its message, and every other tool marked successful; +- a lane per sub-agent, named after its `gen_ai.agent.name`; +- cost on each call if your spans carry `gen_ai.usage.cost`, otherwise unpriced. + +## Troubleshooting + +- **Nothing shows up in Agent Sessions, but the trace is in Traces.** No span has `gen_ai.operation.name`. Maple only picks up traces with at least one classified span. +- **Every message is its own session.** `gen_ai.conversation.id` is missing, or it changes per request. Pass the conversation's id on the `invoke_agent` span of every turn. +- **A session merged two conversations.** The id is a constant, or comes from a module-level variable shared across users. Take it from the request. +- **The transcript is empty, but tokens are there.** The messages are plain text, are in span events or logs, or were cut by an attribute length limit. Set them as JSON strings on the span and unset `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT`. +- **The transcript shows raw JSON instead of messages.** The value isn't an array of `{role, parts}` messages, or it was set as a structured attribute instead of a string. +- **Streamed replies have no tokens.** Add `stream_options: { include_usage: true }` and read usage from the final chunk. +- **Anthropic calls show too many input tokens.** `input_tokens` includes cache while the provider is `anthropic`. Send Anthropic's raw `input_tokens`. +- **Tool calls are missing from the tool pages.** The span has no `gen_ai.operation.name: execute_tool` or no `gen_ai.tool.name`. +- **A failed tool shows as successful.** The span has neither `ERROR` status nor `error.type`. +- **Model and tool spans are separate traces.** The context was lost: the spans weren't started inside the `invoke_agent` span's callback, the provider wasn't registered with a context manager (use `provider.register()` in Node), or the work ran in a new thread. +- **Every model call appears twice.** A provider auto-instrumentation (OpenAI, Anthropic, OpenLLMetry, OpenInference) is also active. Keep either your `chat` spans or the instrumentation, not both. See [provider SDKs](/docs/agent-tracing/provider-sdks) for the instrumentation route. +- **Tokens are missing although the span has usage.** The key is a spelling Maple doesn't read, like `gen_ai.usage.total_tokens`, `gen_ai.usage.reasoning_tokens` or `gen_ai.usage.cache_read_input_tokens`. Use the keys in the table above. +- **Exports fail with 401.** The header is missing or malformed. Some SDKs need the space encoded: `Authorization=Bearer%20YOUR_INGEST_KEY`. +- **Nothing arrives from a script.** The process exited before the batch was sent. Call `shutdown()` at the end. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [All agent tracing guides](/docs/agent-tracing) +- [Provider SDKs](/docs/agent-tracing/provider-sdks), for auto-instrumented OpenAI, Anthropic and Gemini clients +- [OpenTelemetry GenAI semantic conventions](https://github.com/open-telemetry/semantic-conventions-genai) +- [GenAI spans](https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-spans.md) and [GenAI agent spans](https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-agent-spans.md) diff --git a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md new file mode 100644 index 0000000000..aa25f42076 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md @@ -0,0 +1,475 @@ +--- +title: "Trace agents built on the OpenAI, Anthropic and Gemini SDKs" +description: "Trace your own agent loop on the OpenAI, Anthropic or Google Gen AI SDK so each conversation is one Maple Agent Session with its transcript, tool calls, failures and tokens, in Python or TypeScript." +group: "AI Agents" +order: 50 +navLabel: "OpenAI, Anthropic & Gemini SDKs" +icon: "openai" +--- + +If your agent is your own loop around `client.chat.completions.create`, `client.messages.create` or `client.models.generate_content`, an instrumentation library can record each model call: the prompt, the reply, the model and the tokens. It can't know where a user's turn starts, which conversation it belongs to, or that your code ran a tool between two calls. Those spans are yours to add, and it's about 40 lines of code. + +Without them, every model call is its own trace, and Maple files every trace without a conversation id as its own session. A four-message chat with two tool calls shows up as six one-call sessions, with no tool calls and nothing tying them together. The second surprise is that the instrumentations record no prompts or replies until you turn content capture on. + +This guide covers Python 3.10+ with `openai` 3.x, `anthropic` 1.x and `google-genai` 2.x, and TypeScript on Node.js with `openai` 7.x (the same pattern works for `@anthropic-ai/sdk` and `@google/genai`). If you use an agent framework on top of these SDKs, such as the OpenAI Agents SDK, LangChain or Pydantic AI, use [that framework's guide](/docs/agent-tracing) instead. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-provider-sdks](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-provider-sdks) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for the OpenAI, Anthropic or Gemini SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-provider-sdks -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Which instrumentation to use + +Maple builds the transcript from GenAI span attributes: `gen_ai.input.messages` and `gen_ai.output.messages` as JSON arrays of `{role, parts}`. It doesn't read span events or OpenTelemetry logs. That rules out several popular options: + +| Language | Use | Why not the alternatives | +| --- | --- | --- | +| Python | The OpenTelemetry GenAI instrumentations: `opentelemetry-instrumentation-genai-openai`, `opentelemetry-instrumentation-genai-anthropic`, `opentelemetry-instrumentation-google-genai` | They write the messages, tokens (including cache and reasoning), response id and finish reasons onto the span in the exact shape Maple reads. | +| TypeScript | A small helper that records the model call itself (below) | `@opentelemetry/instrumentation-openai` 0.20 only patches `openai` below 7.0 and writes messages to log events. There is no OpenTelemetry instrumentation for `@anthropic-ai/sdk` or `@google/genai`. | + +Other packages you'll find: + +- **`opentelemetry-instrumentation-openai-v2`** is deprecated. Its README says it gets security fixes only and points to `opentelemetry-instrumentation-genai-openai`. +- **`pip install opentelemetry-instrumentation-openai`** (without `genai`) installs OpenLLMetry by Traceloop, a different project. The same goes for `opentelemetry-instrumentation-anthropic`. OpenLLMetry records prompts and replies by default. Pick one family and don't install both. +- **OpenInference** (`openinference-instrumentation-openai`) is recognized by Maple as "OpenInference · OpenAI" and its tokens count, but the transcript is the raw request JSON as one message, with no turn labels. The Anthropic and Gemini OpenInference packages show as Unidentified with the same raw transcript. + +The GenAI instrumentations aren't a vendor Maple recognizes by name either, so sessions show **Unidentified** in the framework column. Everything else (transcript, tools, tokens, failures) is read in full. + +## Export spans to Maple + +Both languages use the standard OpenTelemetry exporter variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT="SPAN_ONLY" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. `SPAN_ONLY` is explained in [Record prompts, replies and tool calls](#record-prompts-replies-and-tool-calls). + +### Python + +Install the SDK, the exporter and the instrumentation for each provider you call: + +```bash +pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ + "opentelemetry-instrumentation-genai-openai>=1.2b0" +# Anthropic: opentelemetry-instrumentation-genai-anthropic>=1.2b0 +# Gemini: opentelemetry-instrumentation-google-genai>=1.2b0 +``` + +Set up tracing once, before your first model call: + +```py +# tracing.py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.instrumentation.genai.openai import OpenAIInstrumentor +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider(resource=Resource.create({"service.name": "support-agent"})) +# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +OpenAIInstrumentor().instrument() +# from opentelemetry.instrumentation.genai.anthropic import AnthropicInstrumentor +# from opentelemetry.instrumentation.google_genai import GoogleGenAiSdkInstrumentor +``` + +Import `tracing` at the top of your entry point. The instrumentors patch the SDK classes, so they cover every client, but the patch has to be in place before the first request. If your app already has a `TracerProvider` (Sentry, Logfire, Datadog), add the `BatchSpanProcessor` to that one instead of creating a second. + +### TypeScript + +```bash +npm install @opentelemetry/api @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-proto +``` + +```ts +// instrumentation.ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { NodeSDK, tracing } from "@opentelemetry/sdk-node" + +// The exporter reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. +export const spanProcessor = new tracing.BatchSpanProcessor(new OTLPTraceExporter()) + +export const sdk = new NodeSDK({ serviceName: "support-agent", spanProcessors: [spanProcessor] }) +sdk.start() +``` + +Import it first in your entry point. The helper below creates its spans through `@opentelemetry/api`, so nothing gets monkey-patched and import order only matters for the provider being registered before the first span. + +## Group turns into one session + +Maple groups traces into a session by `gen_ai.conversation.id`. It only needs the id on one AI span per trace, and the natural place is a span that wraps the whole turn: `invoke_agent`, with the agent's name. The model calls and tool calls made inside it become its children, so a turn is one trace. + +Use the id your app already has for the conversation: the chat thread's database id, a support ticket id, the Slack thread timestamp. Don't generate a UUID per request (every turn becomes its own session) or per process (every user shares one). + +### Python + +```py +# agent_tracing.py +import json +import os +from contextlib import contextmanager + +from opentelemetry import trace +from opentelemetry.trace import Status, StatusCode + +tracer = trace.get_tracer("support-agent") + +# Same switch the instrumentors read, so one env var controls content everywhere. +CAPTURE_CONTENT = os.environ.get( + "OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT", "" +).upper() in ("SPAN_ONLY", "SPAN_AND_EVENT") + + +@contextmanager +def agent_span(agent_name: str, conversation_id: str | None = None): + """One agent run. With a conversation id, it's the turn Maple files under that session.""" + attributes = {"gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agent_name} + if conversation_id: + attributes["gen_ai.conversation.id"] = conversation_id + with tracer.start_as_current_span(f"invoke_agent {agent_name}", attributes=attributes) as span: + yield span + + +def run_tool(call_id: str, name: str, arguments: str, tool) -> str: + """One tool call. A failing tool returns its error to the model, and the span still says it failed.""" + with tracer.start_as_current_span( + f"execute_tool {name}", + attributes={ + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.call.id": call_id, + }, + ) as span: + if CAPTURE_CONTENT: + span.set_attribute("gen_ai.tool.call.arguments", arguments) + try: + result = json.dumps(tool(**json.loads(arguments))) + except Exception as exc: + # The exception never leaves this block, so mark the span failed by hand. + span.record_exception(exc) + span.set_status(Status(StatusCode.ERROR, str(exc))) + span.set_attribute("error.type", type(exc).__qualname__) + result = json.dumps({"error": str(exc)}) + if CAPTURE_CONTENT: + span.set_attribute("gen_ai.tool.call.result", result) + return result +``` + +Your loop then looks like this. Only the two `with` and `run_tool` lines are tracing: + +```py +# agent.py +import tracing # noqa: F401 (first import) +from openai import OpenAI + +from agent_tracing import agent_span, run_tool + +client = OpenAI() +MODEL = "gpt-4o-mini" +# get_weather, fetch_transport_data and TOOL_SCHEMAS are your own tools and their JSON schemas. +TOOLS = {"get_weather": get_weather, "fetch_transport_data": fetch_transport_data} + + +def chat_turn(conversation_id: str, history: list, user_text: str) -> str: + """One user message in, one reply out. `history` is this conversation's stored messages.""" + with agent_span("support_agent", conversation_id): + history.append({"role": "user", "content": user_text}) + while True: + response = client.chat.completions.create(model=MODEL, messages=history, tools=TOOL_SCHEMAS) + message = response.choices[0].message + history.append(message.model_dump(include={"role", "content", "tool_calls"}, exclude_none=True)) + if not message.tool_calls: + return message.content + for call in message.tool_calls: + result = run_tool(call.id, call.function.name, call.function.arguments, TOOLS[call.function.name]) + history.append({"role": "tool", "tool_call_id": call.id, "content": result}) +``` + +With Anthropic, the loop is the same shape: run each `tool_use` block with `run_tool(block.id, block.name, json.dumps(block.input), ...)` and send the results back as `tool_result` blocks. With Gemini's automatic function calling, the SDK runs your Python functions itself and the instrumentation records an `execute_tool` span for each, so wrap the turn in `agent_span` but don't call `run_tool`. + +### TypeScript + +The helper records all three spans: the turn, each tool call, and each model call with the attributes the Python instrumentations would write. + +```ts +// agent-tracing.ts +import { type Attributes, type Span, SpanKind, SpanStatusCode, trace } from "@opentelemetry/api" +import type OpenAI from "openai" + +const tracer = trace.getTracer("support-agent") + +// Same switch as the Python instrumentations, so one env var controls content everywhere. +const captureContent = ["SPAN_ONLY", "SPAN_AND_EVENT"].includes( + (process.env.OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT ?? "").toUpperCase(), +) + +/** One agent run. With a conversation id, it's the turn Maple files under that session. */ +export function agentSpan(agentName: string, conversationId: string | undefined, fn: () => Promise) { + const attributes: Attributes = { "gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agentName } + if (conversationId) attributes["gen_ai.conversation.id"] = conversationId + return withSpan(`invoke_agent ${agentName}`, SpanKind.INTERNAL, attributes, () => fn()) +} + +/** One tool call. A failing tool returns its error to the model, and the span still says it failed. */ +export function runTool(callId: string, name: string, args: string, tool: (args: any) => unknown) { + const attributes = { "gen_ai.operation.name": "execute_tool", "gen_ai.tool.name": name, "gen_ai.tool.call.id": callId } + return withSpan(`execute_tool ${name}`, SpanKind.INTERNAL, attributes, async (span) => { + if (captureContent) span.setAttribute("gen_ai.tool.call.arguments", args) + let result: string + try { + result = JSON.stringify(await tool(JSON.parse(args))) + } catch (error) { + markFailed(span, error) + result = JSON.stringify({ error: String(error) }) + } + if (captureContent) span.setAttribute("gen_ai.tool.call.result", result) + return result + }) +} + +type ChatParams = Omit + +/** One model call. Pass `onText` to stream; usage still arrives. */ +export function tracedChat(client: OpenAI, params: ChatParams, onText?: (delta: string) => void) { + const attributes = { "gen_ai.operation.name": "chat", "gen_ai.provider.name": "openai", "gen_ai.request.model": params.model } + return withSpan(`chat ${params.model}`, SpanKind.CLIENT, attributes, async (span) => { + if (captureContent) { + span.setAttribute("gen_ai.input.messages", JSON.stringify(params.messages.map(toGenAiMessage))) + } + let completion: OpenAI.Chat.ChatCompletion + if (onText) { + const started = performance.now() + let firstChunkAt: number | undefined + // Without include_usage, a streamed call reports no tokens at all. + const stream = client.chat.completions.stream({ ...params, stream_options: { include_usage: true } }) + stream.on("content", (delta) => { + firstChunkAt ??= performance.now() + onText(delta) + }) + completion = await stream.finalChatCompletion() + if (firstChunkAt !== undefined) { + span.setAttribute("gen_ai.response.time_to_first_chunk", (firstChunkAt - started) / 1000) + } + } else { + completion = await client.chat.completions.create(params) + } + span.setAttributes({ + "gen_ai.response.id": completion.id, + "gen_ai.response.model": completion.model, + "gen_ai.response.finish_reasons": completion.choices.map((c) => c.finish_reason), + }) + if (completion.usage) { + span.setAttributes({ + "gen_ai.usage.input_tokens": completion.usage.prompt_tokens, + "gen_ai.usage.output_tokens": completion.usage.completion_tokens, + "gen_ai.usage.cache_read.input_tokens": completion.usage.prompt_tokens_details?.cached_tokens ?? 0, + "gen_ai.usage.reasoning.output_tokens": completion.usage.completion_tokens_details?.reasoning_tokens ?? 0, + }) + } + if (captureContent) { + const output = completion.choices.map((c) => ({ ...toGenAiMessage(c.message), finish_reason: c.finish_reason })) + span.setAttribute("gen_ai.output.messages", JSON.stringify(output)) + } + return completion + }) +} + +// OpenAI chat messages -> the {role, parts} shape Maple renders as a transcript. Text and tool calls only. +function toGenAiMessage(message: OpenAI.Chat.ChatCompletionMessageParam | OpenAI.Chat.ChatCompletionMessage) { + if (message.role === "tool") { + return { role: "tool", parts: [{ type: "tool_call_response", id: message.tool_call_id, response: message.content }] } + } + const parts: object[] = [] + if (typeof message.content === "string" && message.content) parts.push({ type: "text", content: message.content }) + if (message.role === "assistant") { + for (const call of message.tool_calls ?? []) { + if (call.type === "function") { + parts.push({ type: "tool_call", id: call.id, name: call.function.name, arguments: call.function.arguments }) + } + } + } + return { role: message.role, parts } +} + +function withSpan(name: string, kind: SpanKind, attributes: Attributes, fn: (span: Span) => Promise) { + return tracer.startActiveSpan(name, { kind, attributes }, async (span) => { + try { + return await fn(span) + } catch (error) { + markFailed(span, error) + throw error + } finally { + span.end() + } + }) +} + +function markFailed(span: Span, error: unknown) { + const err = error instanceof Error ? error : new Error(String(error)) + span.recordException(err) + span.setStatus({ code: SpanStatusCode.ERROR, message: err.message }) + span.setAttribute("error.type", err.name) +} +``` + +```ts +// agent.ts +import "./instrumentation.ts" +import OpenAI from "openai" +import { agentSpan, runTool, tracedChat } from "./agent-tracing.ts" + +const client = new OpenAI() +const model = "gpt-4o-mini" +// tools: Record unknown> and toolSchemas: OpenAI.Chat.ChatCompletionTool[] are your own. + +/** One user message in, one reply out. `history` is this conversation's stored messages. */ +export function chatTurn( + conversationId: string, + history: OpenAI.Chat.ChatCompletionMessageParam[], + userText: string, + onText?: (delta: string) => void, +) { + return agentSpan("support_agent", conversationId, async () => { + history.push({ role: "user", content: userText }) + while (true) { + const completion = await tracedChat(client, { model, messages: history, tools: toolSchemas }, onText) + const message = completion.choices[0].message + history.push(message) + if (!message.tool_calls?.length) return message.content ?? "" + for (const call of message.tool_calls) { + if (call.type !== "function") continue + const result = await runTool(call.id, call.function.name, call.function.arguments, tools[call.function.name]) + history.push({ role: "tool", tool_call_id: call.id, content: result }) + } + } + }) +} +``` + +For `@anthropic-ai/sdk` or `@google/genai`, copy `tracedChat` and change what it reads. Keep each provider's raw token counts: Maple knows that Anthropic's `input_tokens` excludes cached tokens and Gemini's `candidatesTokenCount` excludes thinking tokens, and it picks the right arithmetic from `gen_ai.provider.name`. + +| Attribute | Anthropic Messages | Gemini `generateContent` | +| --- | --- | --- | +| `gen_ai.operation.name` | `chat` | `generate_content` | +| `gen_ai.provider.name` | `anthropic` | `gcp.gemini` (`gcp.vertex_ai` on Vertex) | +| `gen_ai.response.id` / `.model` | `id` / `model` | `responseId` / `modelVersion` | +| `gen_ai.response.finish_reasons` | `[stop_reason]` | `candidates[].finishReason` | +| `gen_ai.usage.input_tokens` | `usage.input_tokens` | `usageMetadata.promptTokenCount` | +| `gen_ai.usage.output_tokens` | `usage.output_tokens` | `usageMetadata.candidatesTokenCount` | +| `gen_ai.usage.cache_read.input_tokens` | `usage.cache_read_input_tokens` | `usageMetadata.cachedContentTokenCount` | +| `gen_ai.usage.cache_write.input_tokens` | `usage.cache_creation_input_tokens` | not reported | +| `gen_ai.usage.reasoning.output_tokens` | not reported separately | `usageMetadata.thoughtsTokenCount` | + +For messages, map text blocks to `{type: "text", content}`, tool calls (`tool_use`, `functionCall`) to `{type: "tool_call", id, name, arguments}` and tool results (`tool_result`, `functionResponse`) to `{type: "tool_call_response", id, response}`. Pass the system prompt as `gen_ai.system_instructions`, a JSON array of text parts. + +If you skip the turn span, each model call is its own trace and its own session, and tool calls float as separate one-span traces. If you keep the span but drop the id, you get one session per turn. + +## Record prompts, replies and tool calls + +The instrumentations record no message content by default. `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` takes four values: + +| Value | Where content goes | In Maple | +| --- | --- | --- | +| `NO_CONTENT` (default) | Nowhere | Turns, models, tokens and tools, with an empty transcript | +| `SPAN_ONLY` | Span attributes | Full transcript. Use this. | +| `EVENT_ONLY` | Log events | Empty transcript: Maple doesn't read logs | +| `SPAN_AND_EVENT` | Both | Full transcript, plus a second copy in your logs if you export them | + +The 1.x GenAI packages (July 2026 onward) don't need `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental`: the current GenAI conventions are the only ones they emit. Older guides that tell you to set it, or to set the capture variable to `true`, describe the deprecated `openai-v2` package, where `true` meant log events. + +The helpers above read the same variable, so tool arguments, tool results and (in TypeScript) messages follow one switch. + +Content capture sends everything your users type and everything the model answers. To keep it off in production, leave the variable unset there: you still get the session, turns, tools, tokens and failures. To keep content but drop secrets, redact in your code before the call, or run an OpenTelemetry Collector with a `redaction` or `transform` processor on `gen_ai.input.messages` and `gen_ai.output.messages`. Don't lower `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` to trim them: it cuts the JSON mid-string, and Maple drops a message attribute that doesn't parse. + +## Tools, errors and sub-agents + +Each `run_tool` call is an `execute_tool ` span with the tool name, the model's call id, the arguments and the result. The call id is what lets Maple pair the tool span with the tool call in the model's reply. + +Most agent loops catch a tool's exception and hand the error back to the model, which is right for the agent but hides the failure from tracing: the exception never escapes, so the span would end as a success. The helpers mark the span failed themselves (status `ERROR`, `error.type`, the exception recorded), and Maple counts it under the session's tool errors and on the Tools page, grouped by the error message. + +A sub-agent is another agent run inside a tool call. Call it through `run_tool` and wrap its loop in `agent_span` with its own name: + +```py +def ask_weather_worker(task: str) -> str: + # Runs inside the orchestrator's execute_tool span, in the same trace. + with agent_span("weather_worker"): + ... # its own model calls and tools +``` + +It doesn't need the conversation id, because it's in the orchestrator's trace. Maple opens a separate lane for each distinct `gen_ai.agent.name`, and an `execute_tool` span whose only child is an agent span shows as a delegation with the task and the answer. Two agents with the same name merge into one lane. + +If you run tools in parallel, `asyncio.gather` and `Promise.all` keep the trace context. A `ThreadPoolExecutor` doesn't: submit `contextvars.copy_context().run` with your function, or the tool spans start new traces outside the turn. + +## Tokens and cost + +The Python instrumentations record input and output tokens on every model call, plus cached input and reasoning tokens when the provider reports them. The TypeScript helper does the same for OpenAI. + +Streaming is the exception. OpenAI's Chat Completions API only sends usage on streamed calls when you ask for it, in a last chunk with no `choices`: + +```py +stream = client.chat.completions.create( + model=MODEL, messages=history, stream=True, stream_options={"include_usage": True} +) +reply = "".join(chunk.choices[0].delta.content or "" for chunk in stream if chunk.choices) +``` + +Without `stream_options`, the streamed call shows 0 tokens in Maple. Anthropic and Gemini always send usage on a stream. The Python instrumentations also record time to first chunk on streamed calls, which Maple shows per model call. + +None of these instrumentations record cost, and Maple doesn't price tokens itself, so sessions show as **unpriced**. In the TypeScript helper you own the span: if your gateway returns the cost (OpenRouter puts it in `usage.cost`), set it as `gen_ai.usage.cost` in USD. + +## Short-lived processes + +`BatchSpanProcessor` sends spans every 5 seconds, so the last turn of a script can still be in memory when the process exits. + +- **Python:** the `TracerProvider` flushes when the interpreter exits normally. In a serverless handler, a notebook or anything that ends with `os._exit`, call `provider.force_flush()` after each turn. +- **TypeScript:** call `await sdk.shutdown()` before a CLI or script exits. In a serverless handler, call `await spanProcessor.forceFlush()` before returning, or in `waitUntil` once a streamed response is sent. + +## Check that it works + +Run one conversation of two or three messages with at least one tool call, flush, and open **Agent Sessions** in Maple. Sessions appear within a few seconds of the export. + +- **One session per conversation**, with the conversation id you passed. The framework column says **Unidentified**, which is expected for this setup. +- **One turn per user message**, labeled with the first line of that message (with content capture on). Each turn is an `invoke_agent support_agent` span (your agent name) holding the rest. +- **Model calls** named `chat ` for OpenAI and Anthropic (`chat gpt-4o-mini`) or `generate_content ` for Gemini, with input and output tokens. +- **Tool calls** named `execute_tool get_weather` with their arguments and results, and a failing tool counted as a tool error. +- **A transcript** with the user messages, the replies and the tool calls between them. +- **Cost** shown as unpriced. + +## Troubleshooting + +- **Every model call is its own session.** The call ran outside `agent_span`, or the span was ended before the call. Wrap the whole turn, including the tool loop, in one `agent_span` with the conversation id. +- **One session per turn instead of per conversation.** The conversation id changes per request. Pass the id stored with the conversation, not a new UUID. +- **Turns show models and tokens but the transcript is empty.** Content capture is off, or set to `EVENT_ONLY` or `true`. Set `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` in the process that makes the calls. +- **No model spans in Python, only yours.** The instrumentor wasn't installed or `instrument()` ran after the first request. Check that `tracing` is imported first, and that you installed `opentelemetry-instrumentation-genai-openai`, not `opentelemetry-instrumentation-openai`. +- **Every model call appears twice.** Two instrumentations wrap the same SDK: the GenAI package plus OpenLLMetry, OpenInference, the deprecated `openai-v2` package, `logfire.instrument_openai()`, Sentry's OpenAI integration or a framework's own tracing. `opentelemetry-instrument` loads every instrumentation package that's installed, so uninstall the extras rather than just not calling them. +- **Twin traces for every call when you use OpenRouter.** OpenRouter Broadcast is also exporting the same calls. Keep one source; see the [OpenRouter guide](/docs/agent-tracing/openrouter). +- **The streamed turn has 0 tokens.** Add `stream_options={"include_usage": True}` (OpenAI Chat Completions). +- **A failing tool shows as successful.** The tool's exception was caught outside `run_tool`. Let `run_tool` catch it, or set the span status and `error.type` where you catch it. +- **Sub-agent calls land in the orchestrator's lane.** The sub-agent's `agent_span` has the same name as the orchestrator's, or it was never wrapped. Give each agent its own name. +- **Two conversation ids in one trace.** With the OpenAI Responses API and a `conversation` parameter, the instrumentation also sets `gen_ai.conversation.id` to that `conv_...` id. Pass the same id to `agent_span`, or Maple picks one of the two and can split the turn. +- **Cached Anthropic calls show more input tokens than you were billed for (Python).** The Anthropic GenAI instrumentation reports input tokens including cached ones, while Maple reads Anthropic's figure as excluding them, so cache reads count twice in the totals. Calls without prompt caching are unaffected. +- **Nothing arrives, and the exporter logs 401.** The key or region is wrong. EU keys only work with `ingest.eu.maple.dev`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Agent tracing guides](/docs/agent-tracing) +- [OpenRouter](/docs/agent-tracing/openrouter), if your calls go through OpenRouter +- [OpenTelemetry GenAI instrumentations for Python](https://github.com/open-telemetry/opentelemetry-python-genai) +- [OpenTelemetry GenAI semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/) diff --git a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md new file mode 100644 index 0000000000..945fb5e01d --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md @@ -0,0 +1,301 @@ +--- +title: "Trace Pydantic AI agents with OpenTelemetry" +description: "Send Pydantic AI's built-in OpenTelemetry spans to Maple so each conversation is one Agent Session with its transcript, tool calls, sub-agents and tokens, with or without Logfire." +group: "AI Agents" +order: 20 +navLabel: "Pydantic AI" +icon: "pydantic" +--- + +Pydantic AI has its own OpenTelemetry instrumentation, and it's good. Every `agent.run()` emits an `invoke_agent` span, a `chat` span per model request with the prompt, the reply and the token counts, and an `execute_tool` span per tool call with its arguments and result. All of it follows the GenAI semantic conventions Maple reads, so you don't need Logfire or an extra instrumentation package. You give Pydantic AI a `TracerProvider` that exports to Maple. + +The part that goes wrong is the conversation id. Every span carries `gen_ai.conversation.id`, so the traces look grouped, but unless you pass an id, Pydantic AI generates a new UUID7 for each run. A chat backend that handles one message per request gets one session per message, and an agent that delegates to other agents gets a different id for every delegate. + +This guide covers Pydantic AI 2.x on Python 3.10 or newer. It was tested with `pydantic-ai-slim` 2.51.0 and the OpenTelemetry Python SDK 1.45.0, and with Logfire 5.1.1 for the Logfire setup. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-pydantic-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-pydantic-ai) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Pydantic AI in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-pydantic-ai -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Export Pydantic AI spans to Maple + +Install the OpenTelemetry SDK and the OTLP/HTTP exporter next to Pydantic AI: + +```bash +pip install "pydantic-ai-slim[openai]>=2.51" "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Swap `[openai]` for the extras of the providers you use (`anthropic`, `google`, `openrouter`, ...). The full `pydantic-ai` package works the same way. + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +Then set up tracing once, when your process starts: + +```py +# tracing.py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor +from pydantic_ai import Agent, InstrumentationSettings + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +Agent.instrument_all( + InstrumentationSettings( + tracer_provider=provider, + include_content=True, + include_binary_content=False, + ) +) +``` + +Import `tracing` at the top of your entry point (`main.py`, the FastAPI app module, the worker). `Agent.instrument_all()` sets the default for every agent, including agents created before it runs, so the order relative to your agent modules doesn't matter. What matters is that it runs before the first `agent.run()`. + +If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Sentry or your own setup), don't create a second one. Add the `BatchSpanProcessor` above to the existing provider and call `Agent.instrument_all(InstrumentationSettings(include_content=True, include_binary_content=False))` without `tracer_provider`, so Pydantic AI uses the global one. + +Leave `version` at its default. Pydantic AI's instrumentation format is versioned separately from the package: 5 is the default, 2 to 4 are deprecated and emit a warning, and 6 is an opt-in that sends tool results with `role: "tool"`. Version 2 used different span names; this guide was tested with 5. + +### Already using Logfire + +Logfire sets up the same instrumentation and can export to any OTLP backend. It brings its own OpenTelemetry SDK and OTLP exporter, so skip the `pip install` line above: Logfire 5.1 requires `opentelemetry-sdk` below 1.45, and the `>=1.45` pins make the install fail to resolve. Keep the three `OTEL_EXPORTER_OTLP_*` variables and configure Logfire like this: + +```py +import logfire + +logfire.configure(service_name="support-agent", environment="production", send_to_logfire=False) +logfire.instrument_pydantic_ai() +``` + +Logfire adds an OTLP exporter whenever `OTEL_EXPORTER_OTLP_ENDPOINT` is set. With `send_to_logfire=True` (or a Logfire token in the environment), spans go to both Logfire and Maple. + +Logfire scrubs attributes by default, which changes what reaches Maple. The conversation id and the prompt and reply messages are exempt, but tool arguments, tool results and the run's `final_result` are not. A tool result that contains `session`, `auth`, `password`, `cookie` or `secret` anywhere in its text arrives as `[Scrubbed due to 'session']`. See [Record prompts, responses and tool calls](#record-prompts-responses-and-tool-calls) for how to keep it. + +## Group every turn of a conversation into one session + +Maple groups traces into sessions by `gen_ai.conversation.id`. Pydantic AI puts it on every span it emits, and picks the value in this order: + +1. the `conversation_id=` you pass to `run()`, `run_stream()` or `iter()`; +2. the id stamped on the last message of `message_history`; +3. a new UUID7. + +So a chat backend that stores the message history and passes it back keeps one id, as long as the first run of the conversation got one. A backend that starts each request without history, or that rebuilds the history from its own database, gets a new session for every message. Pass your own id on every run, from the chat or thread id your app already has: + +```py +from pydantic_ai import Agent + +support = Agent("openai:gpt-4o-mini", name="support") + + +async def handle_message(chat_id: str, text: str, history: list) -> str: + result = await support.run(text, conversation_id=chat_id, message_history=history) + return result.output +``` + +The id must be stable for the whole conversation and different between conversations. A per-process constant merges every user into one session. + +If you skip this, each request shows up in **Agent Sessions** as its own one-turn session, named after a UUID. + +Streaming works the same way. Pass `conversation_id=` to `run_stream()`, and keep the `async with` block open until the stream is finished, because the `invoke_agent` span ends when the block exits: + +```py +async def stream_reply(chat_id: str, text: str, history: list): + async with support.run_stream(text, conversation_id=chat_id, message_history=history) as run: + async for delta in run.stream_text(delta=True): + yield delta +``` + +## Record prompts, responses and tool calls + +Content capture is on by default in Pydantic AI (`include_content=True`). Each `chat` span carries `gen_ai.input.messages`, `gen_ai.output.messages` and `gen_ai.system_instructions` as JSON in the GenAI message format, and each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Maple builds the transcript from those attributes. + +Two settings are worth changing: + +- `include_binary_content=False` keeps images, audio and documents out of the spans. With it on, a single uploaded PDF can put megabytes of base64 into every later `chat` span of that run, since each request repeats the history. +- `include_content=False` drops prompts, replies, tool arguments and tool results. The messages keep their roles and part types, so the transcript shows the shape of the conversation with no text. Turns, models, tool names, tokens and errors are unaffected. Exception messages are dropped too; only the exception type is kept. + +To turn content off for one agent only, set it on that agent: + +```py +from pydantic_ai import Agent, InstrumentationSettings +from pydantic_ai.capabilities import Instrumentation + +billing = Agent( + "openai:gpt-4o-mini", + name="billing", + capabilities=[Instrumentation(settings=InstrumentationSettings(include_content=False))], +) +``` + +An agent with its own `Instrumentation` capability ignores `Agent.instrument_all()`, and without `tracer_provider=` it uses the global provider you set in `tracing.py`. + +Logfire's scrubbing does not protect prompts. The message attributes are exempt from it, so a user who types a password into the chat sends it to Maple either way. If you need pattern-based redaction of message content, run it in an OpenTelemetry Collector between your app and Maple. + +With Logfire, keep tool content intact by letting those attributes through the scrubber: + +```py +import logfire + +TOOL_CONTENT = {"gen_ai.tool.call.arguments", "gen_ai.tool.call.result", "final_result"} + + +def keep_tool_content(match: logfire.ScrubMatch): + if len(match.path) > 1 and match.path[1] in TOOL_CONTENT: + return match.value # keep; returning None redacts + + +logfire.configure( + service_name="support-agent", + send_to_logfire=False, + scrubbing=logfire.ScrubbingOptions(callback=keep_tool_content), +) +``` + +## Tools, errors and sub-agents + +Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the provider's `gen_ai.tool.call.id`, and the arguments and result. Maple matches each call to the model reply that requested it by that id. + +How a tool fails decides what Maple shows: + +| In your tool | Span status | Run continues | In Maple | +| --- | --- | --- | --- | +| `raise ToolFailed("...")` | ERROR | yes, the model sees the message | failed call, message as its result | +| `raise ModelRetry("...")` | ERROR | yes, the model retries | one failed call per retry | +| any other exception | ERROR | no, the run raises | failed call and failed turn | +| `return {"error": "..."}` | UNSET | yes | successful call | + +For an upstream error the model should work around, raise `ToolFailed` (Pydantic AI 2.16 or newer). It marks the span failed, records the message as the tool result, and doesn't use up the tool's retry budget: + +```py +from pydantic_ai import ToolFailed + + +@support.tool_plain +def fetch_transport_data(city: str) -> dict: + """Fetch live public-transport data for a city.""" + raise ToolFailed("transport data service unavailable (503)") +``` + +Returning an error payload keeps the span green, and Maple counts the call as a success. + +Approval-gated tools (`requires_approval=True`) pause the run without a tool span. The call only gets an `execute_tool` span when the resumed run executes it, so pass the same `conversation_id=` to the run that sends `DeferredToolResults`. + +### Sub-agents: pass the conversation id down + +The common multi-agent pattern in Pydantic AI is delegation: a tool on the orchestrator calls `worker.run()`. The worker's run is nested in the orchestrator's trace, under the tool span, but it resolves its own conversation id, and with no history and no explicit id, that is a new UUID7. One trace then carries several ids. Maple uses one of them for the whole trace, and not necessarily yours, and it can split the turn into one turn per id. + +Pass the caller's id and usage to every delegate: + +```py +from pydantic_ai import Agent, RunContext + +weather_worker = Agent("openai:gpt-4o-mini", name="weather_worker", tools=[get_weather]) +orchestrator = Agent("openai:gpt-4o-mini", name="orchestrator") + + +@orchestrator.tool +async def research_weather(ctx: RunContext[None], city: str) -> str: + """Delegate to the weather worker.""" + result = await weather_worker.run( + f"What is the current weather in {city}?", + usage=ctx.usage, + conversation_id=ctx.conversation_id, + ) + return result.output +``` + +Give every agent an explicit `name=`. It becomes `gen_ai.agent.name` on all of its spans, and Maple draws a lane per agent name. An `execute_tool research_weather` span whose only child is `invoke_agent weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. When several delegation tools are called in one model reply, Pydantic AI runs them concurrently and the lanes overlap in time. + +A pipeline of separate top-level runs (orchestrator, then a summary agent) produces one trace per run. With the same `conversation_id=` on each, they land in one session as consecutive turns. + +## Tokens and cost + +Each `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. Maple sums them per model call. + +One exception: with Anthropic models on Pydantic AI's `anthropic` provider and prompt caching, Maple currently counts cached input tokens twice. Pydantic AI reports input tokens including the cache for every provider, and Maple applies Anthropic's own convention, where they're separate. OpenAI, OpenRouter and Gemini calls aren't affected. + +The `invoke_agent` span reports the run's own total under `gen_ai.aggregated_usage.*`, which Maple doesn't add to the session total, so nothing is counted twice. A delegate's tokens stay on the delegate's spans. + +Streaming needs nothing extra. Pydantic AI requests usage on OpenAI-compatible streams (`stream_options.include_usage`), so streamed calls have token counts too. The streamed `chat` span also records time to first chunk, under a key Maple doesn't read yet. + +Cost shows as unpriced. Pydantic AI prices each call and writes the result to `operation.cost` on the `chat` span, but Maple reads cost only from `gen_ai.usage.cost` and never prices tokens itself. Tokens, models and call counts are complete. + +## Short-lived processes + +`BatchSpanProcessor` exports every few seconds, and the SDK flushes on a normal interpreter exit. That doesn't cover a Lambda that freezes after the handler returns, a worker killed by its supervisor, `os._exit()`, or a notebook kernel that never exits. Flush explicitly in those cases: + +```py +import asyncio + +from opentelemetry import trace + + +def handler(event, context): + try: + return asyncio.run(handle_message(event["chat_id"], event["text"], [])) + finally: + trace.get_tracer_provider().force_flush() +``` + +In a script or CLI, call `provider.shutdown()` at the end. With Logfire, use `logfire.force_flush()` and `logfire.shutdown()`. + +## Check that it works + +Before you look in Maple, check the console. Pydantic AI prints an `observability: off` banner on the first run when no instrumentation is set up. If it's still there, `tracing.py` didn't run before your first `agent.run()`. + +Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: + +- one session per conversation, with the id you passed as `conversation_id`, and framework **Pydantic AI**; +- one turn per `run()`, each labeled with the user's message, and a transcript with the prompts, replies and tool calls; +- `chat ` spans for model calls, `execute_tool ` spans for tool calls, and `invoke_agent ` spans for runs; +- a lane per sub-agent, named after its `name=`; +- token counts on every model call, including streamed ones; +- failed tool calls marked as failed, with the `ToolFailed` message as the result; +- cost shown as unpriced. + +## Troubleshooting + +- **Every message is its own session.** No `conversation_id=` was passed and the history didn't carry one. Pass it on every `run()`, `run_stream()` and `iter()`. +- **A multi-agent run shows several turns, or lands in another session.** Delegates minted their own ids. Pass `conversation_id=ctx.conversation_id` to every nested `run()`. +- **All sub-agents share one lane called `agent`.** The agents have no `name=`. Set one on each. +- **Nothing arrives from a script or Lambda.** The process ended before the batch was exported. Call `force_flush()` or `shutdown()` in a `finally`. +- **The transcript has messages but no text.** `include_content=False` is set, in `instrument_all()` or in that agent's `Instrumentation` capability. +- **Tool arguments or results read `[Scrubbed due to ...]`.** Logfire's scrubbing matched a word in them. Add a scrubbing callback or set `scrubbing=False`. +- **A failed tool shows as successful.** The tool returned an error value instead of raising. Raise `ToolFailed`. +- **`chat` spans are huge or exports fail with 413.** Images or documents are being recorded as base64. Set `include_binary_content=False`. +- **Spans show up twice.** Pydantic AI and a second instrumentor (Logfire's `instrument_openai()`, OpenInference, OpenLLMetry) both trace the same model calls. Keep Pydantic AI's instrumentation and remove the other one for the model client. +- **The install fails to resolve `logfire` and `opentelemetry-sdk`.** Logfire 5.1 pins the SDK below 1.45. On the Logfire path, don't add the OpenTelemetry packages yourself; Logfire installs them. +- **A `PydanticAIDeprecationWarning` about instrumentation versions 2, 3 and 4.** Remove `version=` from `InstrumentationSettings` to use the default. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [All agent tracing guides](/docs/agent-tracing) +- [Pydantic AI: debugging and monitoring with OpenTelemetry](https://pydantic.dev/docs/ai/integrations/logfire/) +- [Instrument a Python application](/docs/guides/instrumentation-python) diff --git a/apps/landing/src/content/docs/agent-tracing/smolagents.md b/apps/landing/src/content/docs/agent-tracing/smolagents.md new file mode 100644 index 0000000000..c054c38461 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/smolagents.md @@ -0,0 +1,233 @@ +--- +title: "Trace Hugging Face smolagents with OpenTelemetry" +description: "Send smolagents runs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools and managed agents in their own lanes." +group: "AI Agents" +order: 25 +navLabel: "smolagents" +icon: "huggingface" +--- + +smolagents has no OpenTelemetry code of its own. Every span comes from OpenInference's `openinference-instrumentation-smolagents`, which patches `MultiStepAgent.run`, each agent step, every model's `generate` and `Tool.__call__`. By default those spans use OpenInference attribute names (`llm.input_messages.0.message.role`, `llm.token_count.prompt`), and every `agent.run()` starts a new trace with no conversation id. + +Two defaults have to change for Maple. Maple's session page reads the OpenTelemetry GenAI attributes (`gen_ai.*`) for smolagents, not OpenInference's own, so without the instrumentor's GenAI dual-write the session list shows token counts and the session page shows no transcript. And without `using_session(...)` around each run, a ten-message chat becomes ten one-turn sessions. This guide covers smolagents 1.26 with `openinference-instrumentation-smolagents` 0.1.40 on Python 3.10 or later, for both `ToolCallingAgent` and `CodeAgent`. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-smolagents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-smolagents) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for smolagents in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-smolagents -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install the instrumentor and export to Maple + +```bash +pip install "smolagents[openai]>=1.26" "openinference-instrumentation-smolagents>=0.1.40" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Skip the `smolagents[telemetry]` extra from the smolagents docs. It also installs the Arize Phoenix server, which you don't need to send traces to Maple. The `[openai]` extra is for `OpenAIServerModel`; use `[litellm]` if you run `LiteLLMModel`. + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_SERVICE_NAME=support-agent +export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +``` + +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL, not a signal path: `OTLPSpanExporter()` with no arguments appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. If you pass `endpoint=` in code instead, it's used as is, so it has to end in `/v1/traces` or every export gets a 404. + +Then add a `tracing.py` and import it at the top of your entry point: + +```py +# tracing.py +import json + +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.smolagents import SmolagentsInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class SmolagentsForMaple(SpanProcessor): + """Fixes what the smolagents instrumentor gets wrong for Maple: agent names, run token totals, tool names and arguments.""" + + def on_start(self, span, parent_context=None): + if span.instrumentation_scope.name != "openinference.instrumentation.smolagents": + return + attrs = span.attributes + if span.name.endswith(".run"): + # "weather_worker.run" -> gen_ai.agent.name "weather_worker", so sub-agents get lanes + span.set_attribute("gen_ai.agent.name", span.name.removesuffix(".run")) + # The run's token totals repeat its model calls (with reset=False, every earlier turn's too) + span.set_attribute("gen_ai.usage.input_tokens", 0) + span.set_attribute("gen_ai.usage.output_tokens", 0) + elif "tool.name" in attrs: + # Every @tool span is named "SimpleTool"; name it after the tool instead + span.update_name(f"execute_tool {attrs['tool.name']}") + # The GenAI dual-write copies the tool's input schema here; record the call's arguments + if attrs.get("input.value", "").startswith("{"): + call = json.loads(attrs["input.value"]) + span.set_attribute("gen_ai.tool.call.arguments", json.dumps(call["kwargs"] or call["args"])) + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(SmolagentsForMaple()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +SmolagentsInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +What each part does: + +- **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*` and `gen_ai.tool.*` next to its OpenInference attributes when each span ends. The environment variable `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, but only if it's set before `TraceConfig` is built; passing it in code avoids that ordering trap. +- **`SmolagentsForMaple`** fixes four things in the instrumentor's output: it names each agent, zeroes the token totals on run spans (see [Tokens and cost](#tokens-and-cost)), names tool spans after the tool, and records the call's arguments instead of the tool's schema. It runs in `on_start`, before the dual-write, and the dual-write never overwrites a key that's already set. + +The instrumentor patches smolagents' classes in place, so import order doesn't matter, as long as `instrument()` runs before the first `agent.run()`. It only patches the model classes smolagents exports. A `Model` subclass of your own that overrides `generate` produces no model spans. + +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), don't create a second one. Add `SmolagentsForMaple()` and the OTLP exporter to the existing provider and pass that provider to `instrument()`. + +## Group a conversation into one session + +smolagents has no session, thread or conversation id. Memory lives on the agent object: `agent.run(task, reset=False)` continues the previous conversation, and every call is a new root span and a new trace. Maple groups traces into a session by the `session.id` attribute, and the instrumentor only sets it inside OpenInference's `using_session` context manager: + +```py +from openinference.instrumentation import using_session +from smolagents import OpenAIServerModel, ToolCallingAgent + +agents: dict[str, ToolCallingAgent] = {} + + +def handle_message(conversation_id: str, text: str) -> str: + agent = agents.get(conversation_id) + if agent is None: + agent = agents[conversation_id] = ToolCallingAgent( + tools=[get_weather, calculate], + model=OpenAIServerModel(model_id="gpt-4o-mini"), + name="assistant", + ) + with using_session(conversation_id): + return str(agent.run(text, reset=False)) +``` + +Use your app's own conversation id, the one it already stores the chat under. A new UUID per request gives you one session per message again, and a constant gives every user one shared session. + +`using_session` stores the id in a Python contextvar, not in OpenTelemetry baggage, and the instrumentor copies it onto every span it creates inside the block, including the model and tool spans of managed agents and of tool calls running in parallel threads. It does not reach spans you create with a plain OpenTelemetry tracer. If you add your own spans, pass `dict(get_attributes_from_context())` (also from `openinference.instrumentation`) as their attributes. `using_attributes(session_id=..., user_id=...)` works the same way and also sets `user.id`. + +If you skip this, every `agent.run()` shows up in **Agent Sessions** as its own one-turn session named after its trace id. Setting `gen_ai.conversation.id` yourself doesn't help: Maple reads `session.id` for smolagents. + +One agent object per conversation matters for more than tracing. A single module-level agent with `reset=False` shares its memory across every user who talks to it, and its traces would look like one long conversation. + +## Record prompts, responses and tool calls + +Content capture is on by default. Every model span carries the full message list sent to the model (system prompt, the task, earlier steps, tool results) and the model's reply, and every tool span carries the tool's arguments and result. With the GenAI dual-write on, Maple renders these as the session transcript. + +That full message list is large. smolagents' default system prompt is about 3,100 characters for `ToolCallingAgent` and 8,500 for `CodeAgent`, and with `reset=False` every model span repeats the whole conversation so far. Maple has no per-attribute limit, and ingest accepts requests up to 20 MiB. + +To keep prompts and outputs out of your traces: + +```py +config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) +``` + +`hide_inputs` drops the input messages and replaces `input.value` with `__REDACTED__`; `hide_outputs` does the same for outputs. The session still shows its turns, model and tool calls, tokens and failures, with an empty transcript. Narrower switches exist: `hide_input_text` and `hide_output_text` keep the message structure but redact the text, and `hide_llm_invocation_parameters` drops temperature and max tokens. Each has an `OPENINFERENCE_HIDE_*` environment variable. + +These switches don't cover everything. The run span's `smolagents.task` attribute holds the previous `agent.run()` task in plain text, and the instrumentor doesn't mask it. If prompts can contain personal data, drop `smolagents.task` in an OpenTelemetry Collector with the `attributes` processor, or redact there with the `redaction` processor. + +## Tools, errors and managed agents + +Each tool call is a span of OpenInference kind `TOOL`, with `gen_ai.operation.name` `execute_tool`, the tool's name in `gen_ai.tool.name` and its return value in `gen_ai.tool.call.result`. smolagents' own `final_answer` is a tool too, so every run that finishes normally ends with an `execute_tool final_answer` span. + +A tool that raises is marked failed without extra code. The instrumentor ends the tool span with status `ERROR` and the exception as the status message, for example `RuntimeError: transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. The enclosing `Step N` span is also marked `ERROR`, with the wrapped `AgentToolExecutionError`, and carries two `exception` events for one failure. + +smolagents feeds the error back to the model with "Now let's retry: take care not to repeat previous errors!", so a tool that's down stays down until `max_steps`. You'll see the same failed tool three or four times in one turn, and after `max_steps` the model is asked to answer without tools, which is where it tends to make data up. A `step_callbacks` hook that tells the model to stop after the first failure keeps both the trace and the answer honest. + +With `ToolCallingAgent`, a model reply that calls `final_answer` together with another tool raises `AgentExecutionError`, and that step shows as a failed `Step` span even though the run continues. That's the framework rejecting the model's output, not your code failing. + +Managed agents (`managed_agents=[...]`) show up as their own `.run` spans under the manager's step, with the managed agent's steps, model calls and tools inside. There's no tool span around the delegation. The `SmolagentsForMaple` processor turns the span name into `gen_ai.agent.name`, and Maple opens a lane for each agent whose name differs from its caller's. Give every agent a `name`; an unnamed agent's span is `ToolCallingAgent.run` or `CodeAgent.run`, and two unnamed agents share one lane. + +When a `ToolCallingAgent` model asks for several tools or managed agents in one reply, smolagents runs them in a thread pool (`max_tool_threads`) and copies the context into each thread, so parallel workers stay in the same trace and session. + +`CodeAgent` calls tools from the Python code the model writes, run by `LocalPythonExecutor` in your process. Those calls produce the same tool spans; positional arguments are recorded as a JSON array, like `["Berlin"]`. With a remote executor (`executor_type="e2b"`, `"docker"`, `"modal"` and others), the code and its tool calls run outside your process, so there are no tool spans, only the model and step spans. + +Two gaps remain in what Maple can show for smolagents tools. Tool spans have no `gen_ai.tool.call.id`, because smolagents doesn't pass the model's tool call id to the tool, so Maple can't link a tool span to the exact call in the model's reply. And the step's failure and the tool's failure are both on the trace, so a session with one broken tool has two failed spans. + +## Tokens and cost + +Every model span carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals. The instrumentor doesn't record cached or reasoning tokens, even when the provider reports them. The model is the id you passed, `gen_ai.request.model` `gpt-4o-mini` or `openai/gpt-4o-mini`; smolagents doesn't record the model name the provider returns. + +The provider comes from the model class, not the model: `OpenAIServerModel` is always `openai`, even for an Anthropic model behind OpenRouter. The model span is named after the class too: `OpenAIServerModel` is an alias of `OpenAIModel`, so its spans are `OpenAIModel.generate`. + +Streaming (`stream_outputs=True` on the agent) produces `OpenAIModel.generate_stream` spans. smolagents requests `stream_options={"include_usage": True}` for `OpenAIServerModel`, `LiteLLMModel` and `InferenceClientModel`, so streamed calls from those keep their token counts. + +The instrumentor also copies the agent monitor's token totals onto each `.run` span. Those totals repeat the model spans below it, and with `reset=False` the monitor is never reset, so turn four's run span carries the tokens of turns one to four. Maple can't net them against the model calls, because a `Step N` span sits in between. In our test, a four-turn conversation with 7,675 input tokens of model calls had another 15,626 on its run spans. `SmolagentsForMaple` sets `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` to 0 on run spans. Maple reads those before OpenInference's `llm.token_count.*`, so each model call is counted once, and the OpenInference totals stay on the span for other tools. + +Maple shows cost only when a span carries one, and smolagents never records cost. Sessions show as **unpriced**, with token counts. + +Don't add `openinference-instrumentation-openai` or `openinference-instrumentation-litellm` next to the smolagents instrumentor. Neither checks for an existing model span, so every `OpenAIModel.generate` or `LiteLLMModel.generate` span gets a second model span under it for the same request. + +## Flush spans before the process exits + +`BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` registers an `atexit` handler that flushes on a normal interpreter exit, which covers most scripts and CLIs. It doesn't run when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: + +```py +from tracing import provider + +try: + handle_message("conv-42", "What's the weather in Berlin?") +finally: + provider.force_flush() # serverless: before returning; notebooks: after each run +``` + +Call `provider.shutdown()` instead of `force_flush()` when the process is about to exit and won't trace anything else. + +## Check that it works + +Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: + +- **One session** for the conversation, framework **smolagents**, with one turn per `agent.run()`. Turn traces start at `assistant.run` (your agent's name). +- **The transcript**: your messages and the model's replies. smolagents sends each task to the model as `New task:` followed by your text, and tool results come back as `tool-response` messages. +- **Model calls** named `OpenAIModel.generate` (or `LiteLLMModel.generate`, `InferenceClientModel.generate`), each with a model, input and output tokens. +- **Tool calls** named `execute_tool get_weather` and `execute_tool final_answer`, with arguments and results. +- **Agents**: `assistant`, plus one lane per managed agent if you use them. +- **Cost**: unpriced. + +A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. + +## Troubleshooting + +- **No spans at all.** `instrument()` never ran, or ran after the agent was used. Import `tracing` first in the entry point and look for an `OTLPSpanExporter` error in the logs. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. +- **Tokens in the list, empty session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)`, or set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` before `instrument()` runs. +- **One session per message.** The run isn't inside `using_session(...)`, or the id changes per request. Wrap every `agent.run()` and use the stored conversation id. +- **Two users in one session.** A shared agent object with `reset=False`, or a constant session id. Keep one agent and one id per conversation. +- **Every tool span is named `SimpleTool`.** `@tool` functions all become instances of a class called `SimpleTool`, and the instrumentor names tool spans after the class. `SmolagentsForMaple` renames them; `gen_ai.tool.name` has the real name either way. +- **Tool arguments show the tool's input schema.** The dual-write copies `tool.parameters`, the schema, into `gen_ai.tool.call.arguments`. `SmolagentsForMaple` replaces it with the call's arguments. +- **No lanes for managed agents.** The instrumentor puts the agent's name only in the span name. Add `SmolagentsForMaple` and give each agent a `name`. +- **A failing tool appears three or four times.** smolagents tells the model to retry after every error, up to `max_steps`. Stop it with a `step_callbacks` hook or a lower `max_steps`. +- **Session tokens are about double the model calls, or grow faster every turn.** The run spans' token totals are being counted. Keep the two zero-token lines for `.run` spans in `SmolagentsForMaple`. +- **Every model call appears twice.** A provider instrumentor (`openinference-instrumentation-openai` or `-litellm`) is also installed. Remove it. +- **No tool spans with `CodeAgent`.** A remote executor runs the generated code outside your process. Only `LocalPythonExecutor` produces tool spans. +- **A custom model class has no model spans.** The instrumentor only patches the model classes smolagents exports. Subclass one of them without overriding `generate`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. +- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. +- [Inspecting runs with OpenTelemetry](https://huggingface.co/docs/smolagents/tutorials/inspect_runs): the smolagents docs page on tracing. +- [openinference-instrumentation-smolagents](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-smolagents): the instrumentor's source. +- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/spring-ai.md b/apps/landing/src/content/docs/agent-tracing/spring-ai.md new file mode 100644 index 0000000000..bf99ca2512 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/spring-ai.md @@ -0,0 +1,421 @@ +--- +title: "Trace Spring AI agents with OpenTelemetry" +description: "Send Spring AI ChatClient, model and tool spans to Maple with every turn sampled, one session per chat memory conversation, the transcript on the spans and failed tools marked as failed." +group: "AI Agents" +order: 31 +navLabel: "Spring AI" +icon: "spring" +--- + +Spring AI instruments itself with Micrometer Observations. Add Spring Boot's OpenTelemetry starter and every `ChatClient` call becomes a `spring_ai chat_client` span, with a `chat ` span per model call and an `execute_tool ` span per tool call. The model spans follow the OpenTelemetry GenAI conventions (model, token counts, cache tokens, finish reasons, response id), and Maple recognizes all of it as Spring AI without a separate instrumentation library. + +Four defaults work against you. Spring Boot samples 10% of traces, so nine turns out of ten never arrive. Prompts and replies are never written to spans: `log-prompt` and `log-completion` send them to the application log. A tool that throws ends its span as a success, because Spring AI hands the error message back to the model. And the advisor spans (`tool _calling `, `message_chat_memory`) have names Maple reads as extra tool and model calls. This guide fixes all four with a handful of properties and one configuration class. It covers Spring AI 2.0 (tested on 2.0.1) on Spring Boot 4.1 (4.1.1) and Java 21, with notes for Spring AI 1.1 on Boot 3.5. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-spring-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Spring AI in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-spring-ai -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install the OpenTelemetry starter and export to Maple + +Spring AI 2.0 requires Spring Boot 4. Import the Spring AI BOM and add your model starter, Boot's OpenTelemetry starter and Actuator: + +```xml + + + + org.springframework.ai + spring-ai-bom + 2.0.1 + pom + import + + + + + + + org.springframework.ai + spring-ai-starter-model-openai + + + org.springframework.boot + spring-boot-starter-opentelemetry + + + org.springframework.boot + spring-boot-starter-actuator + + +``` + +`spring-boot-starter-opentelemetry` brings the Micrometer-to-OpenTelemetry tracing bridge, the OpenTelemetry SDK and the OTLP exporter. With Gradle, use the same three artifacts and `platform("org.springframework.ai:spring-ai-bom:2.0.1")`. + +Then point the exporter at Maple in `application.properties`: + +```properties +spring.application.name=support-agent + +management.opentelemetry.tracing.export.otlp.endpoint=https://ingest.maple.dev/v1/traces +management.opentelemetry.tracing.export.otlp.headers.Authorization=Bearer ${MAPLE_INGEST_KEY} +management.tracing.sampling.probability=1.0 +management.opentelemetry.resource-attributes.deployment.environment.name=production + +# The starter also exports metrics, to localhost:4318 unless told otherwise +management.otlp.metrics.export.url=https://ingest.maple.dev/v1/metrics +management.otlp.metrics.export.headers.Authorization=Bearer ${MAPLE_INGEST_KEY} + +# Read by the configuration class below +maple.ai.capture-content=true +# Token usage on streamed OpenAI calls +spring.ai.openai.chat.stream-options.include-usage=true +``` + +EU organizations use `https://ingest.eu.maple.dev`. Unlike the `OTEL_EXPORTER_OTLP_ENDPOINT` variable, this property takes the full URL, so keep `/v1/traces` on the end. The transport defaults to OTLP over HTTP with protobuf, which is what Maple ingest expects. `spring.application.name` becomes `service.name`. + +`management.tracing.sampling.probability=1.0` is the line that matters most. Boot's default is `0.1`, and a sampled-out turn leaves a hole in the session with no error anywhere. If you'd rather not send metrics, replace the two metrics lines with `management.otlp.metrics.export.enabled=false`. + +On Boot 4.1 you can configure the exporter with the standard variables instead: `OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev` and `OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer YOUR_INGEST_KEY` cover traces, metrics and logs, and Boot appends the signal path itself. Keep the sampling property either way. + +## Add the attributes Maple reads + +Spring AI's spans get you model calls and tokens. The transcript, tool names, failed tools and agent names need one configuration class. It uses three standard Micrometer and Spring AI extension points, so no Spring AI class is patched or replaced: + +```java +package com.example.agent; + +import java.util.ArrayList; +import java.util.List; +import java.util.Map; + +import io.micrometer.common.KeyValue; +import io.micrometer.observation.Observation; +import io.micrometer.observation.ObservationFilter; +import io.micrometer.observation.ObservationPredicate; +import io.micrometer.observation.ObservationRegistry; +import tools.jackson.databind.json.JsonMapper; + +import org.springframework.ai.chat.client.advisor.observation.AdvisorObservationContext; +import org.springframework.ai.chat.client.observation.ChatClientObservationContext; +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.messages.Message; +import org.springframework.ai.chat.messages.ToolResponseMessage; +import org.springframework.ai.chat.observation.ChatModelObservationContext; +import org.springframework.ai.tool.execution.DefaultToolExecutionExceptionProcessor; +import org.springframework.ai.tool.execution.ToolExecutionExceptionProcessor; +import org.springframework.ai.tool.observation.ToolCallingObservationContext; +import org.springframework.beans.factory.annotation.Value; +import org.springframework.context.annotation.Bean; +import org.springframework.context.annotation.Configuration; + +/** Adds the gen_ai.* attributes Maple reads on top of Spring AI's own observations. */ +@Configuration(proxyBeanMethods = false) +public class MapleAiObservationConfig { + + /** Agent name for a ChatClient: `.defaultAdvisors(a -> a.param(AGENT_NAME, "support_agent"))`. */ + public static final String AGENT_NAME = "gen_ai.agent.name"; + + // Advisor spans carry nothing Maple reads, and their names ("tool _calling ", + // "message_chat_memory") would be counted as extra tool and LLM calls. + @Bean + ObservationPredicate skipAdvisorObservations() { + return (name, context) -> !(context instanceof AdvisorObservationContext); + } + + @Bean + ObservationFilter mapleGenAiAttributes(@Value("${maple.ai.capture-content:false}") boolean captureContent) { + return context -> { + if (context instanceof ChatClientObservationContext client) { + // One ChatClient call is one agent turn; Spring AI labels it "framework". + client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.operation.name", "invoke_agent")); + if (client.getRequest().context().get(AGENT_NAME) instanceof String agent) { + client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.agent.name", agent)); + } + } + else if (context instanceof ToolCallingObservationContext tool) { + tool.addLowCardinalityKeyValue(KeyValue.of("gen_ai.tool.name", tool.getToolDefinition().name())); + if (tool.getToolCallId() != null) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.id", tool.getToolCallId())); + } + if (captureContent) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.arguments", tool.getToolCallArguments())); + if (tool.getToolCallResult() != null) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.result", tool.getToolCallResult())); + } + } + } + else if (captureContent && context instanceof ChatModelObservationContext chat) { + chat.addHighCardinalityKeyValue(KeyValue.of("gen_ai.input.messages", + JsonMapper.shared().writeValueAsString(chat.getRequest().getInstructions().stream().map(MapleAiObservationConfig::message).toList()))); + if (chat.getResponse() != null) { + chat.addHighCardinalityKeyValue(KeyValue.of("gen_ai.output.messages", + JsonMapper.shared().writeValueAsString(chat.getResponse().getResults().stream().map(g -> message(g.getOutput())).toList()))); + } + } + return context; + }; + } + + // Spring AI hands a failing tool's message back to the model and ends the tool span OK. + // Mark the span failed first, then keep the default behaviour. + @Bean + ToolExecutionExceptionProcessor toolExecutionExceptionProcessor(ObservationRegistry registry) { + ToolExecutionExceptionProcessor fallback = DefaultToolExecutionExceptionProcessor.builder().build(); + return exception -> { + Observation toolCall = registry.getCurrentObservation(); + if (toolCall != null) { + toolCall.error(exception); + } + return fallback.process(exception); + }; + } + + /** One message in the OpenTelemetry GenAI shape: {role, parts: [...]}. */ + private static Map message(Message message) { + List> parts = new ArrayList<>(); + if (message instanceof ToolResponseMessage toolResponse) { + toolResponse.getResponses().forEach(r -> parts.add( + Map.of("type", "tool_call_response", "id", r.id(), "response", r.responseData()))); + } + else { + if (message.getText() != null && !message.getText().isEmpty()) { + parts.add(Map.of("type", "text", "content", message.getText())); + } + if (message instanceof AssistantMessage assistant) { + assistant.getToolCalls().forEach(call -> parts.add( + Map.of("type", "tool_call", "id", call.id(), "name", call.name(), "arguments", call.arguments()))); + } + } + return Map.of("role", message.getMessageType().getValue(), "parts", parts); + } + +} +``` + +What each bean does: + +- **`ObservationPredicate`** drops the advisor observations. Their spans hold only an advisor name and order, and Maple classifies spans without a known operation by name: `tool _calling ` would count as a tool call and `message_chat_memory` as a model call on every turn. Child spans still attach to the `chat_client` span, because Micrometer keeps the scope of a skipped observation. +- **`ObservationFilter`** runs when each observation stops, after Spring AI's own conventions. It relabels the `chat_client` span as `invoke_agent` (Spring AI calls it `framework`, which Maple would otherwise count as a model call because the span name contains "chat"), copies the tool name and call id to the `gen_ai.tool.*` keys, and writes the conversation as `gen_ai.input.messages` and `gen_ai.output.messages`. +- **`ToolExecutionExceptionProcessor`** marks the running tool span as failed before Spring AI turns the exception into a message for the model. See [Tools, errors and sub-agents](#tools-errors-and-sub-agents). + +Boot applies `ObservationPredicate` and `ObservationFilter` beans to the observation registry by itself; there is nothing else to register. The class works the same in Kotlin; for example, the predicate is `ObservationPredicate { _, context -> context !is AdvisorObservationContext }`. + +### Spring AI 1.1 on Spring Boot 3.5 + +The same approach works, with these differences: + +| | Spring AI 2.0, Boot 4 | Spring AI 1.1, Boot 3.5 | +|---|---|---| +| Tracing dependencies | `spring-boot-starter-opentelemetry` | `io.micrometer:micrometer-tracing-bridge-otel` and `io.opentelemetry:opentelemetry-exporter-otlp` | +| Endpoint property | `management.opentelemetry.tracing.export.otlp.endpoint` | `management.otlp.tracing.endpoint` | +| Header property | `management.opentelemetry.tracing.export.otlp.headers.*` | `management.otlp.tracing.headers.*` | +| JSON in the filter | Jackson 3 `JsonMapper.shared()` | Jackson 2 `ObjectMapper` (checked exception) | +| Tool call id | `getToolCallId()` | not available, drop those lines | +| OpenAI model property | `spring.ai.openai.chat.model` | `spring.ai.openai.chat.options.model` | + +Boot 4 still accepts the Boot 3 property names but marks them deprecated. + +## Group turns into one session + +A chat backend handles one request per user message, and each `ChatClient` call is its own trace. Maple joins those traces into one session by the `spring.ai.chat.client.conversation.id` attribute on the `chat_client` span. Spring AI sets it from the `ChatMemory.CONVERSATION_ID` advisor parameter, the same parameter that selects the chat memory history: + +```java +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.chat.client.advisor.MessageChatMemoryAdvisor; +import org.springframework.ai.chat.memory.ChatMemory; +import org.springframework.stereotype.Service; + +@Service +public class ChatService { + + private final ChatClient chatClient; + + public ChatService(ChatClient.Builder builder, ChatMemory chatMemory, SupportTools tools) { + this.chatClient = builder + .defaultSystem("You are a concise support assistant.") + .defaultTools(tools) + .defaultAdvisors(MessageChatMemoryAdvisor.builder(chatMemory).build()) + .defaultAdvisors(a -> a.param(MapleAiObservationConfig.AGENT_NAME, "support_agent")) + .build(); + } + + public String reply(String conversationId, String userMessage) { + return chatClient.prompt() + .user(userMessage) + .advisors(a -> a.param(ChatMemory.CONVERSATION_ID, conversationId)) + .call() + .content(); + } + +} +``` + +Without the parameter, the attribute is missing and Maple shows every message as its own one-turn session, named after its trace id. Things to get right: + +- **Pass the id on every call.** It belongs on the request (`.advisors(...)` on `prompt()`), not on the builder. A constant set with `defaultAdvisors` puts every user into one session. +- **Use the conversation id your app already stores**, not a fresh UUID per request. +- **The id works without chat memory.** If your app sends the history itself, pass the parameter anyway; Spring AI records it from the request whether or not a memory advisor reads it. +- **Sub-agents don't need it.** Maple groups a whole trace by any span in it that carries the id, so the top-level `ChatClient` call is enough. + +Maple reads only `spring.ai.chat.client.conversation.id` for Spring AI spans. Adding `gen_ai.conversation.id` or `session.id` does nothing here. + +## Record prompts, responses and tool calls + +Spring AI has content switches, but none of them put the conversation where Maple reads it: + +- `spring.ai.chat.observations.log-prompt` and `log-completion` (and the `spring.ai.chat.client.observations.*` pair) write the prompt and reply to SLF4J at INFO level, with the trace id for correlation. They never add span attributes, and Maple's session views read span attributes only. +- `spring.ai.tools.observations.include-content` puts tool arguments and results on the span, but under `spring.ai.tool.call.arguments` and `spring.ai.tool.call.result`, which Maple doesn't read. + +With `maple.ai.capture-content=true`, the filter above writes the attributes Maple does read: + +| Span | Attributes | Contents | +|---|---|---| +| `chat ` | `gen_ai.input.messages` | everything sent to the model: system prompt, the history chat memory added, tool calls and tool results | +| `chat ` | `gen_ai.output.messages` | the reply, including tool calls the model asked for | +| `execute_tool ` | `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result` | the JSON arguments and the string returned to the model | + +Maple builds the transcript from these and labels each turn with the first line of the user's message. You can leave the Spring AI switches off; they only add log lines. + +Every `chat` span carries the whole conversation so far, so spans grow with long chats. Don't set `management.opentelemetry.tracing.limits.max-attribute-value-length`: a cut JSON value no longer parses, and Maple drops it. + +### Privacy: turn content off or redact it + +Set `maple.ai.capture-content=false` (the default in the class) and no message or tool content leaves the process. Sessions, turns, model names, tokens, tool names and failures still show up; the transcript is empty. To redact instead, change `message(...)` to mask what you don't want sent, for example email addresses in text parts. Keep personal data out of the conversation id and agent names; they are sent regardless of the setting. + +## Tools, errors and sub-agents + +Every tool call gets an `execute_tool ` span. Spring AI puts the name in `spring.ai.tool.definition.name`; the filter copies it to `gen_ai.tool.name`, which Maple reads for the tool pages. + +### Failed tools + +When a `@Tool` method throws, Spring AI's default `ToolExecutionExceptionProcessor` returns the exception message to the model as the tool result, and the `execute_tool` span ends with status OK. The model usually recovers gracefully, which is good for users and bad for debugging: Maple would count the call as a success. + +The processor bean in the configuration class calls `error()` on the running tool observation first. The span gets status `Error` with the exception message and an `exception` event, Maple counts it as a failed tool call, and the model still gets the message. Successful calls are untouched. + +Two things to know: + +- Defining the bean replaces Spring AI's, so `spring.ai.tools.throw-exception-on-error` no longer applies. To fail the whole call instead, build the fallback with `.alwaysThrow(true)`. +- If your app already defines a `ToolExecutionExceptionProcessor`, add the `error()` call to it instead of adding a second bean. + +A tool that returns an error string instead of throwing is a success as far as any tracer can tell. Throw if you want the failure counted. + +### Sub-agents as tools + +Spring AI has no agent class. The idiomatic multi-agent setup is one `ChatClient` per role, with the orchestrator calling the others through `@Tool` methods. Give each client its own agent name: + +```java +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.tool.annotation.Tool; +import org.springframework.stereotype.Component; + +@Component +public class Workers { + + private final ChatClient weather; + + public Workers(ChatClient.Builder builder, WeatherTools weatherTools) { + this.weather = builder.clone() + .defaultSystem("You answer weather questions using your tools.") + .defaultTools(weatherTools) + .defaultAdvisors(a -> a.param(MapleAiObservationConfig.AGENT_NAME, "weather_worker")) + .build(); + } + + @Tool(name = "weather_worker", description = "Ask the weather specialist about a city") + public String weatherWorker(String task) { + return weather.prompt().user(task).call().content(); + } + +} +``` + +In Maple, `execute_tool weather_worker` with the worker's `chat_client` span under it shows up as a delegation into a `weather_worker` lane, with the task as its input and the worker's answer as its output. A client without an agent name gets no lane; its model and tool calls are drawn in the caller's lane. + +Spring AI runs the tool calls of one model response one after another, on the calling thread, so context flows without help. If you fan work out to your own executor, propagate the current observation to the worker threads with Micrometer's `context-propagation` library (`ContextSnapshot`, or a wrapped executor). Otherwise each worker starts a new trace, and since it carries no conversation id, Maple shows it as a separate session. + +## Tokens and cost + +Every `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports them. Maple reads all four. The `chat_client` spans carry no usage, so nothing is counted twice. + +Streaming has one catch. OpenAI (and OpenAI-compatible gateways such as OpenRouter) only return usage on a stream when the request asks for it. Without `spring.ai.openai.chat.stream-options.include-usage=true`, or `OpenAiChatOptions.builder().streamUsage(true)` per call, streamed turns show zero tokens. + +Spring AI records the provider as `gen_ai.system`, derived from the client class rather than the model. Behind OpenRouter, a Claude model called through the OpenAI starter is labeled `openai`. Maple uses the provider only to decide whether cached tokens are included in the input count, so this matters only for cache figures. + +Spring AI emits no cost attribute, and Maple never prices tokens itself, so sessions show as unpriced. + +## Short-lived processes + +Spring Boot owns the `SdkTracerProvider` and shuts it down when the application context closes, which flushes the batch of pending spans. A web app needs nothing extra. For other shapes: + +- **`CommandLineRunner` apps and batch jobs.** Let `run()` return, or exit through `SpringApplication.exit(context)`. `System.exit()` still runs Boot's shutdown hook; `kill -9` and `Runtime.halt()` lose the last batch. +- **Serverless (Spring Cloud Function on AWS Lambda and similar).** The runtime freezes the process between invocations, so flush before returning from each one. Don't shut the provider down. + +```java +import java.util.concurrent.TimeUnit; + +import io.opentelemetry.sdk.trace.SdkTracerProvider; + +// inject SdkTracerProvider, then at the end of each invocation: +tracerProvider.forceFlush().join(10, TimeUnit.SECONDS); +``` + +The exporter sends a batch every 5 seconds by default, so allow a few seconds after a conversation before looking for it in Maple. + +## Running the OpenTelemetry Java agent too + +Use one tracing setup per service. The OpenTelemetry Java agent does not turn Micrometer Observations into spans, so with the agent alone you get the HTTP spans but no `chat_client`, `chat` or `execute_tool` span. You still need the starter from this guide. + +If the agent has to stay (it also instruments JDBC, Kafka and other libraries), three things change: + +- **Model calls appear twice.** The agent instruments the OpenAI Java SDK, which Spring AI 2.0's OpenAI starter uses underneath, and adds its own `chat ` span to every call. Turn it off with `-Dotel.instrumentation.openai-java.enabled=false`. +- **HTTP requests appear twice**, once from the agent and once from Boot's `http.server.requests` observation. Set `management.observations.enable.http.server.requests=false`. +- **Two exporters run.** Point the agent (`OTEL_EXPORTER_OTLP_*`) and Boot (the properties above) at Maple, or traces arrive with gaps. + +We tested the starter setup end to end; the agent combination above is not covered by our tests. + +## Check that it works + +Run one conversation of two or three messages with the same conversation id, including one that calls a tool, plus a message with a second id. Wait about a minute, then open **Agent Sessions** in Maple. You should see: + +- **One session per conversation id**, with the framework shown as Spring AI and one turn per `ChatClient` call. A list of one-turn sessions named after trace ids means the `ChatMemory.CONVERSATION_ID` parameter is missing. +- **Spans named** `spring_ai chat_client` (as an agent span named after your `gen_ai.agent.name`), `chat openai/gpt-4o-mini` (your model id) and `execute_tool get_weather`. HTTP `POST` spans to the model provider appear muted next to them. +- **No spans named** `tool _calling `, `call` or `message_chat_memory`. If they show up, the `ObservationPredicate` bean isn't loaded. +- **The transcript**: each user message, the assistant's replies, and the tool calls with their arguments and results. +- **Tokens** on every model call, including streamed ones. +- **Sub-agents** as lanes named after each client's agent name, and a tool that threw counted as a failed tool call under its own name. +- **Cost** shown as unpriced. + +## Troubleshooting + +- **Only some turns arrive, or sessions have gaps.** Sampling is at Boot's default of 10%. Set `management.tracing.sampling.probability=1.0`. +- **Every message is its own session.** The `chat_client` span has no `spring.ai.chat.client.conversation.id`. Pass `.advisors(a -> a.param(ChatMemory.CONVERSATION_ID, id))` on every `prompt()` call. +- **All users land in one session.** The conversation id is a constant, usually set once with `defaultAdvisors` on the builder. Pass it per request. +- **Transcript is empty, but tokens and tools show up.** `maple.ai.capture-content` isn't `true`, or `MapleAiObservationConfig` isn't in a package Spring scans. `log-prompt` and `log-completion` don't help; they write to the log. +- **Twice as many model calls as expected, and a tool called `tool _calling ` in the tool list.** The advisor and `chat_client` spans are being counted. Make sure the `ObservationPredicate` and `ObservationFilter` beans are loaded. +- **A tool that threw shows as successful.** The custom `ToolExecutionExceptionProcessor` isn't active, or the app defines its own. Add the `registry.getCurrentObservation().error(exception)` call to the one that runs. +- **Streamed turns show zero tokens.** Add `spring.ai.openai.chat.stream-options.include-usage=true`. +- **A sub-agent shows up as its own session.** It ran on a thread without the caller's trace context. Propagate the observation to the executor, or run the sub-agent on the calling thread. +- **Tool names are missing on the tool pages.** The `ObservationFilter` isn't running; Spring AI alone emits only `spring.ai.tool.definition.name`. +- **Framework shows as Unidentified on some model spans.** With starters other than OpenAI, a `chat` span for a call without tools has no Spring AI marker, so Maple files it as a generic GenAI span. The session, transcript and tokens are unaffected. +- **Nothing arrives from tests.** `@SpringBootTest` turns tracing export off. Add `@AutoConfigureTracing` to the test class if you want spans from tests. +- **`Failed to publish metrics` warnings every minute.** The starter's metrics exporter is still pointed at `localhost:4318`. Set the two `management.otlp.metrics.export.*` lines or disable metrics export. +- **`401` from ingest.** The key is wrong or from the other region. The property must read `...headers.Authorization=Bearer YOUR_INGEST_KEY`. +- **Using LangChain4j instead of Spring AI.** This guide doesn't apply. Follow the [OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry). + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Agent tracing guides](/docs/agent-tracing) +- [Any language: the OpenTelemetry GenAI conventions](/docs/agent-tracing/opentelemetry) +- [Spring AI observability reference](https://docs.spring.io/spring-ai/reference/observability/index.html) +- [Spring AI tool calling](https://docs.spring.io/spring-ai/reference/api/tools.html) +- [Spring Boot tracing reference](https://docs.spring.io/spring-boot/reference/actuator/tracing.html) diff --git a/apps/landing/src/content/docs/agent-tracing/strands.md b/apps/landing/src/content/docs/agent-tracing/strands.md new file mode 100644 index 0000000000..6aa4dba250 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/strands.md @@ -0,0 +1,308 @@ +--- +title: "Trace Strands Agents with OpenTelemetry" +description: "Send Strands Agents traces to Maple with the prompt, reply and tool calls on every span, one session per conversation, and token counts that don't double." +group: "AI Agents" +order: 24 +navLabel: "Strands Agents" +icon: "strands" +--- + +Strands Agents ships its own OpenTelemetry tracer. Every `agent(...)` call becomes one trace with an `invoke_agent` span, an `execute_event_loop_cycle` span per reasoning step, a `chat` span per model call and an `execute_tool` span per tool call. Maple recognizes these spans as Strands without any extra instrumentation library. + +The catch is where Strands puts the conversation. By default, prompts, replies and tool results are written as span events, and Maple reads span attributes only, so the transcript comes out empty even though tokens and tool calls show up. One environment variable fixes it. This guide covers the Python SDK (`strands-agents` 1.54 or newer, tested on 1.57.1) and notes where the TypeScript SDK (`@strands-agents/sdk` 1.19) differs. + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-strands](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-strands) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Strands Agents in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-strands -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Install Strands telemetry and export to Maple + +The `otel` extra adds the OTLP/HTTP exporter. Add the extra for your model provider too (`openai`, `anthropic`, `litellm`; Bedrock needs none). + +```bash +pip install 'strands-agents[otel,openai]>=1.57' +``` + +Configure the exporter and the content settings with environment variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_SERVICE_NAME="support-agent" +export OTEL_RESOURCE_ATTRIBUTES="deployment.environment.name=production" +export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only,gen_ai_use_latest_invocation_tokens" +``` + +EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself, so set the base URL only. + +The last line matters most. Each token does one job: + +- `gen_ai_latest_experimental` switches Strands to the current GenAI conventions: messages as `gen_ai.input.messages` / `gen_ai.output.messages` in `{role, parts}` form, `gen_ai.system_instructions`, and tool arguments and results on `execute_tool` spans. +- `gen_ai_span_attributes_only` writes those messages as span attributes instead of span events. Without it, Maple shows tokens and tools but no transcript. It requires `strands-agents` 1.48 or newer. +- `gen_ai_use_latest_invocation_tokens` makes the `invoke_agent` span report this call's tokens instead of the agent's running total. See [Tokens and cost](#tokens-cost-and-the-agent-level-roll-up). + +Strands reads this variable once, when the first `Agent` is created. Set it in the environment (shell, `.env`, container config), not in code that runs after an agent exists. + +Then start the tracer once, before your first agent runs: + +```py +# telemetry.py: import this from your entry point before creating any Agent +from strands.telemetry import StrandsTelemetry + +telemetry = StrandsTelemetry().setup_otlp_exporter() +``` + +`StrandsTelemetry()` creates a `TracerProvider`, registers it globally and adds a `BatchSpanProcessor` with the stock OTLP/HTTP exporter, so all the standard `OTEL_EXPORTER_OTLP_*` variables apply. + +If your app already sets up OpenTelemetry (a global `TracerProvider` from your web framework, `opentelemetry-instrument`, or the ADOT distro on AgentCore), skip `StrandsTelemetry()` entirely. Strands uses whatever global provider exists; add a `BatchSpanProcessor(OTLPSpanExporter(...))` pointing at Maple to that provider instead of creating a second one. + +## Group turns into one session + +Each `agent(...)` call is its own trace, and Strands has no built-in conversation id on its spans. Without one, Maple shows every message as a separate one-turn session named after its trace id. + +Strands copies `trace_attributes` onto every span it creates for that agent. Maple reads `session.id` for Strands, so pass your conversation id there: + +```py +from strands import Agent +from strands.models.openai import OpenAIModel +from strands.session.file_session_manager import FileSessionManager + +model = OpenAIModel(model_id="gpt-4o-mini", params={"max_tokens": 600}) + + +def handle_message(conversation_id: str, text: str) -> str: + agent = Agent( + name="support_agent", + model=model, + tools=[get_weather, calculate], + system_prompt="You are a concise support assistant.", + session_manager=FileSessionManager(session_id=conversation_id, storage_dir="./sessions"), + trace_attributes={"session.id": conversation_id}, + callback_handler=None, + ) + return str(agent(text)) +``` + +This is the usual shape of a chat backend: one request per user message, history restored by the session manager (`FileSessionManager` here, `S3SessionManager` in production), and the same id used for history and for tracing. The session manager's `session_id` never reaches the spans on its own, so both arguments are needed. + +A few things to get right: + +- **Use the conversation id, not a per-request UUID.** A fresh id per call gives you one session per message again. +- **Don't share one module-level `Agent` across users.** Its `trace_attributes` would carry one user's id into everyone's traces. Create the agent per request, or per conversation. +- **Name every agent.** `name` becomes `gen_ai.agent.name`. Unnamed agents are all called `Strands Agents`, which merges sub-agents into one lane. + +Only `session.id` groups sessions for Strands. `gen_ai.conversation.id` is ignored on Python Strands spans, so there is no need to add it. + +## Record prompts, responses and tool calls + +With the three opt-in tokens set, every span carries the conversation as JSON attributes: + +| Span | What Maple reads | +|---|---| +| `invoke_agent ` | this turn's user message, the final reply, `gen_ai.system_instructions` | +| `chat` | the full message history sent to the model, the model's reply with `finish_reason`, the system prompt | +| `execute_tool ` | `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result` (successful calls only) | + +Maple builds the transcript from these, and labels each turn with the first line of the user's message. + +Leaving out `gen_ai_span_attributes_only` is the most common reason for an empty transcript. The spans look complete in a trace viewer that renders events, and Maple still shows nothing. + +### Privacy: redact or drop content + +Content capture is on by default in Strands; there is no single off switch. Redaction is controlled by one more token in the same variable, `gen_ai_unredacted_attributes=`, followed by a `;`-separated allowlist. Anything not listed is replaced with `[REDACTED]`: + +```bash +# Keep replies and tool results, redact user input and system prompts +export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only,gen_ai_use_latest_invocation_tokens,gen_ai_unredacted_attributes=gen_ai.output.*;gen_ai.tool.call.result" + +# Redact every message (empty allowlist) +export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only,gen_ai_use_latest_invocation_tokens,gen_ai_unredacted_attributes=" +``` + +The attributes covered are `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Only a single trailing `*` works as a wildcard. Redacted values aren't JSON, so Maple leaves those parts of the transcript blank; tokens, tools and errors are unaffected. + +Tool descriptions and schemas (`gen_ai.tool.description`, `gen_ai.tool.json_schema`) and anything you put in `trace_attributes` are never redacted. Keep emails and customer names out of `trace_attributes`. + +## Tools, errors and sub-agents + +Every tool call gets an `execute_tool ` span with `gen_ai.tool.name`, `gen_ai.tool.call.id` and `gen_ai.tool.description`. + +When a tool raises, Strands catches the exception, feeds the error back to the model, and marks the span failed: status `ERROR` with the exception message (for example `transport data service unavailable (503)`) and `gen_ai.tool.status=error`. Maple counts it as a failed tool call and groups it with other failures of the same message on the tool pages. A tool that returns `{"status": "error", ...}` itself is marked the same way. You don't need to add anything. + +### Agents as tools + +`agent.as_tool()` wraps a sub-agent so an orchestrator can call it. The trace nests the way Maple expects: `execute_tool weather_worker` with the sub-agent's `invoke_agent weather_worker` span inside it, which Maple shows as a delegation with its own lane. + +```py +weather_worker = Agent(name="weather_worker", model=model, tools=[get_weather], callback_handler=None) +budget_worker = Agent(name="budget_worker", model=model, tools=[calculate], callback_handler=None) + +orchestrator = Agent( + name="orchestrator", + model=model, + tools=[ + weather_worker.as_tool(description="Weather for a city"), + budget_worker.as_tool(description="Travel budget arithmetic"), + ], + trace_attributes={"session.id": conversation_id}, + callback_handler=None, +) +orchestrator("Produce a mini briefing about Amsterdam: weather and a 3-day budget.") +``` + +Only the orchestrator needs `session.id`. Maple groups sessions per trace, and the sub-agents run inside the orchestrator's trace. Strands runs tool calls from one model response concurrently by default, so parallel sub-agents appear as overlapping sibling spans. + +### Graph and Swarm + +`Graph` and `Swarm` open their own root span (`invoke_graph` / `invoke_swarm`) with each node's `invoke_agent` span beneath it. The root carries the multi-agent object's `trace_attributes`, not the agents'. + +```py +from strands.multiagent import GraphBuilder, Swarm + +swarm = Swarm([researcher, writer], trace_attributes={"session.id": conversation_id}) + +builder = GraphBuilder() +builder.add_node(researcher, "research") +builder.add_node(writer, "write") +builder.add_edge("research", "write") +graph = builder.build() +graph.trace_attributes = {"session.id": conversation_id} # GraphBuilder has no setter for it +``` + +`GraphBuilder.build()` doesn't forward trace attributes, which is why the last line exists. Giving each node agent `trace_attributes={"session.id": ...}` works too, since one span per trace is enough. + +The graph node id is not exported. Maple identifies each node by its agent's `name`, so name node agents after what they do. + +### Human approval (interrupts) + +An interrupt raised from a `BeforeToolCallEvent` hook ends the current trace, and the resumed call starts a new one. Both carry the agent's `session.id`, so they stay in one session. The interrupted tool appears twice, once ended early (status OK, no result) and once with the real outcome; both spans share the same `gen_ai.tool.call.id`. The tool body runs once, but Maple counts both spans as tool calls, so an approved call shows as two calls, one of them without a result. + +## Tokens, cost and the agent-level roll-up + +Each `chat` span reports `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`. Input includes cached tokens, which matches how Maple reads them. Since 1.54, prompt-cache hits and writes use the names Maple reads (`gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_creation.input_tokens`). Older versions only emit `cache_read_input_tokens` / `cache_write_input_tokens`, which Maple ignores. + +Strands streams every model call internally and reads usage from the stream's final event, so streamed turns report tokens too. The OpenAI model provider requests `stream_options.include_usage` automatically. + +The `invoke_agent` span also carries usage, and this is where counts go wrong: + +- **Without `gen_ai_use_latest_invocation_tokens`,** it reports the agent instance's lifetime total. On a long-lived agent, turn 8 reports the sum of turns 1 to 8, and any backend that sums spans over-counts by several times. +- **With it,** the span reports only this call's tokens, which Maple's detail page recognizes as a roll-up of the `chat` spans below and doesn't count twice. + +A per-request agent (as in the session example above) never accumulates across turns, which is another reason to create agents per request. + +Strands doesn't emit cost, so Maple shows sessions as unpriced. Maple never prices tokens itself. + +Two more gaps: Strands records time to first token as `gen_ai.server.time_to_first_token` in milliseconds, which Maple doesn't read, and `chat` spans carry no `gen_ai.response.id`. The missing response id means Maple can't recognize the same call reported twice, so don't also enable a gateway export such as OpenRouter Broadcast for the same traffic. + +## Flush before short-lived processes exit + +Spans go through a `BatchSpanProcessor`, which exports every few seconds. A long-running server needs nothing extra. Scripts, CLIs, notebooks and jobs lose their last spans unless they flush: + +```py +from telemetry import telemetry + +try: + handle_message(conversation_id, "What's the weather in Berlin?") +finally: + telemetry.tracer_provider.force_flush() + telemetry.tracer_provider.shutdown() +``` + +On AWS Lambda, call `telemetry.tracer_provider.force_flush()` at the end of each invocation and don't call `shutdown()`, because the warm container reuses the provider. + +A call cut off by `asyncio.wait_for` or task cancellation may export incomplete spans ([strands-agents/harness-sdk#3609](https://github.com/strands-agents/harness-sdk/issues/3609)). Ended spans still flush normally. + +## TypeScript SDK differences + +The TypeScript SDK emits the same span tree and honours the same `OTEL_SEMCONV_STABILITY_OPT_IN` tokens, except `gen_ai_use_latest_invocation_tokens`, which it doesn't support. Its OpenTelemetry packages are optional peer dependencies, so install them explicitly: + +```bash +npm install @strands-agents/sdk @opentelemetry/api @opentelemetry/sdk-trace-base @opentelemetry/sdk-trace-node @opentelemetry/resources @opentelemetry/exporter-trace-otlp-http @opentelemetry/sdk-metrics @opentelemetry/exporter-metrics-otlp-http +``` + +Set `OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only"` and the same `OTEL_EXPORTER_OTLP_*` variables as above, then: + +```ts +import { Agent, FileStorage, SessionManager } from "@strands-agents/sdk" +import { OpenAIModel } from "@strands-agents/sdk/models/openai" +import { setupTracer } from "@strands-agents/sdk/telemetry" + +const provider = setupTracer({ exporters: { otlp: true } }) // reads OTEL_EXPORTER_OTLP_* env vars + +const model = new OpenAIModel({ modelId: "gpt-4o-mini", params: { max_tokens: 600 } }) + +// One request per user message: build the agent, restore history, answer. +async function handleMessage(conversationId: string, text: string): Promise { + const agent = new Agent({ + name: "support_agent", + model, + tools: [getWeather], + systemPrompt: "You are a concise support assistant.", + traceAttributes: { "session.id": conversationId, "gen_ai.conversation.id": conversationId }, + sessionManager: new SessionManager({ + sessionId: conversationId, + storage: { snapshot: new FileStorage("./sessions") }, + }), + printer: false, + }) + return String(await agent.invoke(text)) +} + +try { + await handleMessage(conversationId, "What's the weather in Berlin?") +} finally { + await provider.forceFlush() + await provider.shutdown() +} +``` + +Set both keys in `traceAttributes`. The TypeScript SDK names its tracer and `gen_ai.provider.name` after `OTEL_SERVICE_NAME`, so with a custom service name Maple can't tell the spans come from Strands. It files them as generic GenAI spans, which group by `gen_ai.conversation.id`, and shows the framework as "Unidentified". Transcript, tools, failures and tokens still work. The TypeScript SDK writes `traceAttributes` on the `invoke_agent` span only, which is enough, since Maple groups sessions per trace. + +Create the agent per request, as above. The TypeScript `invoke_agent` span always reports the agent instance's running total, so a long-lived agent reports turns 1 to N again on turn N and Maple's totals grow with every turn. A fresh agent per request, with history restored by `SessionManager`, reports only this turn. + +The exporter sends OTLP/HTTP JSON, which Maple ingest accepts. `setupTracer()` only flushes on Node's `beforeExit`, which never fires after `process.exit()`, so flush explicitly as above. Failed `Graph` nodes currently end with status OK ([harness-sdk#4166](https://github.com/strands-agents/harness-sdk/issues/4166)); tool failures are marked correctly. + +## Check that it works + +Run one conversation of two or three messages, including one that calls a tool, then open **Agent Sessions** in Maple. Within a minute you should see: + +- **One session per conversation id**, with the framework shown as Strands Agents and one turn per `agent(...)` call. A list of one-turn sessions means `session.id` is missing. +- **The transcript**: each user message, the assistant's replies and the tool calls with their arguments and results. +- **Spans named** `invoke_agent support_agent` (your agent's `name`), `execute_event_loop_cycle`, `chat` and `execute_tool get_weather`. +- **Tokens** on every model call, including streamed ones, and the model id you passed (for example `gpt-4o-mini`, or a Bedrock id such as `us.anthropic.claude-sonnet-4-5-20250929-v1:0`). +- **Sub-agents** as separate lanes named after each agent, with failed tool calls counted under the tool's name. +- **Cost** shown as unpriced. + +## Troubleshooting + +- **Transcript is empty, but tokens and tools show up.** Content is still in span events. Add `gen_ai_span_attributes_only` (and `gen_ai_latest_experimental`) to `OTEL_SEMCONV_STABILITY_OPT_IN`, make sure it's set before the first `Agent` is created, and upgrade to 1.48 or newer. +- **Every message is its own session.** No `session.id` on the trace. Pass `trace_attributes={"session.id": conversation_id}` to the agent, or to the `Swarm` / `Graph` that runs it. Setting only the session manager's `session_id` isn't enough. +- **Two users' messages land in one session.** A shared `Agent` instance carries one `trace_attributes` dict for everyone. Create the agent per request. +- **Token totals look several times too high.** The `invoke_agent` span reports the agent's lifetime usage. Add `gen_ai_use_latest_invocation_tokens`, or create the agent per request. +- **Cache tokens are zero with prompt caching on.** Versions before 1.54 use `cache_read_input_tokens`, which Maple doesn't read. Upgrade. +- **Every sub-agent is called "Strands Agents".** The agents have no `name`. Set `Agent(name=...)` on each one. +- **Graph session splits from the rest of the conversation.** `GraphBuilder.build()` drops trace attributes. Set `graph.trace_attributes` after building. +- **No model name on spans.** A custom `Model` subclass that only implements `get_config()` gets no `gen_ai.request.model` ([harness-sdk#4205](https://github.com/strands-agents/harness-sdk/issues/4205)). Give it a `config` dict with `model_id`. +- **Every model call appears twice.** Another instrumentation (OpenLIT, OpenLLMetry, OpenInference, an OpenAI or Bedrock GenAI instrumentor) is wrapping the same calls. Strands' own spans are enough; remove the other one. +- **Nothing arrives from a script.** The process exited before the batch exported. Call `force_flush()` and `shutdown()` in a `finally` block. +- **`401` from ingest.** The key is wrong or from the other region. The header must be written `Authorization=Bearer YOUR_INGEST_KEY` in `OTEL_EXPORTER_OTLP_HEADERS`. +- **TypeScript: framework shows "Unidentified" and sessions split.** Add `gen_ai.conversation.id` next to `session.id` in `traceAttributes`. +- **TypeScript: tokens grow with every turn.** A reused `Agent` reports its running total on `invoke_agent`. Create the agent per request and restore history with `SessionManager`. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [Agent tracing guides](/docs/agent-tracing) +- [Strands Agents traces documentation](https://strandsagents.com/docs/user-guide/observability-evaluation/traces/) +- [Strands telemetry tracer API reference](https://strandsagents.com/docs/api/python/strands.telemetry.tracer/) diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md new file mode 100644 index 0000000000..3baa006824 --- /dev/null +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -0,0 +1,409 @@ +--- +title: "Trace Vercel AI SDK agents with OpenTelemetry" +description: "Send the AI SDK's OpenTelemetry spans to Maple so each chat conversation is one Agent Session with its transcript, tool calls, sub-agents and tokens, in Node.js or Next.js." +group: "AI Agents" +order: 10 +navLabel: "Vercel AI SDK" +icon: "vercel" +--- + +The Vercel AI SDK traces itself. Every `generateText`, `streamText` and `ToolLoopAgent` call emits an `invoke_agent` span, a `chat` span per model request with the prompt, the reply and the token counts, and an `execute_tool` span per tool call with its arguments and result. The spans follow the OpenTelemetry GenAI semantic conventions, prompts and replies are recorded by default, and Maple reads all of it without an extra instrumentation package. + +Two things go wrong by default. In AI SDK 7, nothing is traced until you call `registerTelemetry()` once at startup: without it the SDK emits zero spans, with no warning. And the AI SDK has no conversation id. Each call is its own trace with nothing linking it to the previous message, so a chat backend that handles one message per request shows up in Maple as one session per message. This guide fixes both. + +It covers AI SDK 7 (`ai` 7.0.106 or newer) on Node.js 22 or newer, in a plain Node.js service and in Next.js. It was tested with `ai` 7.0.118, `@ai-sdk/otel` 1.0.118, OpenTelemetry JS 0.222.0 and `@openrouter/ai-sdk-provider` 3.1.0 on Node.js 26. The Next.js setup uses `@vercel/otel` 2.1.3. AI SDK 5 and 6 work too, with less detail in the transcript; see [AI SDK 5 and 6](#ai-sdk-5-and-6). + +## Quick setup with a coding agent + +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-vercel-ai-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-vercel-ai-sdk) skill, which contains every step of this guide. + +```text +Set up Maple agent tracing for Vercel AI SDK in this project. + +Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-vercel-ai-sdk -y`, then follow it. + +My Maple ingest key is maple_pk_... and my organization is in the US region. +``` + +Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. + +## Export AI SDK spans to Maple + +In AI SDK 7, OpenTelemetry support moved out of the `ai` package into `@ai-sdk/otel`. It creates spans; the OpenTelemetry SDK exports them. Install both: + +```bash +npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/api @opentelemetry/sdk-node \ + @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources +``` + +Point the exporter at Maple with the standard OpenTelemetry variables: + +```bash +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +``` + +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. + +Then set up tracing once, when your process starts: + +```ts +// instrumentation.ts +import { OpenTelemetry } from "@ai-sdk/otel" +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { resourceFromAttributes } from "@opentelemetry/resources" +import { NodeSDK } from "@opentelemetry/sdk-node" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { registerTelemetry } from "ai" + +// Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) + +export const sdk = new NodeSDK({ + resource: resourceFromAttributes({ + "service.name": "support-agent", + "deployment.environment.name": process.env.NODE_ENV ?? "development", + }), + spanProcessors: [spanProcessor], +}) +sdk.start() + +registerTelemetry( + new OpenTelemetry({ + usage: true, + runtimeContext: true, + // The conversation id, explained in the next section + enrichSpan: ({ runtimeContext }) => + typeof runtimeContext?.conversationId === "string" + ? { "gen_ai.conversation.id": runtimeContext.conversationId } + : undefined, + }), +) +``` + +Import it as the first line of your entry point (`import "./instrumentation"`). The AI SDK doesn't patch any modules, so what matters is that `registerTelemetry()` runs before the first AI SDK call. Call it exactly once: it appends to a global list, and every registered `OpenTelemetry` instance emits its own copy of every span. + +`usage: true` and `runtimeContext: true` add a few AI SDK-specific `ai.*` attributes next to the GenAI ones: the uncached and text token counts, and the conversation id you pass in. Maple identifies AI SDK spans by those `ai.*` keys on the `gen_ai` tracer. Without them, the `invoke_agent` and `chat` spans read as generic GenAI spans, the session's framework shows as **Unidentified**, and time to first token isn't picked up. + +If your app already starts an OpenTelemetry SDK (auto-instrumentation, Sentry, your own `NodeTracerProvider`), don't start a second one. Add the `BatchSpanProcessor` above to the existing provider and keep the `registerTelemetry()` call. `@ai-sdk/otel` uses the global tracer provider. Don't pass it a `tracer` from another tracer name either: Maple expects the default `gen_ai` scope. + +### Next.js + +Next.js calls `register()` in `instrumentation.ts` once per server runtime. Use `@vercel/otel` there, as in the [Next.js guide](/docs/guides/instrumentation-nextjs), and register the AI SDK integration next to it: + +```ts +// instrumentation.ts (project root, or src/instrumentation.ts) +import { OpenTelemetry } from "@ai-sdk/otel" +import { OTLPHttpProtoTraceExporter, registerOTel } from "@vercel/otel" +import { registerTelemetry } from "ai" + +export function register() { + registerOTel({ + serviceName: "support-chat", + traceExporter: new OTLPHttpProtoTraceExporter({ + url: "https://ingest.maple.dev/v1/traces", // EU: https://ingest.eu.maple.dev/v1/traces + headers: { authorization: `Bearer ${process.env.MAPLE_INGEST_KEY}` }, + }), + }) + + registerTelemetry( + new OpenTelemetry({ + usage: true, + runtimeContext: true, + enrichSpan: ({ runtimeContext }) => + typeof runtimeContext?.conversationId === "string" + ? { "gen_ai.conversation.id": runtimeContext.conversationId } + : undefined, + }), + ) +} +``` + +The AI SDK spans then nest under the request span Next.js creates for the route handler, like `POST /api/chat`. + +`@vercel/otel` ends every span of a trace that is still open when the trace's root span ends. AI SDK work that keeps running after the request span has ended, such as a stream read inside `after()`, loses the end of its spans: the reply and the token counts. Read the stream inside the request, for example by returning it as the response with `toUIMessageStreamResponse()` or `createAgentUIStreamResponse()`. + +## Group every turn of a conversation into one session + +Maple groups traces into sessions by `gen_ai.conversation.id`. The AI SDK never sets it: there is no session or thread concept in `generateText`, and `ToolLoopAgent` doesn't keep one either. What it does have is `runtimeContext`, a per-call object your code passes in. The `enrichSpan` callback above copies `runtimeContext.conversationId` onto every span as `gen_ai.conversation.id`. + +Runtime context stays out of telemetry unless you list the key in `telemetry.includeRuntimeContext`, so each call needs both: + +```ts +import { streamText } from "ai" + +const result = streamText({ + model, + messages, + runtimeContext: { conversationId: chatId }, + telemetry: { + functionId: "support_agent", + includeRuntimeContext: { conversationId: true }, + }, +}) +``` + +Use the chat or thread id your app already stores. It must be stable for the whole conversation and different between conversations. A per-process constant merges every user into one session. + +If you skip this, each request shows up in **Agent Sessions** as its own one-turn session named `trace:`. + +### ToolLoopAgent + +`agent.generate()` and `agent.stream()` don't take `runtimeContext` per call. Declare the id as a call option and turn it into runtime context in `prepareCall`: + +```ts +// agent.ts +import { openai } from "@ai-sdk/openai" +import { ToolLoopAgent } from "ai" +import { z } from "zod" + +export const assistant = new ToolLoopAgent({ + model: openai("gpt-4o-mini"), + instructions: "You are a helpful assistant.", + tools: { get_weather: getWeather, fetch_transport_data: fetchTransportData }, + callOptionsSchema: z.object({ conversationId: z.string() }), + prepareCall: ({ options, ...rest }) => ({ + ...rest, + runtimeContext: { conversationId: options.conversationId }, + }), + telemetry: { + functionId: "support_agent", + includeRuntimeContext: { conversationId: true }, + }, +}) + +const result = await assistant.generate({ messages, options: { conversationId: chatId } }) +``` + +`functionId` becomes `gen_ai.agent.name`. The agent's `id` isn't exported to telemetry, so set `functionId` on every agent you want to see by name. + +### Chat routes with useChat + +`useChat` already sends a stable chat id with every request, as `id` in the JSON body. Pass it through: + +```ts +// app/api/chat/route.ts +import { createAgentUIStreamResponse, type UIMessage } from "ai" +import { assistant } from "@/agent" + +export async function POST(req: Request) { + const { id, messages }: { id: string; messages: UIMessage[] } = await req.json() + + return createAgentUIStreamResponse({ + agent: assistant, + uiMessages: messages, + options: { conversationId: id }, + }) +} +``` + +With `streamText` directly, pass `runtimeContext: { conversationId: id }` as in the first example and return `result.toUIMessageStreamResponse()`. + +Tool approvals work the same way. In AI SDK 7 you mark a tool with `toolApproval: { delete_file: "user-approval" }` on the call or the agent (`needsApproval` on the tool is deprecated). The call then ends with a `tool-approval-request`, and resuming after the user answers is a new `generate()` or `stream()` call, so a new trace. Pass the same conversation id to both and they land in the same session as consecutive turns: one approval shows as two turns, the one that asked and the one that ran the tool. The approved `execute_tool` span sits directly under `invoke_agent`, not under a `step`, because the tool runs before the resumed call's first step. + +## Record prompts, responses and tool calls + +Content capture is on by default. With the setup above: + +- `invoke_agent` spans carry the call's `gen_ai.system_instructions`, `gen_ai.input.messages` and `gen_ai.output.messages`; +- each `chat` span carries the messages sent to the model and its reply, plus `gen_ai.tool.definitions`; +- each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. + +All of it is JSON in the GenAI message format, which Maple renders as the transcript. + +To keep content out of Maple, turn it off per call or per agent: + +```ts +telemetry: { + functionId: "billing_agent", + recordInputs: false, // no prompts, system instructions, tool definitions or tool arguments + recordOutputs: false, // no replies or tool results +}, +``` + +Sessions keep their turns, models, tool names, tokens and errors, but the transcript is empty. `telemetry: { isEnabled: false }` drops the call's spans entirely. + +Two more things to know: + +- **Files and images are recorded inline.** An image or PDF in the messages is written as base64 into the `chat` span, and every later step of the loop repeats the history. Keep attachments out of traced calls or set `recordInputs: false` on them. +- **Content is not redacted.** Anything a user types reaches Maple. For pattern-based redaction, run an OpenTelemetry Collector between your app and Maple, or turn recording off for the calls that handle sensitive data. + +`runtimeContext` values are only exported when listed in `includeRuntimeContext`, so user ids, tokens or tenant data you keep there for your tools stay out of the spans. + +## Tools, errors and sub-agents + +Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the model's `gen_ai.tool.call.id`, and the arguments and result. Maple matches each call to the model reply that requested it by that id. + +When a tool's `execute` throws, the AI SDK marks the span ERROR with the error message as the status message, records an `exception` event, and passes the error back to the model as a `tool-error` result. The loop continues, and Maple counts the call as failed: + +```ts +import { tool } from "ai" +import { z } from "zod" + +const fetchTransportData = tool({ + description: "Fetch live public-transport data for a city", + inputSchema: z.object({ city: z.string() }), + execute: async ({ city }): Promise => { + throw new Error(`transport data service unavailable (503) for ${city}`) + }, +}) +``` + +A tool that returns an error value (`return { error: "..." }`) keeps the span green, and Maple counts the call as a success. Throw instead. + +The explicit `Promise` return type matters in TypeScript: without it, an `execute` that always throws infers `never` and the tool fails to typecheck. + +### Sub-agents + +The common multi-agent pattern is agents as tools: the orchestrator has a tool whose `execute` runs another agent. The worker's spans nest under that tool span in the same trace: + +```ts +const weatherWorker = new ToolLoopAgent({ + model: openai("gpt-4o-mini"), + tools: { get_weather: getWeather }, + telemetry: { functionId: "weather_worker" }, +}) + +export const orchestrator = new ToolLoopAgent({ + model: openai("gpt-4o-mini"), + tools: { + delegate_weather: tool({ + description: "Ask the weather worker", + inputSchema: z.object({ task: z.string() }), + execute: async ({ task }) => (await weatherWorker.generate({ prompt: task })).text, + }), + }, + callOptionsSchema: z.object({ conversationId: z.string() }), + prepareCall: ({ options, ...rest }) => ({ + ...rest, + runtimeContext: { conversationId: options.conversationId }, + }), + telemetry: { functionId: "orchestrator", includeRuntimeContext: { conversationId: true } }, +}) +``` + +The worker doesn't need the conversation id: Maple needs it on one span per trace, and the orchestrator's spans have it. It does need its own `functionId`. Maple draws a lane per `gen_ai.agent.name`, and an `execute_tool delegate_weather` span whose only child is `invoke_agent` for `weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. When the model asks for several delegate tools in one reply, the AI SDK runs them concurrently and the lanes overlap in time. + +A pipeline of separate top-level calls (an orchestrator, then a summary agent) produces one trace per call. Pass the same conversation id to each, and they land in one session as consecutive turns. + +## Tokens and cost + +Each `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. With `usage: true`, reasoning tokens are added as `ai.usage.outputTokenDetails.reasoningTokens`. Maple reads all of them. + +The `invoke_agent` span repeats the call's total. Maple counts each model call once: a span's usage is netted against the usage its descendants already reported, so the total isn't doubled, and a sub-agent's tokens stay on the sub-agent's spans. + +Streaming needs nothing extra. The AI SDK normalizes provider usage, so `streamText` and `agent.stream()` calls have token counts too, and the streamed `chat` span records time to first chunk, which Maple shows as TTFT. + +Cost shows as unpriced. The AI SDK doesn't price calls, and Maple never prices tokens itself. Tokens, models and call counts are complete. If your model calls go through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call, and Maple matches them to the AI SDK's `chat` spans by response id. + +## Short-lived processes + +`BatchSpanProcessor` exports every few seconds. A script, a CLI or a serverless function can end before that. Flush before exiting: + +```ts +import { sdk } from "./instrumentation" + +try { + await runConversation() +} finally { + await sdk.shutdown() // flushes, then stops the SDK +} +``` + +In a serverless handler that is reused between invocations, flush without stopping the SDK: + +```ts +import { spanProcessor } from "./instrumentation" + +export async function handler(event: { chatId: string; text: string }) { + try { + return await turn(event.chatId, event.text) + } finally { + await spanProcessor.forceFlush() + } +} +``` + +Streaming matters here. The `invoke_agent` span ends when the stream is fully read, not when `streamText` returns. In a script, read the stream to the end (`for await (const chunk of result.textStream)` or `await result.consumeStream()`) before flushing. A stream nobody reads never ends its spans, and they are never exported. + +On Vercel, `@vercel/otel` flushes at the end of each request through `waitUntil`. Work that runs outside a request, such as a queue consumer, a cron job or a workflow step, doesn't get that flush. Call `forceFlush()` on your span processor at the end of each unit of work there. + +## AI SDK 5 and 6 + +Before AI SDK 7, OpenTelemetry was built into the `ai` package and off by default. Turn it on per call with `experimental_telemetry`, and pass the conversation id as metadata: + +```ts +const result = await generateText({ + model, + messages, + experimental_telemetry: { + isEnabled: true, + functionId: "support_agent", + metadata: { conversationId: chatId }, + }, +}) +``` + +Those versions emit the older span format: `ai.generateText` or `ai.streamText` for the call, `ai.generateText.doGenerate` or `ai.streamText.doStream` for each model request, and `ai.toolCall` for each tool call. The metadata lands as `ai.telemetry.metadata.conversationId`, which Maple doesn't read, so copy it to `gen_ai.conversation.id` with a span processor: + +```ts +import type { Context } from "@opentelemetry/api" +import type { ReadableSpan, Span, SpanProcessor } from "@opentelemetry/sdk-trace-base" + +export class ConversationIdProcessor implements SpanProcessor { + onStart(span: Span, _parentContext: Context) { + const id = span.attributes["ai.telemetry.metadata.conversationId"] + if (typeof id === "string") span.setAttribute("gen_ai.conversation.id", id) + } + onEnd(_span: ReadableSpan) {} + forceFlush() { + return Promise.resolve() + } + shutdown() { + return Promise.resolve() + } +} +``` + +Add it before the exporting processor: `spanProcessors: [new ConversationIdProcessor(), spanProcessor]`. + +Maple reads the old format's models, tokens, prompts and tool calls. The assistant's reply is only recorded as plain text (`ai.response.text`), which Maple doesn't render, so transcripts show the user and tool messages without the final answer. Upgrading to AI SDK 7 fixes that. The `npx @ai-sdk/codemod v7` migration renames `experimental_telemetry` to `telemetry`, but you still need to install `@ai-sdk/otel`, call `registerTelemetry()`, and move the id from `metadata` (gone in AI SDK 7) to `runtimeContext` yourself. Drop the span processor once you have. + +## Check that it works + +Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: + +- one session per conversation, with the id you passed as `conversationId`, and framework **Vercel AI SDK**; +- one turn per `generate()` or `stream()` call, each labeled with the user's message, and a transcript with the prompts, replies and tool calls; +- per turn, an `invoke_agent ` span, a `step ` span per loop iteration, a `chat ` span per model request and an `execute_tool ` span per tool call; +- the agent name from `functionId` in the agent filter, and a lane per sub-agent; +- token counts on every model call, including streamed ones, and TTFT on streamed calls; +- failed tool calls marked as failed, with the error message; +- cost shown as unpriced. + +Span names carry the model id, not the agent name, so two agents on the same model have identically named `invoke_agent` spans. The agent name is on the span as `gen_ai.agent.name`. + +## Troubleshooting + +- **No AI spans at all.** `registerTelemetry()` was never called, or ran in a module that isn't loaded. AI SDK 7 emits nothing without it, and `experimental_telemetry: { isEnabled: true }` alone does nothing. Import `@ai-sdk/otel` and register `new OpenTelemetry()` at startup. +- **Every message is its own session.** No `gen_ai.conversation.id` on the spans. Check all three parts: `enrichSpan` on the integration, `runtimeContext: { conversationId }` on the call (or `prepareCall` for an agent), and `includeRuntimeContext: { conversationId: true }`. Without the last one, `enrichSpan` receives an empty object. +- **Sessions show framework "Unidentified".** The spans carry no `ai.*` attributes. Set `usage: true` and `runtimeContext: true` on `new OpenTelemetry()`, and don't pass a custom `tracer` with another name. +- **Every span shows up twice.** `registerTelemetry()` ran twice, both `OpenTelemetry` and `LegacyOpenTelemetry` are registered, or a second OpenTelemetry SDK exports the same spans (Sentry without `skipOpenTelemetrySetup: true` next to `@vercel/otel`, for example). Register one integration and one exporter. +- **One call has no spans while others do.** It passes `telemetry.integrations`, which replaces the globally registered integrations for that call. Add the `OpenTelemetry` instance to that list, or remove the option. +- **Nothing arrives from a script or function.** The process ended before the batch was exported. Call `sdk.shutdown()` or `spanProcessor.forceFlush()` in a `finally`. +- **A streamed turn is missing, or its spans have no tokens.** The stream was never read to the end, so its spans never ended. Consume the stream before flushing. Before `ai` 7.0.106, a provider error in the middle of a stream also left the spans open; upgrade. +- **Sub-agents share one lane, or have none.** The worker agents have no `functionId`. Set a distinct one on each. +- **A failed tool shows as successful.** The tool returned an error value. Throw from `execute`. +- **Spans arrive with no prompts or replies.** `recordInputs: false` or `recordOutputs: false` is set on that call or agent. +- **Exports fail with 413.** Images or documents in the messages are recorded as base64 on every step. Keep them out of traced calls or set `recordInputs: false` there. +- **Transcripts end at the tool results, with no final answer.** You are on AI SDK 5 or 6, which records the reply as plain text. Upgrade to AI SDK 7. + +## Related + +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [All agent tracing guides](/docs/agent-tracing) +- [AI SDK telemetry](https://ai-sdk.dev/docs/ai-sdk-core/telemetry) +- [Next.js instrumentation](/docs/guides/instrumentation-nextjs) +- [Node.js instrumentation](/docs/guides/instrumentation-nodejs) +- [OpenRouter Broadcast](/docs/agent-tracing/openrouter) diff --git a/apps/landing/src/lib/agent-tracing-guides.ts b/apps/landing/src/lib/agent-tracing-guides.ts new file mode 100644 index 0000000000..3f291a7ca1 --- /dev/null +++ b/apps/landing/src/lib/agent-tracing-guides.ts @@ -0,0 +1,69 @@ +// Cards on /docs/agent-tracing, one per framework or gateway guide, grouped by +// ecosystem. Every entry is a shipping page under content/docs/agent-tracing. +import type { GuideCard } from "./instrumentation-guides" +import type { BrandMarkId } from "./brand-marks" + +const guide = (slug: string, name: string, hint: string, mark: BrandMarkId): GuideCard => ({ + name, + hint, + href: `/docs/agent-tracing/${slug}`, + icon: { mark }, +}) + +export interface AgentGuideSection { + id: string + title: string + cards: readonly GuideCard[] +} + +export const AGENT_GUIDE_SECTIONS: readonly AgentGuideSection[] = [ + { + id: "typescript", + title: "TypeScript & JavaScript", + cards: [ + guide("vercel-ai-sdk", "Vercel AI SDK", "generateText, streamText and Agent", "vercel"), + guide("mastra", "Mastra", "Agents, workflows and memory threads", "mastra"), + guide("claude-agent-sdk", "Claude Agent SDK & Claude Code", "TypeScript, Python and the CLI", "claude"), + ], + }, + { + id: "python", + title: "Python", + cards: [ + guide("openai-agents", "OpenAI Agents SDK", "Handoffs and agents as tools", "openai"), + guide("langchain", "LangChain & LangGraph", "Graphs, threads and interrupts", "langchain"), + guide("pydantic-ai", "Pydantic AI", "Built-in GenAI instrumentation", "pydantic"), + guide("crewai", "CrewAI", "Crews, flows and delegation", "crewai"), + guide("google-adk", "Google ADK", "Runners, sub-agents and sessions", "googleadk"), + guide("llamaindex", "LlamaIndex", "FunctionAgent and AgentWorkflow", "llamaindex"), + guide("strands", "Strands Agents", "Swarms, graphs and agents as tools", "strands"), + guide("smolagents", "smolagents", "CodeAgent and managed agents", "huggingface"), + guide("agno", "Agno", "Agents and teams", "agno"), + guide("dspy", "DSPy", "Modules, ReAct and optimizers", "python"), + guide("haystack", "Haystack", "Agent component and pipelines", "haystack"), + ], + }, + { + id: "jvm-dotnet", + title: "Java & .NET", + cards: [ + guide("spring-ai", "Spring AI", "ChatClient, advisors and tools", "spring"), + guide( + "microsoft-agent-framework", + "Microsoft Agent Framework", + "And Semantic Kernel · Python and .NET", + "dotnet", + ), + ], + }, + { + id: "gateways", + title: "Gateways & provider SDKs", + cards: [ + guide("openrouter", "OpenRouter", "Broadcast traces from the gateway", "openrouter"), + guide("litellm", "LiteLLM", "SDK and proxy", "litellm"), + guide("provider-sdks", "OpenAI, Anthropic & Gemini SDKs", "Your own agent loop", "openai"), + guide("opentelemetry", "Any language", "Emit the OpenTelemetry GenAI conventions", "opentelemetry"), + ], + }, +] diff --git a/apps/landing/src/lib/brand-marks.ts b/apps/landing/src/lib/brand-marks.ts index 2b5600c4bb..0f3f32ad6d 100644 --- a/apps/landing/src/lib/brand-marks.ts +++ b/apps/landing/src/lib/brand-marks.ts @@ -131,6 +131,91 @@ export const BRAND_MARKS = { hex: "#000000", path: "M13.85 0a4.16 4.16 0 0 0-2.95 1.217L1.456 10.66a.835.835 0 0 0 0 1.18.835.835 0 0 0 1.18 0l9.442-9.442a2.49 2.49 0 0 1 3.541 0 2.49 2.49 0 0 1 0 3.541L8.59 12.97l-.1.1a.835.835 0 0 0 0 1.18.835.835 0 0 0 1.18 0l.1-.098 7.03-7.034a2.49 2.49 0 0 1 3.542 0l.049.05a2.49 2.49 0 0 1 0 3.54l-8.54 8.54a1.96 1.96 0 0 0 0 2.755l1.753 1.753a.835.835 0 0 0 1.18 0 .835.835 0 0 0 0-1.18l-1.753-1.753a.266.266 0 0 1 0-.394l8.54-8.54a4.185 4.185 0 0 0 0-5.9l-.05-.05a4.16 4.16 0 0 0-2.95-1.218c-.2 0-.401.02-.6.048a4.17 4.17 0 0 0-1.17-3.552A4.16 4.16 0 0 0 13.85 0m0 3.333a.84.84 0 0 0-.59.245L6.275 10.56a4.186 4.186 0 0 0 0 5.902 4.186 4.186 0 0 0 5.902 0L19.16 9.48a.835.835 0 0 0 0-1.18.835.835 0 0 0-1.18 0l-6.985 6.984a2.49 2.49 0 0 1-3.54 0 2.49 2.49 0 0 1 0-3.54l6.983-6.985a.835.835 0 0 0 0-1.18.84.84 0 0 0-.59-.245", }, + // AI agent frameworks and gateways, for /docs/agent-tracing. simple-icons (CC0) or + // lobe-icons (MIT) where they exist; Agno and Strands use their own marks, + // LiteLLM (which has none) the MDI train-variant glyph. + vercel: { + name: "Vercel", + hex: "#000000", + path: "m12 1.608 12 20.784H0Z", + }, + mastra: { + name: "Mastra", + hex: "#000000", + path: "M14.74 9.977l-1.584 1.583h3.563v.942h-3.563l1.583 1.583-.666.666-2.053-2.054-2.054 2.054-.666-.666 1.583-1.583H7.32v-.942h3.562L9.299 9.977l.666-.666 2.055 2.054 2.054-2.054.666.666z M11.99.083c6.575 0 11.905 5.332 11.905 11.907 0 3.198-1.263 6.099-3.315 8.237a5.727 5.727 0 01-.353.353 11.862 11.862 0 01-8.237 3.316C5.415 23.895.084 18.566.084 11.99c0-3.188 1.255-6.081 3.296-8.219a5.75 5.75 0 01.393-.394A11.864 11.864 0 0111.99.083zM1.93 16.354c1.687 3.884 5.556 6.6 10.06 6.6 1.42 0 2.776-.274 4.022-.765a9.709 9.709 0 01-2.23-.186c-1.665-.328-3.396-1.069-5.036-2.18-1.946-.374-3.695-1.074-5.104-2.02a9.703 9.703 0 01-1.712-1.448zm19.154.899a10.46 10.46 0 01-.746.55c-2.157 1.447-5.11 2.327-8.349 2.327-.296 0-.59-.011-.88-.025.96.463 1.922.791 2.854.975 2.345.462 4.427.015 5.78-1.338.669-.67 1.116-1.518 1.341-2.49zM11.989 4.79c-3.077 0-5.841.838-7.823 2.167a8.863 8.863 0 00-1.44 1.195 8.86 8.86 0 00.172 1.864c.462 2.34 1.826 4.888 4.001 7.064.705.705 1.45 1.323 2.213 1.855.916.164 1.88.254 2.877.254 3.077 0 5.843-.838 7.825-2.168a8.858 8.858 0 001.438-1.194 8.865 8.865 0 00-.172-1.864c-.461-2.341-1.824-4.888-4-7.064a16.323 16.323 0 00-2.213-1.855 16.32 16.32 0 00-2.878-.254zM1.836 9.278c-.528.846-.81 1.764-.81 2.711 0 1.913 1.155 3.701 3.14 5.032.788.53 1.701.978 2.708 1.33a17.78 17.78 0 01-.64-.606c-2.29-2.29-3.756-5-4.258-7.548a10.48 10.48 0 01-.14-.919zm15.27-3.65c.216.195.43.396.64.605 2.29 2.29 3.756 5 4.258 7.548.06.308.106.614.139.917.527-.845.811-1.762.811-2.709 0-1.913-1.156-3.7-3.14-5.032-.789-.529-1.701-.978-2.708-1.33zM11.99 1.025c-1.42 0-2.777.272-4.022.764a9.707 9.707 0 012.23.186c1.665.328 3.398 1.07 5.038 2.181 1.945.374 3.694 1.075 5.102 2.02a9.704 9.704 0 011.71 1.448A10.965 10.965 0 0011.99 1.025zm-1.974 1.873c-2.345-.462-4.427-.014-5.78 1.338-.67.67-1.117 1.519-1.342 2.49.237-.192.487-.375.748-.55C5.798 4.729 8.75 3.85 11.989 3.85c.296 0 .589.009.88.023a11.917 11.917 0 00-2.853-.975z", + }, + openai: { + name: "OpenAI", + hex: "#000000", + path: "M22.2819 9.8211a5.9847 5.9847 0 0 0-.5157-4.9108 6.0462 6.0462 0 0 0-6.5098-2.9A6.0651 6.0651 0 0 0 4.9807 4.1818a5.9847 5.9847 0 0 0-3.9977 2.9 6.0462 6.0462 0 0 0 .7427 7.0966 5.98 5.98 0 0 0 .511 4.9107 6.051 6.051 0 0 0 6.5146 2.9001A5.9847 5.9847 0 0 0 13.2599 24a6.0557 6.0557 0 0 0 5.7718-4.2058 5.9894 5.9894 0 0 0 3.9977-2.9001 6.0557 6.0557 0 0 0-.7475-7.0729zm-9.022 12.6081a4.4755 4.4755 0 0 1-2.8764-1.0408l.1419-.0804 4.7783-2.7582a.7948.7948 0 0 0 .3927-.6813v-6.7369l2.02 1.1686a.071.071 0 0 1 .038.052v5.5826a4.504 4.504 0 0 1-4.4945 4.4944zm-9.6607-4.1254a4.4708 4.4708 0 0 1-.5346-3.0137l.142.0852 4.783 2.7582a.7712.7712 0 0 0 .7806 0l5.8428-3.3685v2.3324a.0804.0804 0 0 1-.0332.0615L9.74 19.9502a4.4992 4.4992 0 0 1-6.1408-1.6464zM2.3408 7.8956a4.485 4.485 0 0 1 2.3655-1.9728V11.6a.7664.7664 0 0 0 .3879.6765l5.8144 3.3543-2.0201 1.1685a.0757.0757 0 0 1-.071 0l-4.8303-2.7865A4.504 4.504 0 0 1 2.3408 7.872zm16.5963 3.8558L13.1038 8.364 15.1192 7.2a.0757.0757 0 0 1 .071 0l4.8303 2.7913a4.4944 4.4944 0 0 1-.6765 8.1042v-5.6772a.79.79 0 0 0-.407-.667zm2.0107-3.0231l-.142-.0852-4.7735-2.7818a.7759.7759 0 0 0-.7854 0L9.409 9.2297V6.8974a.0662.0662 0 0 1 .0284-.0615l4.8303-2.7866a4.4992 4.4992 0 0 1 6.6802 4.66zM8.3065 12.863l-2.02-1.1638a.0804.0804 0 0 1-.038-.0567V6.0742a4.4992 4.4992 0 0 1 7.3757-3.4537l-.142.0805L8.704 5.459a.7948.7948 0 0 0-.3927.6813zm1.0976-2.3654l2.602-1.4998 2.6069 1.4998v2.9994l-2.5974 1.4997-2.6067-1.4997Z", + }, + langchain: { + name: "LangChain", + hex: "#7FC8FF", + path: "M13.796 0a6.93 6.93 0 0 0-4.91 2.019L5.451 5.455l3.273 3.27 3.432-3.432a2.284 2.284 0 0 1 3.277 0 2.28 2.28 0 0 1 0 3.275L12 12.001l3.273 3.273 3.433-3.435c2.692-2.692 2.692-7.127 0-9.82A6.92 6.92 0 0 0 13.796 0m-5.07 8.728-3.433 3.434c-2.692 2.693-2.692 7.126 0 9.819A6.92 6.92 0 0 0 10.203 24a6.93 6.93 0 0 0 4.911-2.02l3.432-3.432-3.271-3.272-3.433 3.433a2.284 2.284 0 0 1-3.277 0 2.28 2.28 0 0 1 0-3.276L12 12z", + }, + pydantic: { + name: "Pydantic", + hex: "#E92063", + path: "m23.826 17.316-4.23-5.866-6.847-9.496c-.348-.48-1.151-.48-1.497 0l-6.845 9.494-4.233 5.868a.925.925 0 0 0 .46 1.417l11.078 3.626h.002a.92.92 0 0 0 .572 0h.002l11.077-3.626c.28-.092.5-.31.59-.592a.916.916 0 0 0-.13-.825h.002ZM12.001 4.07l4.44 6.158-4.152-1.36c-.032-.01-.066-.008-.098-.016a.8.8 0 0 0-.096-.016c-.032-.004-.062-.016-.094-.016s-.062.012-.094.016a.74.74 0 0 0-.096.016c-.032.006-.066.006-.096.016L7.59 10.221l-.026.008 4.44-6.158h-.002Zm-6.273 8.7 4.834-1.583.516-.168v9.19L2.41 17.372l3.317-4.6Zm7.197 7.437V11.02l5.35 1.752 3.316 4.598-8.666 2.838Z", + }, + crewai: { + name: "CrewAI", + hex: "#FF5A50", + path: "M12.482.18C7.161 1.319 1.478 9.069 1.426 15.372c-.051 5.527 3.1 8.68 8.68 8.627 6.716-.05 14.259-6.87 12.09-10.9-.672-1.292-1.396-1.344-2.687-.207-1.602 1.395-1.654.31-.207-2.893 1.757-3.98 1.705-5.322-.31-7.544C17.03.388 14.962-.388 12.482.181Zm5.322 2.068c2.273 2.015 2.376 4.236.465 8.42-1.395 3.1-2.17 3.515-3.824 1.86-1.24-1.24-1.343-3.46-.258-6.044 1.137-2.635.982-3.1-.568-1.653-3.72 3.358-6.458 9.765-5.424 12.503.464 1.189.825 1.395 2.737 1.395 2.79 0 6.303-1.705 7.957-3.926 1.756-2.274 2.79-2.274 2.79-.052 0 3.875-6.459 8.627-11.625 8.627-6.251 0-9.351-4.752-7.491-11.47.878-2.995 4.443-7.904 7.077-9.66 3.255-2.17 5.684-2.17 8.164 0z", + }, + googleadk: { + name: "Google ADK", + hex: "#4285F4", + path: "M12.48 10.92v3.28h7.84c-.24 1.84-.853 3.187-1.787 4.133-1.147 1.147-2.933 2.4-6.053 2.4-4.827 0-8.6-3.893-8.6-8.72s3.773-8.72 8.6-8.72c2.6 0 4.507 1.027 5.907 2.347l2.307-2.307C18.747 1.44 16.133 0 12.48 0 5.867 0 .307 5.387.307 12s5.56 12 12.173 12c3.573 0 6.267-1.173 8.373-3.36 2.16-2.16 2.84-5.213 2.84-7.667 0-.76-.053-1.467-.173-2.053H12.48z", + }, + llamaindex: { + name: "LlamaIndex", + hex: "#BC8DEB", + path: "M15.855 17.122c-2.092.924-4.358.545-5.23.24 0 .21-.01.857-.048 1.78-.038.924-.332 1.507-.475 1.684.016.577.029 1.837-.047 2.26a1.93 1.93 0 01-.476.914H8.295c.114-.577.555-.946.761-1.058.114-1.193-.11-2.229-.238-2.597-.126.449-.437 1.49-.665 2.068a6.418 6.418 0 01-.713 1.299h-.951c-.048-.578.27-.77.475-.77.095-.177.323-.731.476-1.54.152-.807-.064-2.324-.19-2.981v-2.068c-1.522-.818-2.092-1.636-2.473-2.55-.304-.73-.222-1.843-.142-2.308-.096-.176-.373-.625-.476-1.25-.142-.866-.063-1.491 0-1.828-.095-.096-.285-.587-.285-1.78 0-1.192.349-1.811.523-1.972v-.529c-.666-.048-1.331-.336-1.712-.721-.38-.385-.095-.962.143-1.154.238-.193.475-.049.808-.145.333-.096.618-.192.76-.48C4.512 1.403 4.287.448 4.16 0c.57.077.935.577 1.046.818V0c.713.337 1.997 1.154 2.425 2.934.342 1.424.586 4.409.665 5.723 1.823.016 4.137-.26 6.229.193 1.901.412 2.757 1.25 3.755 1.25.999 0 1.57-.577 2.282-.096.714.481 1.094 1.828.999 2.838-.076.808-.697 1.074-.998 1.106-.38 1.27 0 2.485.237 2.934v1.827c.111.16.333.655.333 1.347 0 .693-.222 1.154-.333 1.299.19 1.077-.08 2.18-.238 2.597h-1.283c.152-.385.412-.481.523-.481.228-1.193.063-2.293-.048-2.693-.722-.424-1.188-1.17-1.331-1.491.016.272-.029 1.029-.333 1.875-.304.847-.76 1.347-.95 1.491v1.01h-1.284c0-.615.348-.737.523-.721.222-.4.76-1.01.76-2.212 0-1.015-.713-1.492-1.236-2.405-.248-.434-.127-.978-.047-1.203z", + }, + strands: { + name: "Strands Agents", + hex: "#00FF77", + viewBox: "-86.5 0 463 463", + path: "M259.147 0.981812C271.389 -2.57498 284.197 4.46571 287.754 16.7074C291.311 28.9492 284.27 41.757 272.028 45.3138L71.1727 103.671C40.7142 112.521 37.1976 154.262 65.7459 168.083L241.343 253.093C307.872 285.302 299.794 382.546 228.862 403.336L30.4041 461.502C18.1707 465.088 5.34708 458.078 1.76153 445.844C-1.8239 433.611 5.18637 420.787 17.4197 417.202L215.878 359.035C246.277 350.125 249.739 308.449 221.226 294.645L45.6297 209.635C-20.9834 177.386 -12.7772 79.9893 58.2928 59.3402L259.147 0.981812Z", + }, + huggingface: { + name: "Hugging Face", + hex: "#FFD21E", + path: "M12.025 1.13c-5.77 0-10.449 4.647-10.449 10.378 0 1.112.178 2.181.503 3.185.064-.222.203-.444.416-.577a.96.96 0 0 1 .524-.15c.293 0 .584.124.84.284.278.173.48.408.71.694.226.282.458.611.684.951v-.014c.017-.324.106-.622.264-.874s.403-.487.762-.543c.3-.047.596.06.787.203s.31.313.4.467c.15.257.212.468.233.542.01.026.653 1.552 1.657 2.54.616.605 1.01 1.223 1.082 1.912.055.537-.096 1.059-.38 1.572.637.121 1.294.187 1.967.187.657 0 1.298-.063 1.921-.178-.287-.517-.44-1.041-.384-1.581.07-.69.465-1.307 1.081-1.913 1.004-.987 1.647-2.513 1.657-2.539.021-.074.083-.285.233-.542.09-.154.208-.323.4-.467a1.08 1.08 0 0 1 .787-.203c.359.056.604.29.762.543s.247.55.265.874v.015c.225-.34.457-.67.683-.952.23-.286.432-.52.71-.694.257-.16.547-.284.84-.285a.97.97 0 0 1 .524.151c.228.143.373.388.43.625l.006.04a10.3 10.3 0 0 0 .534-3.273c0-5.731-4.678-10.378-10.449-10.378M8.327 6.583a1.5 1.5 0 0 1 .713.174 1.487 1.487 0 0 1 .617 2.013c-.183.343-.762-.214-1.102-.094-.38.134-.532.914-.917.71a1.487 1.487 0 0 1 .69-2.803m7.486 0a1.487 1.487 0 0 1 .689 2.803c-.385.204-.536-.576-.916-.71-.34-.12-.92.437-1.103.094a1.487 1.487 0 0 1 .617-2.013 1.5 1.5 0 0 1 .713-.174m-10.68 1.55a.96.96 0 1 1 0 1.921.96.96 0 0 1 0-1.92m13.838 0a.96.96 0 1 1 0 1.92.96.96 0 0 1 0-1.92M8.489 11.458c.588.01 1.965 1.157 3.572 1.164 1.607-.007 2.984-1.155 3.572-1.164.196-.003.305.12.305.454 0 .886-.424 2.328-1.563 3.202-.22-.756-1.396-1.366-1.63-1.32q-.011.001-.02.006l-.044.026-.01.008-.03.024q-.018.017-.035.036l-.032.04a1 1 0 0 0-.058.09l-.014.025q-.049.088-.11.19a1 1 0 0 1-.083.116 1.2 1.2 0 0 1-.173.18q-.035.029-.075.058a1.3 1.3 0 0 1-.251-.243 1 1 0 0 1-.076-.107c-.124-.193-.177-.363-.337-.444-.034-.016-.104-.008-.2.022q-.094.03-.216.087-.06.028-.125.063l-.13.074q-.067.04-.136.086a3 3 0 0 0-.135.096 3 3 0 0 0-.26.219 2 2 0 0 0-.12.121 2 2 0 0 0-.106.128l-.002.002a2 2 0 0 0-.09.132l-.001.001a1.2 1.2 0 0 0-.105.212q-.013.036-.024.073c-1.139-.875-1.563-2.317-1.563-3.203 0-.334.109-.457.305-.454m.836 10.354c.824-1.19.766-2.082-.365-3.194-1.13-1.112-1.789-2.738-1.789-2.738s-.246-.945-.806-.858-.97 1.499.202 2.362c1.173.864-.233 1.45-.685.64-.45-.812-1.683-2.896-2.322-3.295s-1.089-.175-.938.647 2.822 2.813 2.562 3.244-1.176-.506-1.176-.506-2.866-2.567-3.49-1.898.473 1.23 2.037 2.16c1.564.932 1.686 1.178 1.464 1.53s-3.675-2.511-4-1.297c-.323 1.214 3.524 1.567 3.287 2.405-.238.839-2.71-1.587-3.216-.642-.506.946 3.49 2.056 3.522 2.064 1.29.33 4.568 1.028 5.713-.624m5.349 0c-.824-1.19-.766-2.082.365-3.194 1.13-1.112 1.789-2.738 1.789-2.738s.246-.945.806-.858.97 1.499-.202 2.362c-1.173.864.233 1.45.685.64.451-.812 1.683-2.896 2.322-3.295s1.089-.175.938.647-2.822 2.813-2.562 3.244 1.176-.506 1.176-.506 2.866-2.567 3.49-1.898-.473 1.23-2.037 2.16c-1.564.932-1.686 1.178-1.464 1.53s3.675-2.511 4-1.297c.323 1.214-3.524 1.567-3.287 2.405.238.839 2.71-1.587 3.216-.642.506.946-3.49 2.056-3.522 2.064-1.29.33-4.568 1.028-5.713-.624", + }, + agno: { + name: "Agno", + hex: "#FF4017", + viewBox: "17.7 18.8 80 80", + path: "M21.7395 79.3576V89.966H52.6926V79.3576H21.7395Z M38.8391 27.6018V38.2111H61.6887L81.1604 89.966H93.7395L69.0686 27.6018H38.8391Z", + }, + haystack: { + name: "Haystack", + hex: "#0EAF9C", + path: "M2.0084 0C.8992 0 0 .8992 0 2.0084v19.9832C0 23.1006.8992 24 2.0084 24h19.9832C23.1006 24 24 23.1007 24 21.9916V2.0084C24 .8992 23.1007 0 21.9916 0Zm9.9624 3.84c3.4303 0 6.2108 2.7626 6.2108 6.1709v6.4875a.2688.2688 0 0 1-.2697.2681c-1.3425 0-2.4306-1.0811-2.4306-2.415v-4.3409c0-1.9265-1.572-3.488-3.5105-3.488s-3.424 1.562-3.424 3.488v1.608a.2633.2633 0 0 0 .259.2681h1.5394a.2693.2693 0 0 0 .2753-.263V9.9453c0-.7412.6044-1.3414 1.3503-1.3414s1.3502.6002 1.3502 1.3414V20.029a.2747.2747 0 0 1-.2807.2682c-1.3362 0-2.4198-1.0766-2.4198-2.4043v-3.2307a.2747.2747 0 0 0-.2753-.268H8.8114a.2637.2637 0 0 0-.2646.263v1.0789c0 1.3338-1.1746 2.4152-2.517 2.4152a.2688.2688 0 0 1-.2698-.268v-7.8724c0-3.4083 2.7805-6.1709 6.2108-6.1709Z", + }, + spring: { + name: "Spring", + hex: "#6DB33F", + path: "M21.8537 1.4158a10.4504 10.4504 0 0 1-1.284 2.2471A11.9666 11.9666 0 1 0 3.8518 20.7757l.4445.3951a11.9543 11.9543 0 0 0 19.6316-8.2971c.3457-3.0126-.568-6.8649-2.0743-11.458zM5.5805 20.8745a1.0174 1.0174 0 1 1-.1482-1.4323 1.0396 1.0396 0 0 1 .1482 1.4323zm16.1991-3.5806c-2.9385 3.9263-9.2601 2.5928-13.2852 2.7904 0 0-.7161.0494-1.4323.1481 0 0 .2717-.1234.6174-.2469 2.8398-.9877 4.1732-1.1853 5.9018-2.0743 3.2349-1.6545 6.4698-5.2844 7.1118-9.0379-1.2347 3.6053-4.9881 6.7167-8.3959 7.9761-2.3459.8643-6.5685 1.7039-6.5685 1.7039l-.1729-.0988c-2.8645-1.4076-2.9632-7.6304 2.2718-9.6306 2.2966-.889 4.4696-.395 6.9637-.9877 2.6422-.6174 5.7043-2.5929 6.939-5.1857 1.3828 4.1732 3.062 10.643.0493 14.6434z", + }, + dotnet: { + name: ".NET", + hex: "#512BD4", + path: "M24 8.77h-2.468v7.565h-1.425V8.77h-2.462V7.53H24zm-6.852 7.565h-4.821V7.53h4.63v1.24h-3.205v2.494h2.953v1.234h-2.953v2.604h3.396zm-6.708 0H8.882L4.78 9.863a2.896 2.896 0 0 1-.258-.51h-.036c.032.189.048.592.048 1.21v5.772H3.157V7.53h1.659l3.965 6.32c.167.261.275.442.323.54h.024c-.04-.233-.06-.629-.06-1.185V7.529h1.372zm-8.703-.693a.868.829 0 0 1-.869.829.868.829 0 0 1-.868-.83.868.829 0 0 1 .868-.828.868.829 0 0 1 .869.829Z", + }, + openrouter: { + name: "OpenRouter", + hex: "#94A3B8", + path: "M16.778 1.844v1.919q-.569-.026-1.138-.032-.708-.008-1.415.037c-1.93.126-4.023.728-6.149 2.237-2.911 2.066-2.731 1.95-4.14 2.75-.396.223-1.342.574-2.185.798-.841.225-1.753.333-1.751.333v4.229s.768.108 1.61.333c.842.224 1.789.575 2.185.799 1.41.798 1.228.683 4.14 2.75 2.126 1.509 4.22 2.11 6.148 2.236.88.058 1.716.041 2.555.005v1.918l7.222-4.168-7.222-4.17v2.176c-.86.038-1.611.065-2.278.021-1.364-.09-2.417-.357-3.979-1.465-2.244-1.593-2.866-2.027-3.68-2.508.889-.518 1.449-.906 3.822-2.59 1.56-1.109 2.614-1.377 3.978-1.466.667-.044 1.418-.017 2.278.02v2.176L24 6.014Z", + }, + litellm: { + name: "LiteLLM", + hex: "#4F46E5", + path: "M18,10H6V5H18M12,17C10.89,17 10,16.1 10,15C10,13.89 10.89,13 12,13A2,2 0 0,1 14,15A2,2 0 0,1 12,17M4,15.5A3.5,3.5 0 0,0 7.5,19L6,20.5V21H18V20.5L16.5,19A3.5,3.5 0 0,0 20,15.5V5C20,1.5 16.42,1 12,1C7.58,1 4,1.5 4,5V15.5Z", + }, warpstream: { name: "WarpStream", hex: "#E32645", diff --git a/apps/landing/src/lib/docs-nav.ts b/apps/landing/src/lib/docs-nav.ts index 47e2e603b4..154a6f3a68 100644 --- a/apps/landing/src/lib/docs-nav.ts +++ b/apps/landing/src/lib/docs-nav.ts @@ -31,8 +31,8 @@ export const SECTIONS = [ id: "instrumentation", icon: "Instrumentation", label: "Instrumentation", - blurb: "Languages, frameworks, hosts and clusters.", - groups: ["Instrumentation", "Infrastructure"], + blurb: "Languages, frameworks, AI agents, hosts and clusters.", + groups: ["Instrumentation", "AI Agents", "Infrastructure"], }, { id: "local", @@ -74,6 +74,7 @@ export const INSTRUMENTATION_SLUG = "instrumentation" export const GROUP_BLURBS = { "Getting Started": "What Maple is and the three steps to first data.", Instrumentation: "Setup guides for every language, framework and runtime.", + "AI Agents": "Trace an agent framework or LLM gateway so its conversations show up as Agent Sessions.", Concepts: "How Maple reads OpenTelemetry data and what it expects from yours.", Explore: "Search traces, logs and metrics, and read services and the service map.", Errors: "How errors become issues, and how to triage and resolve them.", diff --git a/skills/maple-agent-tracing-agno/SKILL.md b/skills/maple-agent-tracing-agno/SKILL.md new file mode 100644 index 0000000000..4e58e3f759 --- /dev/null +++ b/skills/maple-agent-tracing-agno/SKILL.md @@ -0,0 +1,185 @@ +--- +name: maple-agent-tracing-agno +description: "Trace Agno agents with Maple: installs the OpenInference Agno instrumentor with an OTLP exporter and GenAI attributes, and sets session_id per conversation so each chat is one Maple Agent Session with transcript, tool calls, team members, tokens and cost. Triggers on 'trace my agno agent', 'add Maple to agno', 'agent sessions for agno', 'OpenTelemetry for agno'." +--- + +# Maple agent tracing for Agno + +Goal: every conversation with the Agno app shows up in Maple **Agent Sessions** as exactly one session, one turn per `run()`, with transcript, model calls, tool calls (failures marked), team-member lanes, tokens and cost where the provider returns it. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/agno + +Mechanism: `openinference-instrumentation-agno` (the instrumentor Agno's own `setup_tracing()` uses) + OTel SDK + OTLP/HTTP exporter to Maple. Agno's `setup_tracing(db=...)` and `AgentOS(tracing=True)` only write to the AgentOS database; they never export OTLP. + +## Step 0: Detect versions and existing setup + +1. Find the Python project file (`pyproject.toml`, `requirements*.txt`, `uv.lock`, `poetry.lock`) and the installed `agno` version. Target `agno>=3.0` (verified 3.0.11). On Agno 2.x, `openinference-instrumentation-agno` needs `agno>=2.5.0`; the rest of this skill applies unchanged. +2. Grep for existing tracing: `TracerProvider(`, `set_tracer_provider`, `AgnoInstrumentor`, `setup_tracing(`, `tracing=True`, `phoenix.otel.register`, `openlit.init`, `logfire.configure`, `langfuse`, `OpenAIInstrumentor`, `LiteLLMInstrumentor`. + - Existing `TracerProvider` of the app's own: reuse it. Add a `BatchSpanProcessor(OTLPSpanExporter(...))` for Maple to it. Do not create a second provider. + - Existing `AgnoInstrumentor().instrument(...)`: edit that call (add `config=`); never call `instrument()` a second time (the second call is a silent no-op). + - `AgentOS(tracing=True)` or `setup_tracing(db=...)`: see Step 2c. + - Any other LLM instrumentor on the same calls (OpenAI/LiteLLM OpenInference instrumentors, OpenLIT, `register(auto_instrument=True)`): remove it or scope it away from Agno, or every model call is recorded twice. +3. Find every place an `Agent`, `Team` or `Workflow` is run: `.run(`, `.arun(`, `.print_response(`, `.aprint_response(`, `.continue_run(`, `.acontinue_run(`. Note where the conversation/thread id lives in the request. + +## Step 1: Key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. Protocol `http/protobuf`. +- Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from **Settings → Ingestion**. +- Private `maple_sk_` keys never go in browser code. Agno runs server-side; a `maple_pk_` ingest key is write-only. +- Follow the repo's existing secret/env convention (`.env`, settings module, secret manager). If there is none, inline is acceptable because ingest keys are write-only. + +## Step 2: Install and initialize + +### 2a. Packages + +Add to the project's dependency file with its package manager (uv/poetry/pip): + +``` +agno>=3.0 +openinference-instrumentation-agno>=1.0.10 +opentelemetry-sdk +opentelemetry-exporter-otlp-proto-http +``` + +`>=1.0.8` is required for human-in-the-loop `continue_run()` spans; `1.0.10` adds `llm.cost.total`. If the app uses `SqliteDb`/`AsyncSqliteDb` and imports fail with "requires ... 'greenlet'", add `greenlet` (and `aiosqlite` or `agno[sqlite]`). + +### 2b. Environment + +```bash +OTEL_SERVICE_NAME= +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name= +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev # EU: https://ingest.eu.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +AGNO_TELEMETRY=false +``` + +The exporter appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. If you use `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` instead, give the full `.../v1/traces` URL. `AGNO_TELEMETRY=false` disables Agno's anonymous usage pings (unrelated to OTel). + +### 2c. Tracing module + +Create `tracing.py` (or add to the app's existing observability module): + +```py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.agno import AgnoInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider() # resource from OTEL_SERVICE_NAME / OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +AgnoInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +- `enable_genai_semconv=True` is REQUIRED. It dual-writes `gen_ai.*` (messages in `{role, parts}` form, usage, tool name/args/result, agent name, `gen_ai.operation.name`). Without it Maple's session detail page has no transcript for Agno. Env equivalent: `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` (use it when you can't edit the `instrument()` call, e.g. AgentOS owns it). +- Import `tracing` at the top of the entry point (`main.py`, `app.py`, the ASGI module), before building agents, teams or `AgentOS`. +- AgentOS: `AgentOS(tracing=True)` and `setup_tracing(db=...)` skip their setup when a real `TracerProvider` is already registered, so AgentOS's traces view stops getting spans once `tracing.py` runs first. If the user wants to keep that view, add Agno's DB exporter to the same provider (use the `db` the AgentOS uses): + + ```py + from agno.tracing.exporter import DatabaseSpanExporter + provider.add_span_processor(BatchSpanProcessor(DatabaseSpanExporter(db=db))) + ``` + + If `setup_tracing`/`tracing=True` must run first instead, don't create a provider: call `trace.get_tracer_provider().add_span_processor(BatchSpanProcessor(OTLPSpanExporter()))` after it and set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` in the environment before the process starts. + +## Step 3: Session id (one conversation = one session) + +Maple groups Agno traces by `session.id` on the run span. Agno sets it from `session_id=`. + +- Pass `session_id=` on EVERY `run`, `arun`, `print_response`, `aprint_response`, `continue_run`, `acontinue_run`, and on `team.run/arun` and `workflow.run/arun`. Use the id the app already stores for the chat/thread; pass `user_id=` too if available. +- Without `session_id=`, Agno mints a `uuid4()` on the first run and stores it on the `Agent`/`Team` instance; every later run of that instance reuses it. A module-level shared agent then merges all users into one session. Fix it; don't rely on the default. +- Do not mint a new id per request (every turn becomes its own session). +- Team members inherit the team's session id automatically. Do not pass a different id to members. +- Do not add `using_session()` or a custom span processor for the session; Agno's run span already carries it. Maple ignores `gen_ai.conversation.id` for Agno spans (it reads `session.id`). +- If the Agent has a `db`, `session_id` also selects the stored history: keep them the same id. + +## Step 4: Content + +- Content capture is ON by default in OpenInference. With Step 2c, prompts/replies land in `gen_ai.input.messages` / `gen_ai.output.messages` on span attributes. Nothing else to enable. +- If the user asks for no content / PII-sensitive: set before start `OPENINFERENCE_HIDE_INPUT_MESSAGES=true`, `OPENINFERENCE_HIDE_OUTPUT_MESSAGES=true`, `OPENINFERENCE_HIDE_INPUTS=true`, `OPENINFERENCE_HIDE_OUTPUTS=true`. Tell them the transcript will be empty but tokens/tools/failures remain, and that tool ARGUMENTS (`tool.parameters`, `gen_ai.tool.call.arguments`) are NOT masked by any of these variables; dropping them needs an OTel Collector `redaction`/`transform` processor. +- Tool results are recorded as `str(result)` (the same string Agno sends to the model). A tool returning a `dict`/`list` produces a Python repr (`{'city': 'Berlin'}`), which is not JSON, so Maple shows plain text. Recommend returning `json.dumps(...)` from structured tools; tell the user rather than silently changing tool return types. + +## Step 5: Tools, errors, sub-agents + +- Tool spans are named after the function and carry `gen_ai.tool.name`, arguments and result. A tool that RAISES gets status ERROR with the exception message; Maple counts it as failed. A tool that RETURNS an error string counts as success. If the user's tools swallow errors into strings, point this out; don't change behavior unasked. +- Give every `Agent` and `Team` a `name=` (distinct per member). Unnamed ones produce `Agent.run`/`Team.run` spans with no `gen_ai.agent.name`, and Maple can't draw a lane for them. +- Teams: one trace per team run. Member runs (`.arun`) are children of the team run span, siblings of the leader's `delegate_task_to_member` tool spans (not nested inside them). Maple draws a lane per named member. Nothing to add. +- Team context leak (instrumentor 1.0.10): after `team.run()`/`team.arun()` returns, the team span stays attached to the thread/task, so any LATER agent run in the same thread/task nests into the finished team trace. Per-request tasks/copied contexts (FastAPI, Starlette) are unaffected. In scripts, loops, workers, CLIs and tests, isolate each team run: + + ```py + result = await asyncio.create_task(team.arun(prompt, session_id=conversation_id)) # async + result = contextvars.copy_context().run(team.run, prompt, session_id=conversation_id) # sync + ``` + + Plain `Agent` runs (sync, async, streamed) do not leak. +- Human-in-the-loop (`@tool(requires_confirmation=True)` → `requirement.confirm()/reject()` → `agent.continue_run(run_id=..., requirements=..., session_id=...)`, agent needs a `db`): + + ```py + response = agent.run(message, session_id=conversation_id) + if response.is_paused: + for requirement in response.active_requirements: + requirement.confirm() # or requirement.reject(note="...") + response = agent.continue_run( + run_id=response.run_id, requirements=response.requirements, session_id=conversation_id + ) + ``` + + The resumed run is its own trace with a `.continue_run` span carrying `session.id`, so it joins the session as a second turn. Needs instrumentor `>=1.0.8` and `session_id=` on `continue_run`. + +## Step 6: Flush + +- Long-running server (FastAPI, AgentOS): nothing to add; `TracerProvider` flushes at normal exit. +- Script / CLI / worker: wrap the entry point: + + ```py + from tracing import provider + try: + main() + finally: + provider.force_flush() + provider.shutdown() + ``` + +- Took the `setup_tracing` path in Step 2c (no `provider` of your own): flush with `trace.get_tracer_provider().force_flush()`. +- Serverless handler (Lambda, Cloud Run jobs, etc.): `provider.force_flush()` before returning from EVERY invocation; never `shutdown()`. +- Notebook: `provider.force_flush()` after the cell that runs the agent. + +## Step 7: Verify + +Run one real conversation: 2-3 turns with the same `session_id` including one tool call, plus a second conversation with a different id. Flush. Wait ~1 minute. In Maple **Agent Sessions**, filtered by the service name (or via the Maple MCP `list_agent_sessions` + `get_agent_session`), check: + +- [ ] Exactly one session per conversation id; none named `trace:` (that means a run span had no `session.id`). +- [ ] The two conversations are two different sessions (not merged by a sticky auto id). +- [ ] Framework shows **Agno** (not Unidentified). +- [ ] One turn per `run()`/`arun()`; turn labels are the user messages. +- [ ] Transcript shows system prompt, user and assistant messages, tool calls. Empty transcript with non-zero tokens = `enable_genai_semconv` not active (check Step 2c, check `instrument()` isn't called elsewhere first). +- [ ] Model calls (`.invoke|ainvoke|invoke_stream|ainvoke_stream`) show the model id and non-zero input/output tokens, including streamed turns. +- [ ] Each tool call appears once with its real name; a raised tool error is marked failed and nothing else is. Structured results render as JSON (not a Python repr). +- [ ] Teams: one lane per named member; the team run and all member spans are in one trace, same session; a run started after a team run is NOT inside the team's trace (else see Step 5 context leak). +- [ ] Cost shown if the provider returns it (OpenRouter does); otherwise "unpriced" is expected. +- [ ] No span attribute contains the model provider API key or `Bearer`. + +Raw span check (optional, e.g. with a console exporter in a scratch run): run spans `.run` have `session.id` + `gen_ai.operation.name=invoke_agent` + `gen_ai.agent.name`; model spans have `gen_ai.operation.name=chat`, `gen_ai.input.messages`, `gen_ai.usage.input_tokens`; tool spans have `gen_ai.operation.name=execute_tool`. + +## Do not + +- Do not rely on `setup_tracing()` or `AgentOS(tracing=True)` to reach Maple; they only write to the database. +- Do not omit `TraceConfig(enable_genai_semconv=True)` / `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true`. +- Do not call `AgnoInstrumentor().instrument()` twice or create a second `TracerProvider`. +- Do not stack another LLM instrumentor (OpenAI/LiteLLM OpenInference, OpenLIT, `auto_instrument=True`) on the same calls. +- Do not run agents without `session_id=` in a server; do not generate a fresh id per request. +- Do not put a different `session_id` on team members or resumed runs than on the conversation. +- Do not use `gen_ai.conversation.id` or `maple_ai.session.id` workarounds; `session.id` from Agno is what Maple reads (stamping `maple_ai.session.id` on Agno spans re-vendors them and loses decoding). +- Do not use a console/stdout exporter in production, and do not use `SimpleSpanProcessor` in servers (it exports synchronously on the request path). +- Do not skip the flush in scripts, notebooks, CLIs and serverless. +- Do not run an agent after a team run in the same thread/task without isolating the team run (Step 5). +- Do not expect `gen_ai.response.id`, `gen_ai.tool.call.id` on tool spans, TTFT or reasoning-token attributes from this instrumentor; they are not emitted and nothing needs fixing. +- Do not set `OTEL_SDK_DISABLED=true` (it turns off all tracing). `AGNO_TELEMETRY=false` only stops Agno's product analytics and is safe. diff --git a/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md b/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md new file mode 100644 index 0000000000..a7082b96d2 --- /dev/null +++ b/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md @@ -0,0 +1,221 @@ +--- +name: maple-agent-tracing-claude-agent-sdk +description: "Trace Claude Agent SDK agents (TypeScript and Python) and Claude Code CLI sessions with Maple: configure Claude Code's built-in OpenTelemetry so each conversation becomes one Maple Agent Session with prompts, model calls, tool calls and tokens. Triggers on 'trace my claude agent sdk agent', 'add Maple to claude agent sdk', 'agent sessions for claude code', 'OpenTelemetry for claude agent sdk', 'send my claude code sessions to Maple'." +--- + +# Maple agent tracing: Claude Agent SDK and Claude Code + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/claude-agent-sdk + +## Goal + +One conversation = one Maple Agent Session, one turn per user message, with the user prompts, every model call (model, tokens, TTFT), every tool call (name, args, result, failures). + +How it works: the Agent SDK emits nothing itself. `query()` spawns the Claude Code CLI, which has OpenTelemetry built in and exports spans `claude_code.interaction` (turn), `claude_code.llm_request` (model call), `claude_code.tool` (tool call), with phase children `claude_code.tool.blocked_on_user` / `claude_code.tool.execution`. All configuration is environment variables for that child process. Maple detects the `com.anthropic.claude_code` scope, groups by the `session.id` span attribute, and restates the spans as `gen_ai.*` at ingest. No instrumentation package, no TracerProvider. + +Known gaps (tell the user, don't try to fix): assistant reply text and cost are only on OTLP log events, which Maple's session views don't read, so transcripts have no assistant text and sessions show "unpriced"; no `gen_ai.agent.name`, so sub-agents get no separate lanes; tool arguments shown only for Bash (command) and Read/Edit/Write (file path). + +## Step 0: Detect + +- Which surface: + - TypeScript: `@anthropic-ai/claude-agent-sdk` in `package.json`. Need >= 0.3.283 (bundles Claude Code 2.1.283). Upgrade if older. + - Python: `claude-agent-sdk` in `pyproject.toml` / `requirements*.txt`. Need >= 0.2.160 (bundles 2.1.283). Upgrade if older. + - The user wants their own `claude` CLI / IDE / desktop sessions in Maple: go to Step 2c. Check `claude --version` >= 2.1.283. + - Check for `pathToClaudeCodeExecutable` (TS) / `cli_path` (Py): a custom CLI binary must also be >= 2.1.283. +- Existing OpenTelemetry in the app: keep it. It can't carry the CLI's spans (the CLI exports on its own), but if the app has an active span when `query()` runs, both SDKs pass it as `TRACEPARENT` and the turn nests under it. Do not add a second SDK/exporter for the agent. +- Remove any hook-based instrumentor for the Agent SDK (OpenInference `openinference-instrumentation-claude-agent-sdk`, Langfuse/LangSmith/Opik wrappers) if the user agrees: it duplicates every model call and its spans don't group by session in Maple. +- Find where env is already set for the CLI: `options.env` / `ClaudeAgentOptions(env=...)`, Dockerfile, deploy manifests. +- Settings files beat `options.env`: when `settingSources` / `setting_sources` is omitted (all sources) or includes `user`/`project`, an `env` block in `~/.claude/settings.json` or the repo's `.claude/settings.json` overrides the same keys passed in `options.env` (verified with `OTEL_SERVICE_NAME`). If those files set `OTEL_*` / `CLAUDE_CODE_*` keys, tell the user; for server apps that don't need file settings, suggest `settingSources: []` (Py `setting_sources=[]`). +- Find the conversation boundary: how the app calls `query()` per user message, and whether it stores a session id (`resume`, `sessionId`, `session_id`, `continue`, `ClaudeSDKClient`). + +## Step 1: Key and region + +- US endpoint `https://ingest.maple.dev`, EU endpoint `https://ingest.eu.maple.dev`. Header `Authorization=Bearer `. +- Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Private `maple_sk_` keys never go in browser code. Ingest keys are write-only. +- Follow the repo's existing secret/env convention (e.g. `MAPLE_INGEST_KEY` in `.env`). If there is none, inlining the ingest key is acceptable. + +## Step 2a: TypeScript SDK + +`npm install @anthropic-ai/claude-agent-sdk@latest zod` (zod ^4 is a peer). + +`options.env` REPLACES the child environment. Always spread `process.env`, and drop inherited `TRACEPARENT`/`TRACESTATE`. Create `maple-env.ts` (adapt service name, environment, key source): + +```ts +const inherited: Record = { ...process.env } +delete inherited.TRACEPARENT +delete inherited.TRACESTATE + +export const mapleEnv: Record = { + ...inherited, + CLAUDE_CODE_ENABLE_TELEMETRY: "1", + CLAUDE_CODE_ENHANCED_TELEMETRY_BETA: "1", + OTEL_TRACES_EXPORTER: "otlp", + OTEL_LOGS_EXPORTER: "otlp", + OTEL_METRICS_EXPORTER: "otlp", + OTEL_EXPORTER_OTLP_PROTOCOL: "http/protobuf", + OTEL_EXPORTER_OTLP_ENDPOINT: "https://ingest.maple.dev", + OTEL_EXPORTER_OTLP_HEADERS: `Authorization=Bearer ${process.env.MAPLE_INGEST_KEY}`, + OTEL_SERVICE_NAME: "support-agent", + OTEL_RESOURCE_ATTRIBUTES: "deployment.environment.name=production", + OTEL_TRACES_EXPORT_INTERVAL: "1000", + OTEL_LOGS_EXPORT_INTERVAL: "1000", + OTEL_LOG_USER_PROMPTS: "1", + OTEL_LOG_TOOL_DETAILS: "1", + OTEL_LOG_TOOL_CONTENT: "1", +} +``` + +Pass `env: mapleEnv` on EVERY `query()` / `startup()` / session call in the codebase. If the call already sets `env`, merge its keys into `mapleEnv` rather than dropping them. + +## Step 2b: Python SDK + +`pip install -U claude-agent-sdk` (or the repo's tool: `uv add`, `poetry add`). Python >= 3.10. + +`ClaudeAgentOptions.env` MERGES over the inherited env, so pass only telemetry vars; remove inherited trace context from `os.environ`. Create `maple_env.py`: + +```py +import os + +os.environ.pop("TRACEPARENT", None) +os.environ.pop("TRACESTATE", None) + +MAPLE_ENV = { + "CLAUDE_CODE_ENABLE_TELEMETRY": "1", + "CLAUDE_CODE_ENHANCED_TELEMETRY_BETA": "1", + "OTEL_TRACES_EXPORTER": "otlp", + "OTEL_LOGS_EXPORTER": "otlp", + "OTEL_METRICS_EXPORTER": "otlp", + "OTEL_EXPORTER_OTLP_PROTOCOL": "http/protobuf", + "OTEL_EXPORTER_OTLP_ENDPOINT": "https://ingest.maple.dev", + "OTEL_EXPORTER_OTLP_HEADERS": f"Authorization=Bearer {os.environ['MAPLE_INGEST_KEY']}", + "OTEL_SERVICE_NAME": "support-agent", + "OTEL_RESOURCE_ATTRIBUTES": "deployment.environment.name=production", + "OTEL_TRACES_EXPORT_INTERVAL": "1000", + "OTEL_LOGS_EXPORT_INTERVAL": "1000", + "OTEL_LOG_USER_PROMPTS": "1", + "OTEL_LOG_TOOL_DETAILS": "1", + "OTEL_LOG_TOOL_CONTENT": "1", +} +``` + +Import `maple_env` before the first `query()` / `ClaudeSDKClient` and pass `env=MAPLE_ENV` (merge with any existing `env` dict) on every `ClaudeAgentOptions`. + +Alternative for both SDKs: set the same variables in the deployment environment (Dockerfile, k8s manifest) and omit `env`. Then make sure no `TRACEPARENT` is set there. + +## Step 2c: Claude Code CLI (the user's own sessions) + +Merge into `~/.claude/settings.json` (user settings; create if missing, keep existing keys): + +```json +{ + "env": { + "CLAUDE_CODE_ENABLE_TELEMETRY": "1", + "CLAUDE_CODE_ENHANCED_TELEMETRY_BETA": "1", + "OTEL_TRACES_EXPORTER": "otlp", + "OTEL_LOGS_EXPORTER": "otlp", + "OTEL_METRICS_EXPORTER": "otlp", + "OTEL_EXPORTER_OTLP_PROTOCOL": "http/protobuf", + "OTEL_EXPORTER_OTLP_ENDPOINT": "https://ingest.maple.dev", + "OTEL_EXPORTER_OTLP_HEADERS": "Authorization=Bearer YOUR_INGEST_KEY", + "OTEL_LOG_USER_PROMPTS": "1", + "OTEL_LOG_TOOL_DETAILS": "1", + "OTEL_LOG_TOOL_CONTENT": "1" + } +} +``` + +- NOT the repo's `.claude/settings.json` / `.claude/settings.local.json`: Claude Code >= 2.1.282 ignores the vars there that turn export on, set the endpoint, or capture content (`CLAUDE_CODE_ENABLE_TELEMETRY`, `OTEL_LOG_*`, ...). +- Existing `OTEL_*` keys in user settings, or managed settings / `~/.claude/remote-settings.json` setting endpoint or headers: stop and ask the user; managed values override the user's, and replacing them would redirect their company's telemetry. +- Tell the user to start a new `claude` session. The current session doesn't reload it. +- Service name is `claude-code` (terminal) or `claude-code-desktop` (desktop Code tab). Session grouping needs no setup: one `claude` session = one Maple session; `/clear` and `--fork-session` start new ones. + +## Step 3: One session per conversation (SDK only) + +Maple's session key for Claude Code is the `session.id` span attribute = the Claude session id. Every `query()` without `resume` starts a NEW session, so a chat backend calling `query()` once per message gets one Maple session per message unless it resumes. + +- Store a UUID per conversation. First turn: `sessionId: uuid` (TS) / `session_id=uuid` (Py). Later turns: `resume: uuid` / `resume=uuid`. `sessionId`/`session_id` must be a valid UUID and can't be combined with `resume`. +- If the app already captures `session_id` from the init/result message and passes `resume`, keep it; that is correct. +- `ClaudeSDKClient` (Py) or a TS `query()` with an async-iterable prompt keeps one session for all its turns: nothing to do. +- Never set `forkSession` / `fork_session` on normal turns (new session id). Never set `OTEL_METRICS_INCLUDE_SESSION_ID=false` (removes `session.id` from spans). +- `resume` needs the transcript under `~/.claude/projects/` on the same host. Multi-host / serverless: tell the user about the `sessionStore` option; don't implement it unasked. +- Do not add `gen_ai.conversation.id` or `maple_ai.session.id`: you can't attribute the CLI's spans, and Maple reads `session.id` for this vendor. + +TS pattern: + +```ts +const firstTurn = !conversation.claudeSessionId +const sessionId = conversation.claudeSessionId ?? randomUUID() +conversation.claudeSessionId = sessionId // persist with the conversation +for await (const message of query({ + prompt: text, + options: { env: mapleEnv, ...(firstTurn ? { sessionId } : { resume: sessionId }) }, +})) { + if (message.type === "result") return message.subtype === "success" ? message.result : undefined +} +``` + +Python pattern: + +```py +first_turn = "claude_session_id" not in conversation +session_id = conversation.setdefault("claude_session_id", str(uuid.uuid4())) +session = {"session_id": session_id} if first_turn else {"resume": session_id} +async for message in query(prompt=text, options=ClaudeAgentOptions(env=MAPLE_ENV, **session)): + ... +``` + +## Step 4: Content + +- `OTEL_LOG_USER_PROMPTS=1`: prompt on `claude_code.interaction` → turn titles + user messages. Without it: ``, untitled turns. +- `OTEL_LOG_TOOL_DETAILS=1`: Bash `full_command`, Read/Edit/Write `file_path` → tool arguments; full error message on failed tools; `subagent_type`. +- `OTEL_LOG_TOOL_CONTENT=1`: `tool.output` span event → tool result (Read, Bash, Edit/Write with DETAILS; MCP/SDK tools, WebFetch, WebSearch on >= 2.1.283). +- Ask the user before enabling content in production if the repo shows compliance constraints (PII handling, HIPAA, etc.); content flags send file contents and command output. Offer to leave them off; the session still works (untitled turns, no args/results). +- Do NOT enable `ENABLE_BETA_TRACING_DETAILED` / `BETA_TRACING_ENDPOINT`: they redirect logs+traces and Maple doesn't read what they add. +- `OTEL_LOG_ASSISTANT_RESPONSES` only affects the `assistant_response` log event (not shown in session views). It defaults to `OTEL_LOG_USER_PROMPTS`; set `0` if prompts are approved and replies are not. + +## Step 5: Tools, errors, sub-agents + +- Nothing to add. Each `claude_code.tool` span is one tool call named by `tool_name` (`Bash`, `Read`, `mcp____`, `Agent`). +- Tool failures: SDK tool handlers must fail by throwing or returning `isError: true` (Py `"is_error": True`), not by returning an ordinary success result that says "error". Maple marks the call failed from `claude_code.tool.execution` `success=false` (`error.type` = `error_class`, result = the error). +- Sub-agents (`agents` option, `.claude/agents/`) run through the `Agent` tool; their spans nest under it in one trace. Maple shows no per-sub-agent lane (no `gen_ai.agent.name`). Don't try to fix it. +- Background sub-agents: Claude Code 2.1.283 may run `Agent` calls in the background even when the model doesn't ask for it (seen with SDK `agents` in string-prompt `query()`). Then the turn ends at once, every finished sub-agent starts a new turn whose prompt is a `` block, one `query()` yields several `result` messages, and SDK MCP tool calls inside those sub-agents can fail with "The tool call was interrupted before a result was received" (Maple counts them as failed tools). If the app uses `agents` and expects one answer per `query()`, add `CLAUDE_CODE_DISABLE_BACKGROUND_TASKS: "1"` to its env (verified: one turn, one trace, parallel sub-agents still overlap). Otherwise consume the iterator to its end, not to the first `result`. + +## Step 6: Flush + +- Keep `OTEL_TRACES_EXPORT_INTERVAL=1000` and `OTEL_LOGS_EXPORT_INTERVAL=1000`. +- Consume every `query()` loop to its `result` message (returning from the loop at `result`, as in Step 3, is fine: verified no span loss). Don't `break`, `close()`, or abort a query before that; it kills the CLI before its final export. +- Scripts / CLIs / one-shot jobs: after the last `query()` completes, wait ~5 s before the process exits (`await new Promise((r) => setTimeout(r, 5000))` / `await asyncio.sleep(5)`). +- Serverless: finish the loop before returning the response. + +## Step 7: Verify + +Run one conversation: two turns (the second resuming the first), one of which calls a tool. Use `MAPLE_TEST` only if no real key; with the sentinel nothing is stored, so ask the user to check in Maple once they have a key. To see export errors: add `CLAUDE_CODE_OTEL_DIAG_STDERR: "1"` to the env and a `stderr` callback (TS `options.stderr`, Py `ClaudeAgentOptions(stderr=...)`); no `[3P telemetry]` errors should appear. For the CLI: `claude --debug-file /tmp/claude.log`, then grep `3P telemetry`. + +Then in Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`) check: + +- Exactly one session for the conversation (not one per message), framework "Claude Agent SDK", service = `OTEL_SERVICE_NAME`. +- A second conversation run in the same process gets a different session. +- One turn per user message, titled with the prompt. +- LLM calls > 0 with a model name and input/output/cache tokens; TTFT present. +- Tool calls listed by real name (`mcp____`, `Bash`, ...), args for Bash/file tools, results when `OTEL_LOG_TOOL_CONTENT=1`. +- A tool that threw is marked failed; successful tools are not. +- Sub-agent model/tool calls appear under the `Agent` tool call in the same trace, and no turn is titled `` (if one is, see background sub-agents in Step 5). +- Expected and not bugs: cost "unpriced"; no assistant text; no sub-agent lanes. + +If spans exist but the turn nests under an unrelated trace, an inherited `TRACEPARENT` survived: fix the env stripping. + +## Do not + +- Do not omit `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`: zero spans without it (metrics/logs still flow, which hides the problem). +- Do not leave `OTEL_EXPORTER_OTLP_PROTOCOL` unset or use `grpc`: Claude Code has no default; Maple ingest is OTLP/HTTP. +- Do not set any exporter to `console` in SDK apps: stdout is the SDK message channel. +- Do not pass a TS `env` without `...process.env`. +- Do not pass an inherited `TRACEPARENT`/`TRACESTATE` to the CLI (Claude Code's Bash tool and CI set them). +- Do not call `query()` per message without `resume`. +- Do not put telemetry vars in a repo's `.claude/settings.json`. +- Do not assume `options.env` wins: `env` in loaded settings files overrides it (Step 0). +- Do not add a second instrumentation (OpenInference/Langfuse/LangSmith hooks) on top of the CLI's spans. +- Do not use `OTEL_METRICS_INCLUDE_SESSION_ID=false`, `forkSession` on normal turns, or detailed beta tracing. +- Do not promise cost or assistant replies in Maple's session views; they are in Logs (`claude_code.api_request` `cost_usd`, `claude_code.assistant_response`) and the `claude_code.cost.usage` metric. +- Do not print or commit `maple_sk_` keys or model API keys. diff --git a/skills/maple-agent-tracing-crewai/SKILL.md b/skills/maple-agent-tracing-crewai/SKILL.md new file mode 100644 index 0000000000..d5f2c4ab2d --- /dev/null +++ b/skills/maple-agent-tracing-crewai/SKILL.md @@ -0,0 +1,215 @@ +--- +name: maple-agent-tracing-crewai +description: "Trace CrewAI crews and flows with Maple: OpenInference CrewAI instrumentor plus the model-SDK instrumentor, GenAI dual-write, one Maple Agent Session per conversation with transcript, model and tool calls, tokens, failed tools and one lane per agent. Triggers on 'trace my crewai agent', 'add Maple to crewai', 'agent sessions for crewai', 'OpenTelemetry for crewai'." +--- + +# Maple agent tracing: CrewAI + +Goal: every conversation the app runs through CrewAI shows up in Maple **Agent Sessions** as ONE session, with the transcript, each model call (model, tokens), each tool call (name, result, failure) and one lane per agent role. Reasoning and background for every step: https://maple.dev/docs/agent-tracing/crewai + +CrewAI exports nothing to your backend. Its built-in telemetry is anonymous analytics to crewai.com on a private provider; its OTel export is AMP-only. All spans come from OpenInference, and the defaults are wrong for Maple in four ways this skill fixes: the CrewAI instrumentor records no model calls (a second, SDK-level instrumentor is required), OpenInference-only attributes (Maple's session page reads `gen_ai.*` for CrewAI's spans), no session id, and no agent-name attribute. + +## Step 0: Detect versions and existing OpenTelemetry + +- Read `pyproject.toml` / `requirements*.txt` / `uv.lock` / `poetry.lock`. Need `crewai>=1.15`, Python 3.10-3.13. If older, upgrade CrewAI first (1.x changed provider routing). +- Find every `LLM(model=...)` / `Agent(llm=...)` string and map it to the SDK CrewAI calls. Install one instrumentor per SDK actually used: + +| Model string | SDK | Instrumentor package / class | +| --- | --- | --- | +| `openai/…`, `openrouter/…`, `deepseek/…`, `ollama/…`, `hosted_vllm/…`, `cerebras/…`, `dashscope/…`, `custom_openai=True`, or any bare name not matched below | `openai` | `openinference-instrumentation-openai` / `OpenAIInstrumentor` | +| `anthropic/…`, `claude/…`, bare `claude-…` | `anthropic` | `openinference-instrumentation-anthropic` / `AnthropicInstrumentor` | +| `gemini/…`, `google/…`, bare `gemini-…` | `google-genai` | `openinference-instrumentation-google-genai` / `GoogleGenAIInstrumentor` | +| `bedrock/…`, `aws/…`, bare `anthropic.claude-…` | `boto3` | `openinference-instrumentation-bedrock` / `BedrockInstrumentor` | +| any other prefix (LiteLLM fallback, needs `crewai[litellm]`) | `litellm` | `openinference-instrumentation-litellm` / `LiteLLMInstrumentor` | + + An agent with no `llm=` uses env `MODEL` / `MODEL_NAME` / `OPENAI_MODEL_NAME`, else `gpt-4.1-mini` (the `openai` row); resolve that string with the same table. `azure/…` uses `azure-ai-inference`, which has no OpenInference instrumentor: tell the user model calls won't be recorded. +- Find every `kickoff` call site (`crew.kickoff`, `kickoff_async`, `akickoff`, `flow.kickoff`, `flow.handle_turn`, `flow.resume`, `Agent.kickoff`) and how conversations are identified (chat id, thread id, session row, flow `state.id`). +- Search for an existing `TracerProvider`, `trace.set_tracer_provider`, `opentelemetry-instrument`, `logfire.configure`, `phoenix.otel.register`, `langfuse`, `sentry_sdk.init`, `CrewAIInstrumentor`, `litellm.callbacks = ["otel"]`. If a provider exists, REUSE it: add Maple's exporter and the processor below to it and pass it to `instrument()`. Never create a second provider. If an instrumentor's `instrument()` already runs, change that call instead of adding another. +- Search for `OTEL_SDK_DISABLED`. If set to true, remove it (it kills the whole SDK) and replace with `CREWAI_DISABLE_TELEMETRY=true`. + +## Step 1: Ingest key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. Protocol: http/protobuf. +- Key given in the prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from **Settings → Ingestion**. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's existing secret/env convention (`.env`, settings module, secret manager) if it has one. Otherwise inlining the key is acceptable: ingest keys are write-only. + +## Step 2: Install and initialize + +```bash +pip install "crewai>=1.15" "openinference-instrumentation-crewai>=1.1.18" \ + "openinference-instrumentation-openai>=0.1.61" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Use the repo's package manager (`uv add`, `poetry add`...). Swap/add the model-SDK instrumentor per the Step 0 table. + +Environment (in the repo's env mechanism): + +```bash +OTEL_SERVICE_NAME= +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name= +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer " +CREWAI_DISABLE_TELEMETRY=true +CREWAI_TRACING_ENABLED=false +``` + +- Base URL only. `OTLPSpanExporter()` with no args appends `/v1/traces`. If you pass `endpoint=` in code, it must end in `/v1/traces`. +- `CREWAI_DISABLE_TELEMETRY=true` stops the analytics export to telemetry.crewai.com. `CREWAI_TRACING_ENABLED=false` stops the AMP uploader and its first-run prompt that waits on stdin at exit. Both must be in the environment before `crewai` is imported (env file loaded by the process, or `os.environ.setdefault(...)` at the top of `tracing.py` if the repo has no env mechanism). + +Create `tracing.py` (adapt the module path to the repo layout): + +```py +# tracing.py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.crewai import CrewAIInstrumentor +from openinference.instrumentation.openai import OpenAIInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class CrewAIAgentNames(SpanProcessor): + """Copies each CrewAI agent's role to gen_ai.agent.name, which Maple uses for agent lanes.""" + + def on_start(self, span, parent_context=None): + # The instrumentor records the role (graph.node.id) just after the agent span starts, + # so name the agent span when its first child starts, while it's still open. + parent = trace.get_current_span(parent_context) + attrs = getattr(parent, "attributes", None) or {} + role = attrs.get("graph.node.id") + if role and "gen_ai.agent.name" not in attrs and parent.is_recording(): + parent.set_attribute("gen_ai.agent.name", role) + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(CrewAIAgentNames()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +config = TraceConfig(enable_genai_semconv=True) +CrewAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) +OpenAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) +``` + +- `import tracing` at the top of every entry point (web app module, worker, CLI main, `main.py` of a `crewai create` project) so `instrument()` runs before the first kickoff. +- Other SDKs: add `AnthropicInstrumentor().instrument(...)` etc. with the same `tracer_provider`, `config` and `skip_dep_check=True`. Only for SDKs the app uses (Step 0). +- Existing provider: skip `TracerProvider()`/`set_tracer_provider`, add `CrewAIAgentNames()` and the exporter to the existing provider, pass it as `tracer_provider=`. +- `enable_genai_semconv=True` on EVERY instrumentor is required: without it CrewAI's agent/tool spans have no operation or tool details on Maple's session page and model calls have no `{role, parts}` transcript. The env var `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` is equivalent only if set before `instrument()`; prefer the code form. +- Keep `skip_dep_check=True`: a failed version check makes `instrument()` skip itself with only an error log. +- Keep `CrewAIAgentNames` exactly, and add it BEFORE the `BatchSpanProcessor`. + +## Step 3: One session per conversation + +CrewAI has no conversation id; each `kickoff()` is its own trace. `crew_id` (new per Crew object) and `crew_key` (same for every user of a crew) are NOT conversation ids. Maple reads `session.id` for CrewAI, and the instrumentors set it only inside `using_session`: + +```py +from crewai import LLM, Agent, Crew, Task +from openinference.instrumentation import using_session + +llm = LLM(model="openai/gpt-4o-mini", temperature=0) + + +def build_crew(text: str, history: str, stream: bool = False) -> Crew: + assistant = Agent( + role="assistant", + goal="Answer the user's questions", + backstory="You are a concise, helpful assistant.", + llm=llm, + tools=[get_weather, calculate], + ) + task = Task( + description=f"{text}\n\nConversation so far:\n{history}", + expected_output="A short, direct reply to the user.", + agent=assistant, + name="reply", + ) + return Crew(name="support", agents=[assistant], tasks=[task], stream=stream) + + +def handle_message(conversation_id: str, text: str, history: str) -> str: + with using_session(conversation_id): + return build_crew(text, history).kickoff().raw +``` + +- Wrap EVERY kickoff call site in `with using_session():`. Use the app's stored conversation/chat/thread id. Never a fresh UUID per request, never a constant. +- Conversational flows: `with using_session(sid): flow.handle_turn(text, session_id=sid)`. Use the same id for both. +- Batch/one-shot crews (no chat): one id per run/job is correct; still wrap the kickoff so the run is a named session instead of `trace:`. +- Give every `Crew` a `name=` (otherwise the root span is `Crew_.kickoff`) and every `Task` a `name=`. Put the user's message FIRST in the task description: Maple labels turns with the first line of `Current Task: …`. +- Do not change the app's own history handling beyond that; CrewAI has no chat memory, so the app already passes history somehow. +- `akickoff()` is NOT instrumented (no crew/agent spans, each model/tool call becomes its own trace). Replace `await crew.akickoff(...)` with `await crew.kickoff_async(...)` (same result, runs instrumented `kickoff` in a thread). Same for `Agent.akickoff` -> `Agent.kickoff` in a thread. Tell the user why. +- `Crew(stream=True)` runs the crew twice (a stub `kickoff` span + the real one in a second trace). Wrap each streamed turn: + +```py +from opentelemetry import trace + +tracer = trace.get_tracer("chat") + + +def stream_message(conversation_id: str, text: str, history: str, send) -> None: + with using_session(conversation_id), tracer.start_as_current_span( + "invoke_agent support", + attributes={ + "gen_ai.operation.name": "invoke_agent", + "gen_ai.conversation.id": conversation_id, + }, + ): + for chunk in build_crew(text, history, stream=True).kickoff(): + send(chunk.content) +``` + + Iterate INSIDE the `with`. `LLM(stream=True)` alone (no `Crew(stream=True)`) needs no wrapper. +- `flow.resume(...)` after `@human_feedback` is not instrumented: wrap it the same way (`using_session` with the same id + the wrapper span). +- Your own spans (plain OTel tracer) don't get `session.id` automatically; give them `attributes=dict(get_attributes_from_context())` (from `openinference.instrumentation`) if you add any beyond the wrapper above. + +## Step 4: Content + +- On by default: model spans carry the messages CrewAI sent (system = role/goal/backstory, user = `Current Task: …` + context) and the reply; agent spans the task and output; tool spans arguments and results. Leave it on unless the user or repo says prompts are sensitive. +- To turn off: `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` passed to every instrumentor (or `OPENINFERENCE_HIDE_INPUTS=true` / `OPENINFERENCE_HIDE_OUTPUTS=true`). Narrower: `hide_input_text`, `hide_output_text`. +- Agent roles, task names and tool names are span names and always recorded. Don't put PII in them. +- Do NOT set `share_crew=True` to get content; it only adds data to CrewAI's analytics. + +## Step 5: Tools, errors, agents + +- Tool exceptions are marked failed automatically (`.run` span status ERROR with the message). Do not catch exceptions inside tools to return an error string: the span stays OK and Maple won't count the failure. +- Tool spans need native function calling (all providers in the Step 0 table). Custom `BaseLLM` subclasses or LiteLLM models without function calling use the ReAct text path, which the instrumentor doesn't patch: no tool spans. Tell the user. +- Every agent needs a distinct `role`; `CrewAIAgentNames` turns roles into lanes. +- `async_execution=True` tasks keep context (siblings under the crew span). Nothing to do. +- `Process.hierarchical`: delegated coworker work (`Delegate work to coworker` / `Ask question to coworker` tools) runs through un-instrumented `Agent.execute_task`: its model calls sit inside the tool span, no lane. Known; not fixable here. +- Known, not fixable here: `gen_ai.tool.call.arguments` holds the tool's JSON schema (OpenInference copies `tool.parameters`); the real arguments are in `input.value`. No `gen_ai.tool.call.id` on tool spans. + +## Step 6: Flush + +- `TracerProvider` flushes on normal interpreter exit (atexit). That covers servers and CLIs that exit normally, including `crewai run`. +- Serverless handlers (Lambda, Cloud Run jobs, Modal functions...): call `provider.force_flush()` in a `finally` before returning. +- Scripts that may be killed or call `os._exit`, and notebooks: `provider.force_flush()` after each run; `provider.shutdown()` at the very end. + +## Step 7: Verify + +Run one real conversation (2-3 messages, same conversation id, at least one tool call), and one message in a second conversation. If the user gave no key (`MAPLE_TEST`), you can't see results in Maple; say so and list what they should check. Otherwise check in Maple **Agent Sessions** (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), filtered to the service name: + +- Exactly one session per conversation id (two here), not one per message and no `trace:` sessions. +- One turn per kickoff; each turn's root span is `.kickoff` (or `.kickoff`, or your `invoke_agent` wrapper when streaming). No empty extra turns. +- Transcript is non-empty (system message from role/goal/backstory, `Current Task: …`, replies). +- Model calls (`ChatCompletion` for the OpenAI instrumentor) have a model and non-zero input/output tokens, including streamed calls. Each call appears once. +- Tool calls `.run` with results; a tool that raised is counted as failed and nothing else is. +- One lane per agent role (`.._execute_core` spans carry `gen_ai.agent.name`). +- Cost: unpriced unless models go through LiteLLM (which records `llm.cost.total`). Expected. +- No spans with scope `crewai.telemetry`, no `coding_agent` attribute (telemetry is off). + +If sessions are split per message: `using_session` missing or id changing. No model spans/tokens: wrong or missing SDK instrumentor. Every call its own trace: `akickoff`. Empty session page details: `enable_genai_semconv` not applied to that instrumentor. Nothing arrives: exporter endpoint/header wrong, `OTEL_SDK_DISABLED=true`, or process exited without flushing. + +## Do not + +- Do not install only `openinference-instrumentation-crewai`: no model calls, no tokens, no transcript. +- Do not add the LiteLLM instrumentor for `openai/`/`openrouter/`/`anthropic/`/`gemini/` models (CrewAI 1.x calls those SDKs natively; LiteLLM records nothing). Do not stack two model-layer instrumentors or `litellm.callbacks=["otel"]` on the same calls (duplicate spans). +- Do not use `OTEL_SDK_DISABLED=true` to silence CrewAI telemetry. +- Do not use `crewai.telemetry`, `share_crew`, or `CREWAI_TRACING_ENABLED=true` as the Maple pipeline. +- Do not use `crew_id`, `crew_key`, `task_id` or a fresh UUID per request as the session id; do not set `gen_ai.conversation.id` alone expecting grouping (Maple reads `session.id` for CrewAI). +- Do not call `akickoff()` on traced crews. +- Do not pass `OTLPSpanExporter(endpoint="https://ingest.maple.dev")` without `/v1/traces`. +- Do not create a second `TracerProvider` when one exists, and do not call `instrument()` twice. +- Do not wrap tools in try/except that returns error strings; let them raise. diff --git a/skills/maple-agent-tracing-dspy/SKILL.md b/skills/maple-agent-tracing-dspy/SKILL.md new file mode 100644 index 0000000000..8d6929fab9 --- /dev/null +++ b/skills/maple-agent-tracing-dspy/SKILL.md @@ -0,0 +1,261 @@ +--- +name: maple-agent-tracing-dspy +description: "Trace DSPy programs and ReAct agents with Maple: OpenInference DSPy instrumentor with GenAI dual-write, a DSPy callback for tokens, cost, tool names and agent spans, using_session for one session per conversation, and thread context for dspy.Parallel. Triggers on 'trace my dspy agent', 'add Maple to dspy', 'agent sessions for dspy', 'OpenTelemetry for dspy'." +--- + +# Maple agent tracing: DSPy + +Goal: every conversation with the user's DSPy program shows up in Maple **Agent Sessions** as one session, with one turn per call to the program, a transcript, model calls with tokens and cost, tool calls with arguments, results and failures, and one lane per worker module. Human guide with the reasoning: https://maple.dev/docs/agent-tracing/dspy + +DSPy emits nothing by itself. Spans come from `openinference-instrumentation-dspy`. It records no tokens, no tool names, no agent spans and no session id; the steps below add all four. Do not skip any step. + +## Step 0: Detect versions and existing setup + +- Read `pyproject.toml` / `requirements*.txt` / `uv.lock`. Require `dspy>=3.4`. On 3.3 or older, tell the user this skill targets 3.4 and ask before upgrading. +- Find how LMs are built: `dspy.LM("provider/model", ...)`. Note any `engine=`, `cache=`, `callbacks=`, `disable_history`, `max_history_size`. +- Search for an existing OpenTelemetry setup: `TracerProvider(`, `set_tracer_provider`, `opentelemetry-instrument`, `logfire.configure`, `mlflow.dspy.autolog`, `phoenix.otel.register`, `LiteLLMInstrumentor`, `OpenAIInstrumentor`. + - An existing `TracerProvider`: reuse it. Add the OTLP exporter to it and pass it to `instrument()`. Never create a second provider. + - `LiteLLMInstrumentor` / `OpenAIInstrumentor` from OpenInference: remove them (they double-count model calls on the LiteLLM engine and record nothing on the native one). + - `mlflow.dspy.autolog()`: leave it if the user wants MLflow too, but it is not the Maple path. +- Find where the program is called per user message (HTTP handler, CLI loop, worker). That is where the session id goes. +- Find `dspy.Parallel`, `dspy.Evaluate`, `ThreadPoolExecutor`, `threading.Thread` around module calls. +- Find `dspy.streamify(...)` calls and where they run relative to `dspy.configure` (see Streaming in Step 3). +- Check for a `dspy.Module` subclass with an attribute named `history` (see "Do not"). + +## Step 1: Ingest key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with a key from **Settings → Ingestion**. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's existing secret/env convention (`.env`, settings module, secret manager). If there is none, inline is acceptable: ingest keys are write-only. + +## Step 2: Install and initialize + +```bash +pip install "dspy>=3.4" "openinference-instrumentation-dspy>=0.1.45" "openinference-instrumentation>=0.1.66" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ + "opentelemetry-instrumentation-threading>=0.66b0" +``` + +Use the repo's package manager (`uv add`, `poetry add`, requirements file). + +Environment (base URL, no `/v1/traces`; the exporter appends it): + +```bash +OTEL_SERVICE_NAME= +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name= +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer " +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +Create `tracing.py`: + +```py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.dspy import DSPyInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.instrumentation.threading import ThreadingInstrumentor +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +DSPyInstrumentor().instrument(tracer_provider=provider, config=TraceConfig(enable_genai_semconv=True)) +ThreadingInstrumentor().instrument() +``` + +- `enable_genai_semconv=True` is required: Maple's session page reads `gen_ai.*` for DSPy, not OpenInference `llm.*` / `input.value`. +- `ThreadingInstrumentor` is required whenever modules run in threads (`dspy.Parallel`, `Evaluate`, executors). Without it every worker is an orphan trace with no session. +- Import `tracing` as the first import of every entry point (web app, CLI, worker), before the DSPy program modules. + +Create `maple_dspy.py` exactly as below (it fills the instrumentor's gaps from DSPy's callback hooks, which run inside the instrumentor's spans): + +```py +import json + +import dspy +from dspy.utils.callback import BaseCallback +from openinference.instrumentation import TraceConfig +from opentelemetry import trace + +# Same switches as the instrumentor: OPENINFERENCE_HIDE_INPUTS / OPENINFERENCE_HIDE_OUTPUTS. +_config = TraceConfig() + + +def _message(role, values): + text = "\n".join(v for v in values if isinstance(v, str)) + return json.dumps([{"role": role, "parts": [{"type": "text", "content": text}]}]) if text else None + + +class MapleCallback(BaseCallback): + """Adds what Maple reads and the OpenInference DSPy instrumentor leaves out: + an agent span per program, tool names and arguments, tokens and cost.""" + + def __init__(self): + self._agents = set() + self._lms = {} + + def on_module_start(self, call_id, instance, inputs): + if type(instance).__module__.startswith("dspy."): + return # Predict, ChainOfThought, ReAct: building blocks, not agents + self._agents.add(call_id) + span = trace.get_current_span() + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", type(instance).__name__) + user = _message("user", [*inputs.get("args", ()), *inputs.get("kwargs", {}).values()]) + if user and not _config.hide_inputs: + span.set_attribute("gen_ai.input.messages", user) + + def on_module_end(self, call_id, outputs, exception): + if call_id not in self._agents: + return + self._agents.discard(call_id) + if isinstance(outputs, dspy.Prediction) and not _config.hide_outputs: + reply = _message("assistant", [v for k, v in outputs.items() if k != "reasoning"]) + if reply: + trace.get_current_span().set_attribute("gen_ai.output.messages", reply) + + def on_adapter_format_start(self, call_id, instance, inputs): + # The span is "ChatAdapter.__call__"; without an operation, "chat" in the name reads as a model call. + trace.get_current_span().set_attribute("gen_ai.operation.name", "invoke_workflow") + + def on_tool_start(self, call_id, instance, inputs): + span = trace.get_current_span() + span.set_attribute("gen_ai.tool.name", instance.name) + span.set_attribute("gen_ai.tool.description", instance.desc or "") + if not _config.hide_inputs: + span.set_attribute("gen_ai.tool.call.arguments", json.dumps(inputs.get("kwargs", {}), default=str)) + + def on_lm_start(self, call_id, instance, inputs): + self._lms[call_id] = instance + + def on_lm_end(self, call_id, outputs, exception): + lm = self._lms.pop(call_id, None) + # The LM's history holds the provider response. Threads share the LM, so match ours by identity. + entry = next((e for e in reversed(lm.history[-16:]) if e["outputs"] is outputs), None) if lm else None + if entry is None or getattr(entry["response"], "cache_hit", False): + return # history is off, or a cache hit that cost nothing + usage = entry["usage"] or {} + attributes = { + "gen_ai.response.id": getattr(entry["response"], "id", None), + "gen_ai.response.model": entry.get("response_model"), + "gen_ai.usage.input_tokens": usage.get("prompt_tokens"), + "gen_ai.usage.output_tokens": usage.get("completion_tokens"), + "gen_ai.usage.cache_read.input_tokens": (usage.get("prompt_tokens_details") or {}).get("cached_tokens"), + "gen_ai.usage.reasoning.output_tokens": (usage.get("completion_tokens_details") or {}).get("reasoning_tokens"), + "gen_ai.usage.cost": entry.get("cost"), + } + trace.get_current_span().set_attributes({k: v for k, v in attributes.items() if v is not None}) +``` + +Register it where the app configures DSPy, keeping existing callbacks: + +```py +import tracing # first + +import dspy +from maple_dspy import MapleCallback + +dspy.configure(lm=dspy.LM("openai/gpt-4o-mini"), callbacks=[MapleCallback()]) # append to any existing callbacks list +``` + +- Every later `dspy.configure(callbacks=...)` or `dspy.context(callbacks=...)` must include the `MapleCallback` instance too, or it is replaced. +- If the app's program is a bare `dspy.ReAct` / `dspy.ChainOfThought` called directly, wrap it in a small `dspy.Module` subclass named after the agent. Only user-defined module classes become agents (`invoke_agent`, `gen_ai.agent.name` = class name). + +## Step 3: One session per conversation + +Maple reads `session.id` for DSPy (not `gen_ai.conversation.id`). Wrap every call to the program in `using_session` with the app's own stored conversation id: + +```py +from openinference.instrumentation import using_session + +def handle_message(conversation_id: str, question: str, turns: list[dict]) -> str: + with using_session(conversation_id): + answer = assistant(question=question, conversation=dspy.History(messages=turns)).answer + turns.append({"question": question, "answer": answer}) + return answer +``` + +- Use the id the app already stores the chat under. Never a new UUID per request, never a constant. +- For a one-shot job (batch, CLI run, orchestration), mint one id per run and wrap the whole run. +- Async handlers work unchanged (`using_session` is context-based). + +Streaming (`dspy.streamify`): + +```py +stream_assistant = dspy.streamify(assistant, stream_listeners=[dspy.streaming.StreamListener("answer")]) + + +async def stream_message(conversation_id: str, question: str, turns: list[dict]): + with using_session(conversation_id): + async for chunk in stream_assistant(question=question, conversation=dspy.History(messages=turns)): + if isinstance(chunk, dspy.streaming.StreamResponse): + yield chunk.chunk + elif isinstance(chunk, dspy.Prediction): + turns.append({"question": question, "answer": chunk.answer}) +``` + +- Call `dspy.streamify(...)` only after `dspy.configure(callbacks=[MapleCallback()])`: it snapshots the callback list when called. A streamified program created earlier (e.g. at import time) streams without the callback: no tokens, no tool names, `ChatAdapter.__call__` counted as a model call. +- `using_session` must wrap the loop that consumes the stream (inside the generator handed to the SSE/`StreamingResponse`), not only the `stream_assistant(...)` call: the program starts on the first iteration. +- Streamed calls get tokens and cost; they have no `gen_ai.response.id` (DSPy's native engine drops it) and no TTFT attribute. Expected. + +## Step 4: Content + +- Default: prompts, replies, tool arguments and results are captured. Maple shows them as the transcript. +- To turn content off: `OPENINFERENCE_HIDE_INPUTS=true` and `OPENINFERENCE_HIDE_OUTPUTS=true`, set before `tracing.py` / `maple_dspy.py` are imported. The callback honours both. +- If the user mentions PII or compliance, ask whether to disable content; do not decide silently. + +## Step 5: Tools, errors, sub-agents + +- Tools need no changes: plain functions passed to `dspy.ReAct(tools=[...])` or `dspy.Tool(...)` get `.__call__` spans. A raised exception marks the tool span `ERROR` with the message; Maple counts it. Do not catch exceptions inside the tool just to return an error string: that hides the failure. +- Give tools real docstrings; the callback records them as `gen_ai.tool.description`. +- Sub-agents: DSPy's idiom is an orchestrator module calling worker modules. Give each worker its own class with a descriptive name (`WeatherWorker`, `BudgetWorker`). Two instances of one class share one lane. +- Parallel workers: `dspy.Parallel` is fine once `ThreadingInstrumentor` runs. `multiprocessing` is not covered; each process starts its own trace. + +## Step 6: Flush + +- Long-running servers: nothing to add; the provider flushes on normal exit. +- Scripts, CLIs, jobs: `provider.shutdown()` in a `finally` at the end of `main`. +- Serverless handlers and notebooks: `provider.force_flush()` before returning / after each run. + +```py +from tracing import provider + +try: + run() +finally: + provider.shutdown() +``` + +## Step 7: Verify + +Run one conversation of 2-3 messages with the same conversation id (one using a tool), with `cache=False` on the LM so every call hits the provider. Then check in Maple **Agent Sessions** (`https://app.maple.dev/agent-sessions`, or `app.eu.maple.dev`), or with the Maple MCP (`list_agent_sessions`, `get_agent_session`): + +- Exactly one session for the conversation, id = the conversation id, framework **DSPy**. A second conversation is a second session. +- One turn per program call; each turn's root is `.forward`, titled with the user's question. +- Model calls named `LM.__call__`, each with model, input and output tokens, and cost where DSPy knows the price. The LLM call count equals the real number of model calls (not double). +- If the app streams: the streamed turn is in the same session and its `LM.__call__` spans have tokens. +- Transcript non-empty (DSPy's `[[ ## field ## ]]` prompt format is expected). +- Tool calls named `.__call__` with `gen_ai.tool.name`, arguments and results. `finish.__call__` (ReAct's end-of-loop tool) is expected. +- A tool that raised is counted as failed; no successful tool or model call is marked failed. +- Worker modules appear as separate agents/lanes; with `dspy.Parallel`, all workers are in the same trace as the orchestrator, none orphaned. +- The process exited cleanly and the last turn is present (flush ran). + +If a check fails, see the troubleshooting list in the human guide. + +## Do not + +- Do not add `openinference-instrumentation-litellm` or `-openai`. DSPy 3.4 runs most models on its own `lm15` engine, where they record nothing; on the LiteLLM engine they add a duplicate model span per call. +- Do not rely on `gen_ai.conversation.id` for grouping; Maple reads `session.id` for DSPy. +- Do not name a `dspy.Module` attribute `history`: DSPy appends LM calls to it and a `dspy.History` there crashes every model call (`TypeError: object of type 'History' has no len()`). Pass `dspy.History` as an input field. +- Do not set `disable_history=True` or `max_history_size=0`: the callback reads token usage from LM history. +- Do not expect tokens for cache hits (`cache=True` is the default); they cost nothing and are skipped. +- Do not use MLflow autolog or `opentelemetry-instrumentation-genai-dspy` as the Maple path: the first has no session id Maple reads, the second (1.2b0) records no model calls and is not identified as DSPy. +- Do not create a second `TracerProvider` when one exists. +- Do not pass `endpoint=` without `/v1/traces`, or set the env endpoint with it. +- Do not trace optimizer runs (`MIPROv2`, `GEPA`, `BootstrapFewShot`) or `dspy.Evaluate` under the production service name; they make hundreds of calls. diff --git a/skills/maple-agent-tracing-google-adk/SKILL.md b/skills/maple-agent-tracing-google-adk/SKILL.md new file mode 100644 index 0000000000..190461d113 --- /dev/null +++ b/skills/maple-agent-tracing-google-adk/SKILL.md @@ -0,0 +1,172 @@ +--- +name: maple-agent-tracing-google-adk +description: "Trace Google ADK (Agent Development Kit, Python) agents with Maple: register an OTLP tracer provider, switch ADK to GenAI message attributes so Maple shows the transcript, and keep one session per ADK session id. Triggers on 'trace my ADK agent', 'add Maple to Google ADK', 'agent sessions for Google ADK', 'OpenTelemetry for google-adk'." +--- + +# Maple agent tracing: Google ADK (Python) + +Goal: one conversation = one ADK session id = one Maple Agent Session, with the transcript (user, assistant, tool calls and results), model calls, tool calls with arguments/results, failures, and tokens. Reasoning for every step: https://maple.dev/docs/agent-tracing/google-adk + +ADK emits its own OTel spans (scope `gcp.vertex.agent`): `invocation` > `invoke_agent {agent}` > `call_llm` > `generate_content {model}`, plus `execute_tool {tool}`. No instrumentation package is needed. You add: a tracer provider (Runner apps only), the env vars below, one plugin, one span processor. + +## Step 0: Detect versions and existing setup + +- Read `pyproject.toml` / `requirements*.txt` / `uv.lock`. Require `google-adk>=2.10`; upgrade if lower (before 2.7, `gen_ai.system` is always `gemini` and failed tools are not marked ERROR). +- ADK caps `opentelemetry-sdk<=1.42.1`. Never pin a newer OTel package; let the resolver choose. +- Find how ADK runs: + - `adk web` / `adk api_server` (CLI): ADK builds the tracer provider from `OTEL_EXPORTER_OTLP_*` env. Do NOT register another provider. + - `Runner(...)` in the app's own code (FastAPI, worker, script, notebook): nothing is exported unless the app registers a provider. You must add one. +- Grep for an existing OTel setup: `set_tracer_provider`, `TracerProvider(`, `logfire.configure`, `sentry_sdk.init`, `maybe_set_otel_providers`, `opentelemetry-instrument`. If one exists, add the Maple processor to that provider instead of creating a second one. +- Grep for double instrumentation and remove it if it only exists for tracing: `litellm.callbacks` containing `"otel"`, `openinference-instrumentation-google-adk` / `GoogleADKInstrumentor`, `openinference-instrumentation-litellm`, `openinference-instrumentation-openai`, `opentelemetry-instrumentation-openai-v2`. Ask the user before removing something another backend relies on. + +## Step 1: Key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. In `OTEL_EXPORTER_OTLP_HEADERS` write the space as `%20`: `Authorization=Bearer%20`. +- Key given in the prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with a key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention (`.env`, settings module, deployment env). If there is none, inline values are acceptable: ingest keys are write-only. + +## Step 2: Install and initialize + +```bash +pip install "google-adk>=2.10" opentelemetry-exporter-otlp-proto-http +``` + +Add `litellm` only if the app uses `google.adk.models.lite_llm.LiteLlm`. Use the repo's package manager (`uv add`, `poetry add`). + +Env vars (all runtimes, including `adk web`/`api_server`): + +```bash +OTEL_SERVICE_NAME= +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer%20 +OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental +OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY +ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false +``` + +- `OTEL_EXPORTER_OTLP_ENDPOINT` gets `/v1/traces` appended. If you use `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` instead, give the full `https://ingest.maple.dev/v1/traces`. +- The env vars must be in the process environment before the first `run_async()`. If the app loads `.env` with python-dotenv, load it before `import telemetry`. + +For Runner apps, create `telemetry.py` next to the entry point: + +```py +# telemetry.py +import json + +from google.adk.plugins.base_plugin import BasePlugin +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class SkipDuplicateToolSpans(BatchSpanProcessor): + """Drops two ADK tool spans that would count a call twice: the + `execute_tool (merged)` summary of parallel calls, and the span of a call + paused for confirmation (it runs again, in its own span, once approved).""" + + def on_end(self, span): + if span.name != "execute_tool (merged)" and not span.attributes.get("adk.awaiting_confirmation"): + super().on_end(span) + + +class ToolCallAttributes(BasePlugin): + """Records each tool call's arguments and result on its `execute_tool` span, + and marks the span of a call that is waiting for confirmation.""" + + def __init__(self): + super().__init__(name="tool_call_attributes") + + async def before_tool_callback(self, *, tool, tool_args, tool_context): + trace.get_current_span().set_attribute("gen_ai.tool.call.arguments", json.dumps(tool_args, default=str)) + + async def after_tool_callback(self, *, tool, tool_args, tool_context, result): + span = trace.get_current_span() + if tool_context.actions.requested_tool_confirmations: + span.set_attribute("adk.awaiting_confirmation", True) + span.set_attribute("gen_ai.tool.call.result", json.dumps(result, default=str)) + + +# Reads OTEL_SERVICE_NAME, OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider = TracerProvider(resource=Resource.create()) +provider.add_span_processor(SkipDuplicateToolSpans(OTLPSpanExporter())) +trace.set_tracer_provider(provider) +``` + +- `import telemetry` as the first import of the entry module (before agents are built and before any run). +- Existing provider found in Step 0: skip `TracerProvider(...)`/`set_tracer_provider`; call `existing_provider.add_span_processor(SkipDuplicateToolSpans(OTLPSpanExporter()))` right after it is created. Keep `ToolCallAttributes`. +- Register the plugin on every `Runner`: `Runner(..., plugins=[telemetry.ToolCallAttributes()])`, appending to existing plugins. For `adk web`/`api_server`, add it to the `App(name=..., root_agent=..., plugins=[...])` in the agent module and do not create a provider (only the plugin is needed). The processor can't be added there without replacing ADK's provider, so `execute_tool (merged)` spans and paused-confirmation spans stay; tell the user tool counts can be inflated under the CLI. + +## Step 3: Session id + +- ADK writes `session.id` as `gen_ai.conversation.id` on `invoke_agent` and `generate_content` spans; Maple groups on it. One `run_async()` = one trace = one turn. +- Make sure every turn of a conversation calls `runner.run_async(user_id=..., session_id=, new_message=...)` with the SAME id. Use the app's own chat/thread id. `Runner(..., auto_create_session=True)` creates the session under that id on the first turn. +- Fix code that calls `session_service.create_session(...)` without `session_id` on every request: that mints a new UUID per turn (and the model loses history). +- Never use a process-global constant session id for all users. +- Prefer `run_async()`; in servers do not use sync `runner.run()`. + +## Step 4: Content + +- The two OTel env vars from Step 2 put `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.definitions` on `generate_content` spans as `[{role, parts}]` JSON. This is the only content Maple reads. +- `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=true` means log records only. Use `SPAN_ONLY`. +- `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false` removes ADK's `gcp.vertex.agent.*` duplicate blobs (Maple ignores them; they double payload size). +- Per-request alternative: `RunConfig(telemetry=TelemetryConfig(genai_semconv_stability_opt_in="experimental", capture_message_content=ContentCapturingMode.SPAN_ONLY))`, imports from `google.adk.telemetry.context`. +- If the user wants no content: omit `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` and delete the two `gen_ai.tool.call.*` `set_attribute` lines from `ToolCallAttributes` (keep the plugin: it still marks paused-confirmation spans). Tell the user the transcript will be empty. + +## Step 5: Tools, errors, sub-agents + +- A tool call is marked failed (ERROR + `error.type`) when the tool raises, or returns a dict with a non-empty `"error"` key (`error.type=TOOL_ERROR`). +- Tools returning `{"status": "error", "error_message": ...}` are NOT marked failed. If the repo uses that shape, tell the user and offer to change it to `{"error": ...}` (changes what the model sees; confirm first). +- A raised tool exception aborts the run. Only if the user wants the agent to continue, add: + +```py +class ToolErrorsAsResults(BasePlugin): + def __init__(self): + super().__init__(name="tool_errors_as_results") + + async def on_tool_error_callback(self, *, tool, tool_args, tool_context, error): + return {"error": str(error)} +``` + +- Human approval (`FunctionTool(func, require_confirmation=True)`): ADK opens an `execute_tool` span for the paused call and another when the approved call runs. `ToolCallAttributes` marks the paused one and `SkipDuplicateToolSpans` drops it, so each call counts once. Send the approval (`FunctionResponse` named `adk_request_confirmation`) with the same `session_id`: it becomes its own trace in the same session. A rejected call is marked failed (`This tool call is rejected.`). +- Every agent needs a distinct `name` (becomes `gen_ai.agent.name`, Maple's lanes). +- Agents as tools: use `sub_agents=[LlmAgent(..., mode="single_turn")]`. Each call produces `execute_tool {agent}` (counted as a tool call, result = the sub-agent's reply) and a sibling `invoke_agent {agent}` under the parent agent, not nested; several delegations in one model response run in parallel. Replace `AgentTool(agent=...)` if present and the user agrees: `AgentTool` runs the sub-agent in a fresh in-memory session, so its spans carry a second `gen_ai.conversation.id` and Maple may file the turn under the wrong session. +- `SequentialAgent` / `ParallelAgent` / `LoopAgent` need no changes. + +## Step 6: Flush + +- Scripts, CLIs, notebooks, jobs, tests: wrap the work in `try/finally` and call `telemetry.provider.force_flush()` then `telemetry.provider.shutdown()`. +- Servers: call `provider.shutdown()` in the shutdown hook (FastAPI `lifespan`). +- CPU-throttled serverless (Cloud Run default, Cloud Functions, Lambda): call `provider.force_flush()` before returning each response. + +## Step 7: Verify + +Run one conversation of 2-3 turns with the same session id, one turn calling a tool, then flush. With a real key, open Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`); data appears within about a minute. Check: + +- [ ] Exactly one session for the conversation, framework **Google ADK**, one turn per `run_async()`. A second conversation is a separate session. +- [ ] Transcript shows user messages, assistant replies, tool calls with arguments and results. +- [ ] LLM calls and input/output tokens are non-zero on every turn, including streamed ones. Model = `gen_ai.request.model` (e.g. `openrouter/openai/gpt-4o-mini`). +- [ ] Tool calls carry real function names; no tool named `(merged tools)`; an approved confirmation-gated call counts once. +- [ ] A failing tool is counted as an error; successful tools are not. +- [ ] Sub-agents appear as separate lanes with their own names. +- [ ] No duplicated model calls (no second instrumentor). +- [ ] Cost shows "unpriced" (expected: ADK records no cost). + +Without a real key (`MAPLE_TEST`), verify locally: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` to the provider, run one turn, and confirm `generate_content` spans have `gen_ai.conversation.id`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.input_tokens`; `execute_tool` spans have `gen_ai.tool.call.arguments`. Remove the console exporter afterwards. + +Tell the user about the known gaps: cost is unpriced; the model is the requested id, not the served one, and there is no `gen_ai.response.id`; with `StreamingMode.SSE` the transcript shows the streamed chunks and then the full reply (ADK records each chunk). + +## Do not + +- Do not rely on `OTEL_EXPORTER_OTLP_*` env alone with a plain `Runner`: nothing is exported. +- Do not set `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=true` (log records only) or leave out `OTEL_SEMCONV_STABILITY_OPT_IN`: the transcript stays empty. +- Do not create a second `TracerProvider` when one exists; only the first global one wins. +- Do not add `litellm.callbacks=["otel"]`, OpenInference ADK/LiteLLM/OpenAI instrumentors, or the OpenAI OTel instrumentor alongside ADK's spans: doubled model calls and tokens. +- Do not create a new ADK session per request, or share one session id across users. +- Do not use `AgentTool` for sub-agents when tracing matters; use `mode="single_turn"` sub-agents. +- Do not drop or filter `call_llm` spans: `generate_content` would lose its parent. +- Do not use the gRPC OTLP exporter or pin OTel packages above ADK's cap. +- Do not skip `force_flush()`/`shutdown()` in short-lived processes. diff --git a/skills/maple-agent-tracing-haystack/SKILL.md b/skills/maple-agent-tracing-haystack/SKILL.md new file mode 100644 index 0000000000..fd9c83c0bb --- /dev/null +++ b/skills/maple-agent-tracing-haystack/SKILL.md @@ -0,0 +1,258 @@ +--- +name: maple-agent-tracing-haystack +description: "Trace Haystack agents with Maple: installs a small Haystack tracer that adds OpenTelemetry GenAI attributes to Haystack's own spans, so each conversation is one Maple Agent Session with transcript, model, tokens, cost and tool failures. Triggers on 'trace my haystack agent', 'add Maple to haystack', 'agent sessions for haystack', 'OpenTelemetry for haystack'." +--- + +# Maple agent tracing: Haystack + +Goal: one conversation = one Maple Agent Session, with transcript, model calls, tool calls (failures marked), tokens and (on OpenRouter) cost. Reasoning and details: https://maple.dev/docs/agent-tracing/haystack + +Haystack's `OpenTelemetryTracer` alone gives Maple structure only (model, tokens, content live in `haystack.*` JSON blobs Maple doesn't read; failed tools end `Unset`; no conversation id). The fix is `MapleHaystackTracer`, a subclass that writes `gen_ai.*` attributes on the live spans. + +## Step 0: Detect + +- `haystack-ai` version: `python -c "import haystack; print(haystack.__version__)"`. Need `>=3.0`; guide verified on 3.2. For 2.x, stop and tell the user this skill targets Haystack 3. +- Find where Agents/Pipelines run (`Agent(`, `Pipeline(`, `.run(`, `AgentTool(`, `PipelineTool(`) and where a chat/thread id is available per request. +- Existing OTel: grep for `TracerProvider(`, `set_tracer_provider`, `configure_otel`, `logfire.configure`, `opentelemetry-instrument`. If a provider exists, reuse it: only add the Maple exporter (if not already exporting to Maple) and `enable_tracing(...)`. Never create a second provider. +- Remove/skip `HaystackInstrumentor().instrument()` (OpenInference) and OpenLLMetry's Haystack instrumentor if present: they double every model call. Remove any existing `enable_tracing(OpenTelemetryTracer(...))` or `OpenTelemetryConnector` component and replace with Step 2. + +## Step 1: Key and region + +- US endpoint `https://ingest.maple.dev`; EU endpoint `https://ingest.eu.maple.dev`. Header `Authorization=Bearer `. Protocol `http/protobuf`. +- Key given in the prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Private `maple_sk_` keys never go in browser code. +- Follow the repo's secret/env convention (`.env`, settings module, deployment env). If none exists, inline is acceptable: ingest keys are write-only. + +## Step 2: Install and init + +```bash +pip install "haystack-ai>=3.2" "opentelemetry-haystack>=1.0" "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Use the repo's package manager (`uv add`, `poetry add`, requirements file). Write this file verbatim as `maple_haystack.py` in the app's package: + +```py +"""Haystack tracer that adds the OpenTelemetry GenAI attributes Maple reads.""" + +import json +import logging +from collections.abc import Iterator +from contextlib import contextmanager +from contextvars import ContextVar +from typing import Any + +from haystack.dataclasses import ChatMessage +from haystack_integrations.tracing.opentelemetry import OpenTelemetrySpan, OpenTelemetryTracer +from opentelemetry import trace +from opentelemetry.trace import StatusCode + +logger = logging.getLogger(__name__) +_conversation_id: ContextVar[str | None] = ContextVar("maple_conversation_id", default=None) + + +@contextmanager +def conversation(conversation_id: str) -> Iterator[None]: + """Every Haystack run inside this block joins the same Maple session.""" + token = _conversation_id.set(conversation_id) + try: + yield + finally: + _conversation_id.reset(token) + + +def _messages(messages: list[ChatMessage]) -> str: + out = [] + for m in messages: + parts: list[dict[str, Any]] = [{"type": "text", "content": t} for t in m.texts] + parts += [{"type": "tool_call", "id": c.id, "name": c.tool_name, "arguments": c.arguments} for c in m.tool_calls] + parts += [{"type": "tool_call_response", "id": r.origin.id, "response": r.result} for r in m.tool_call_results] + out.append({"role": m.role.value, "parts": parts}) + return json.dumps(out, default=str) + + +class MapleSpan(OpenTelemetrySpan): + def __init__(self, span: trace.Span, operation: str | None, content: bool) -> None: + super().__init__(span) + self._operation = operation + self._content = content + + def set_content_tag(self, key: str, value: Any) -> None: + try: + if self._operation == "chat": + self._chat(key, value) + elif self._operation == "execute_tool": + self._tool(key, value) + except Exception: # a tracing bug must never fail the agent run + logger.exception("maple_haystack: could not map %s", key) + if self._content: + self.set_tag(key, value) + + def _chat(self, key: str, value: Any) -> None: + if key.endswith(".input") and self._content: + messages = value["messages"] + system = [{"type": "text", "content": m.text} for m in messages if m.is_from("system")] + if system: + self._span.set_attribute("gen_ai.system_instructions", json.dumps(system)) + self._span.set_attribute("gen_ai.input.messages", _messages([m for m in messages if not m.is_from("system")])) + elif key.endswith(".output"): + replies = value["replies"] + meta = replies[0].meta + usage = meta.get("usage") or {} + prompt_details = usage.get("prompt_tokens_details") or {} + attributes = { + "gen_ai.response.model": meta.get("model"), + "gen_ai.response.finish_reasons": meta.get("finish_reason"), + "gen_ai.usage.input_tokens": usage.get("prompt_tokens"), + "gen_ai.usage.output_tokens": usage.get("completion_tokens"), + "gen_ai.usage.cache_read.input_tokens": prompt_details.get("cached_tokens"), + "gen_ai.usage.cache_write.input_tokens": prompt_details.get("cache_write_tokens"), + "gen_ai.usage.reasoning.output_tokens": (usage.get("completion_tokens_details") or {}).get("reasoning_tokens"), + "gen_ai.usage.cost": usage.get("cost"), # OpenRouter prices every call; other providers leave this out + } + if self._content: + attributes["gen_ai.output.messages"] = _messages(replies) + self._span.set_attributes({k: v for k, v in attributes.items() if v is not None}) + + def _tool(self, key: str, value: Any) -> None: + if key.endswith(".output") and isinstance(value, dict) and "error" in value: + # Haystack records a failed tool as {"error": ...} and leaves the span status unset + self._span.set_status(StatusCode.ERROR, str(value["error"])) + self._span.set_attribute("error.type", "ToolInvocationError") + if self._content: + attribute = "gen_ai.tool.call.arguments" if key.endswith(".input") else "gen_ai.tool.call.result" + self._span.set_attribute(attribute, json.dumps(value, default=str)) + + +class MapleHaystackTracer(OpenTelemetryTracer): + def __init__(self, tracer: trace.Tracer, *, content: bool = True) -> None: + super().__init__(tracer) + self._content = content + + @contextmanager + def trace(self, operation_name: str, tags: dict[str, Any] | None = None, parent_span: Any = None) -> Iterator[MapleSpan]: + tags = dict(tags or {}) + attributes: dict[str, str] = {} + operation = None + if operation_name == "haystack.agent.run": + # The Agent span has no name: use the pipeline component or the AgentTool that runs it + parent = getattr(trace.get_current_span(), "attributes", None) or {} + attributes["gen_ai.operation.name"] = "invoke_agent" + attributes["gen_ai.agent.name"] = parent.get("haystack.component.name") or parent.get("gen_ai.tool.name") or "agent" + elif operation_name == "haystack.agent.step.llm" or str(tags.get("haystack.component.type", "")).endswith("ChatGenerator"): + operation = attributes["gen_ai.operation.name"] = "chat" + elif operation_name == "haystack.agent.step.tool": + operation = attributes["gen_ai.operation.name"] = "execute_tool" + attributes["gen_ai.tool.name"] = tags["haystack.tool.name"] + if conversation_id := _conversation_id.get(): + attributes["gen_ai.conversation.id"] = conversation_id + if not self._content: + tags.pop("haystack.pipeline.input_data", None) # a plain tag, not gated by Haystack's content switch + + with self._tracer.start_as_current_span(operation_name, attributes=attributes) as raw_span: + span = MapleSpan(raw_span, operation, self._content) + span.set_tags(tags) + yield span +``` + +Init once at startup, before the first `pipeline.run()`/`agent.run()` (no import-order constraint beyond that): + +```py +from haystack import tracing +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +from maple_haystack import MapleHaystackTracer + +provider = TracerProvider( + resource=Resource.create({"service.name": "", "deployment.environment.name": ""}) +) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) +tracing.enable_tracing(MapleHaystackTracer(trace.get_tracer("haystack"))) +``` + +Env: + +```bash +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev # EU: https://ingest.eu.maple.dev; /v1/traces is appended +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +- Tracer name must be `"haystack"` (instrumentation scope = how Maple labels the vendor Haystack). +- Set a real `service.name`; never leave `unknown_service`. + +## Step 3: Session id + +Haystack has no conversation concept; the app owns the message history. Wrap every run in `conversation()`: + +```py +from maple_haystack import conversation + +with conversation(chat_id): + result = pipeline.run({"assistant": {"messages": [*history, ChatMessage.from_user(text)]}}) +history = [m for m in result["assistant"]["messages"] if not m.is_from("system")] # Agent re-adds its system prompt +``` + +- `chat_id` = the id the app already stores for the chat/thread (DB id, frontend thread id). Never a per-request UUID, never a process-wide constant. +- Wrap the run call itself (inside the request handler). It is a ContextVar: request-scoped and carried into Haystack's tool threads. +- It writes `gen_ai.conversation.id` on every span. That is the only session key Maple reads for Haystack (`session.id` is ignored). + +## Step 4: Content + +- Default `MapleHaystackTracer(..., content=True)` writes `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments`/`.result`, plus Haystack's own `haystack.*` content tags. +- `content=False` keeps model, tokens, cost, finish reason, tool names and failures; drops messages, tool args/results, `haystack.*` content tags and the ungated `haystack.pipeline.input_data` tag. Use it if the user asks for no prompts/PII in telemetry. The failed tool's status message still quotes its arguments (Haystack's `Failed to invoke Tool ... with parameters {...}`); if arguments can hold PII, redact them in `_tool()` before `set_status`. +- `HAYSTACK_CONTENT_TRACING_ENABLED` is irrelevant with this tracer; don't add it. + +## Step 5: Tools, errors, sub-agents + +- Tool spans: `haystack.agent.step.tool` → `execute_tool` + `gen_ai.tool.name`. A tool that raises becomes `{"error": ...}` in Haystack; the tracer sets status `Error` + `error.type=ToolInvocationError`. Don't catch exceptions inside tools just to return strings; let them raise (or return `{"error": ...}`). +- Agent name (`gen_ai.agent.name`, needed for Maple lanes) = pipeline component name, or the `AgentTool` name, else `agent`. If the app calls `agent.run()` directly for its main agent, prefer running it as a `Pipeline` component with a meaningful name (only if that's a small change; otherwise leave it and mention the `agent` name). +- Sub-agents: use `AgentTool(agent=..., name="", description=...)` (Haystack ≥3.1) or `PipelineTool`; each worker gets its own name/lane. Keep them under the same `conversation()` block. +- Streaming with `OpenAIChatGenerator`: add `generation_kwargs={"stream_options": {"include_usage": True}}` or the streamed turn has no tokens. OpenRouter always sends usage. +- Token mapping reads OpenAI-style `meta["usage"]` keys (`prompt_tokens`, `completion_tokens`, `*_details`). For another generator, print one reply's `meta["usage"]`; if keys differ, extend `_chat()`. +- Cost: only OpenRouter reports `usage.cost` → `gen_ai.usage.cost`. Other providers show "unpriced" in Maple; say so to the user. + +## Step 6: Flush + +Scripts, CLIs, notebooks, jobs, serverless: flush before exit. + +```py +try: + run() +finally: + provider.force_flush() + provider.shutdown() +``` + +Servers: `provider.shutdown()` in the shutdown hook (FastAPI lifespan, atexit). + +## Step 7: Verify + +Run one real conversation (≥2 turns, one tool call). If you can export to a local collector or console first, check the spans; otherwise check Maple (Agent Sessions, filter by service). Check: + +- Every span has `service.name` set; spans' scope is `haystack`. +- Each `haystack.agent.step.llm` span has `gen_ai.operation.name=chat`, `gen_ai.response.model`, `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` (also on the streamed turn). +- `gen_ai.input.messages`/`gen_ai.output.messages` are JSON arrays of `{role, parts}` (content on). +- Each tool call has one `haystack.agent.step.tool` span with the real `gen_ai.tool.name`; a failing tool has status `Error` and `error.type`; successful tools don't. +- `haystack.agent.run` has `gen_ai.operation.name=invoke_agent` and a distinct `gen_ai.agent.name` per sub-agent. +- Every trace of one conversation carries the same `gen_ai.conversation.id`; a second conversation carries a different one. +- No model call appears twice (no second instrumentor). +- No attribute contains an API key or `Bearer `. +- In Maple: one session per conversation labelled Haystack, turns = runs, transcript non-empty, LLM/tool counts match, failed tool counted, cost shown (OpenRouter) or unpriced. + +Tell the user: known gaps are no `gen_ai.provider.name`, no `gen_ai.response.id`, no tool call id on tool spans, and cost only on OpenRouter. + +## Do not + +- Don't rely on `OpenTelemetryTracer`/`OpenTelemetryConnector` alone: Maple gets no model, tokens, transcript or tool failures. +- Don't add `openinference-instrumentation-haystack` or OpenLLMetry's Haystack instrumentor alongside (double model calls; OpenInference has no tool spans and its `session.id` is ignored). +- Don't create a second `TracerProvider`; don't rename the tracer from `"haystack"`. +- Don't mint a conversation id per request or share one across users. +- Don't set `session.id` or `maple_ai.session.id` for Haystack; `gen_ai.conversation.id` via `conversation()` is the key. +- Don't expect Haystack 2.x auto-tracing: 3.x needs `enable_tracing(...)`. +- Don't use a console exporter in production or forget the flush in short-lived processes. +- Don't modify `maple_haystack.py` beyond extending usage keys or redaction. diff --git a/skills/maple-agent-tracing-langchain/SKILL.md b/skills/maple-agent-tracing-langchain/SKILL.md new file mode 100644 index 0000000000..77e8b0fbb9 --- /dev/null +++ b/skills/maple-agent-tracing-langchain/SKILL.md @@ -0,0 +1,204 @@ +--- +name: maple-agent-tracing-langchain +description: "Trace LangChain and LangGraph agents with Maple: export OpenInference LangChain spans with the GenAI dual-write so each thread is one Maple Agent Session with transcript, tool calls, sub-agent lanes and tokens. Triggers on 'trace my langchain agent', 'trace my langgraph agent', 'add Maple to langchain', 'add Maple to langgraph', 'agent sessions for langchain', 'OpenTelemetry for langgraph'." +--- + +# Maple agent tracing: LangChain & LangGraph (Python) + +Goal: every conversation = one Maple Agent Session. Each `invoke()`/`stream()` = one turn (one trace) with a readable transcript, chat model spans with tokens, tool spans with names/results, failed tools marked failed, sub-agents in their own lanes. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/langchain + +Mechanism: `openinference-instrumentation-langchain` (scope `openinference.instrumentation.langchain`) with `TraceConfig(enable_genai_semconv=True)`, which dual-writes `gen_ai.*` (incl. `gen_ai.conversation.id` from run metadata `session_id` > `conversation_id` > `thread_id`, and `gen_ai.input/output.messages` in `{role, parts}` form). Maple classifies these as generic GenAI (framework facet "Unidentified") and reads `gen_ai.conversation.id` as the session key. This beats LangSmith's OTel export for Maple: readable transcript, interrupts not marked ERROR, no middleware noise spans, normal flush. Python only; LangChain.js/LangGraph.js are not covered (tell the user and stop). + +## Step 0: detect + +1. Versions: `python -c "import langchain, langgraph, langchain_core; print(langchain.__version__, langchain_core.__version__)"` and `pip show langgraph` (or read `pyproject.toml` / `uv.lock` / `requirements*.txt`). Tested: langchain 1.4.2, langgraph 1.2.12, langchain-core 1.6.5, langchain-openai 1.6.6, Python 3.12. Python must be >= 3.10. +2. Existing OTel setup. Search for `TracerProvider(`, `set_tracer_provider`, `logfire.configure`, `sentry_sdk.init`, `opentelemetry-instrument`, `Traceloop.init`, `LangChainInstrumentor`, `OpenAIInstrumentor`, `LANGSMITH_OTEL_ENABLED`, `LANGSMITH_TRACING_MODE`. + - A `TracerProvider` exists → add the processors from Step 2 to it; do NOT create a second provider. + - `LANGSMITH_OTEL_ENABLED`/`LANGSMITH_TRACING_MODE=otel` set → it duplicates every run. Ask the user; remove it for Maple (plain `LANGSMITH_TRACING=true` to LangSmith cloud is fine to keep). + - OpenAI/Anthropic OpenInference instrumentors or OpenLLMetry LangChain instrumentor → duplicate model spans. Ask before removing if they serve something else. +3. Find: every `create_agent(`, `create_react_agent(`, `StateGraph(`/`.compile(`, every `.invoke(`/`.ainvoke(`/`.stream(`/`.astream(`/`Command(resume=` call on an agent/graph/chain, where the app's chat/thread id lives, every `ChatOpenAI(` (note `base_url`), and every tool that invokes another agent. +4. LangGraph Server (`langgraph.json` present): `import tracing` at the top of the module(s) `langgraph.json`'s `graphs` points to (the dir holding `tracing.py` must be in `dependencies`); `OTEL_*` env goes in the server env file/container. Server threads already carry `configurable.thread_id`: one Maple session per thread, no code. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention (`.env`, settings module, secret manager) if it has one. Otherwise inline is acceptable: ingest keys are write-only. + +Env vars (`OTLPSpanExporter()` with no args reads them and appends `/v1/traces`): + +```bash +OTEL_SERVICE_NAME= +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name= +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +``` + +If you pass `OTLPSpanExporter(endpoint=...)` in code, it must end in `/v1/traces` (used verbatim). + +## Step 2: install + init + +Add with the repo's package manager (uv/poetry/pip): + +```bash +pip install "openinference-instrumentation-langchain>=0.1.76" "openinference-instrumentation>=0.1.66" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Pin `openinference-instrumentation>=0.1.66` explicitly (the langchain instrumentor allows 0.1.61, which may lack the GenAI dual-write). + +Create `tracing.py`: + +```py +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.langchain import LangChainInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +# The name= of every create_agent(), and graph nodes that act as agents +AGENT_NAMES = {"assistant"} +# LangGraph's tool node and prompt templates: steps, not tool or model calls +STEP_NAMES = {"tools", "ChatPromptTemplate"} + + +class AgentSpans(SpanProcessor): + """Names your agents' spans for Maple (one lane per agent) and keeps graph steps out of the tool and model counts.""" + + def on_start(self, span, parent_context=None): + if span.instrumentation_scope.name != "openinference.instrumentation.langchain": + return + if span.name in AGENT_NAMES: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", span.name) + elif span.name in STEP_NAMES: + span.set_attribute("gen_ai.operation.name", "invoke_workflow") + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(AgentSpans()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +LangChainInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +- Import `tracing` first in the entry point (app module, `main.py`, worker, LangGraph Server graph module). It must run before the first `invoke()`. +- Fill `AGENT_NAMES` with every agent's `name=` from Step 0.3. Give unnamed `create_agent(...)` calls a `name=` (default graph name is `LangGraph`). +- `STEP_NAMES`: `tools` is the tool node of `create_agent` and the usual `ToolNode` name. Add any other graph node whose name contains "tool" (e.g. `add_node("run_tools", ToolNode(...))`); Maple counts unmarked ones as extra tool calls. Never add real tool names. +- Existing provider: add `AgentSpans()` and the `BatchSpanProcessor(OTLPSpanExporter())` to it and pass it as `tracer_provider=`. +- Set a real `service.name` (never `unknown_service`). +- Streaming with `ChatOpenAI(base_url=...)` or `OPENAI_BASE_URL` set: add `stream_usage=True` to the `ChatOpenAI(...)` constructor. ChatOpenAI only requests streamed usage from api.openai.com; servers that don't send it unasked (vLLM, many gateways) give streamed calls no tokens. OpenRouter sends it anyway; set it regardless. + +## Step 3: session id (required) + +Pass the app's conversation id as `thread_id` on EVERY agent/graph call, including streams and HITL resumes: + +```py +result = agent.invoke( + {"messages": [{"role": "user", "content": text}]}, + {"configurable": {"thread_id": conversation_id}}, +) +``` + +```py +agent.invoke(Command(resume={"decisions": [{"type": "approve"}]}), {"configurable": {"thread_id": conversation_id}}) +``` + +- Works with or without a checkpointer (LangGraph copies `configurable.thread_id` into run metadata). If the graph already uses a checkpointer, reuse its existing `thread_id`; don't invent a second id. +- Plain LangChain chains (`prompt | model`, no graph) do NOT copy `configurable`: pass `{"metadata": {"thread_id": conversation_id}}` instead. +- Nested runs (agents called inside tools or nodes) inherit it; don't pass a different id to them. +- The id must be stable per conversation and unique across conversations: no constants, no `uuid4()` per request. No id in the app → ask where the conversation boundary is; single-shot script → one uuid per conversation, reused. +- Do not set `session.id`/`gen_ai.conversation.id`/`maple_ai.session.id` by hand. + +## Step 4: content + +- On by default: `gen_ai.input.messages` / `gen_ai.output.messages` (system prompt included as the first input message) on chat model spans; `gen_ai.tool.call.result` on tool spans. Maple's transcript needs these. +- User wants content off → `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` (or `OPENINFERENCE_HIDE_INPUTS=true` / `OPENINFERENCE_HIDE_OUTPUTS=true`). The gen_ai copies are built from masked values, so they're empty too. Tell them the transcript will be empty; turns, tools, tokens and failures remain. +- Narrower: `hide_input_text`, `hide_output_text`. Pattern redaction → OTel Collector `redaction` processor. + +## Step 5: tools, errors, sub-agents + +1. Tool exceptions mark the tool span ERROR with the message automatically. `create_agent` re-raises them and aborts the run; if the app should continue, add (ask the user if behaviour changes matter): + +```py +from langchain.agents.middleware import wrap_tool_call +from langchain_core.messages import ToolMessage + + +@wrap_tool_call +def tool_errors_to_model(request, handler): + try: + return handler(request) + except Exception as e: + return ToolMessage(content=f"Tool error: {e}", tool_call_id=request.tool_call["id"], status="error") +``` + + and `middleware=[tool_errors_to_model, ...]` on `create_agent`. For `ToolNode` in a `StateGraph`: `ToolNode(tools, handle_tool_errors=True)`. The tool span stays ERROR either way. + - Tools that `return "Error: ..."` show as successful calls. Prefer raising, where the user agrees. +2. Interrupts (`interrupt()`, `HumanInTheLoopMiddleware`) end with status OK; the resume is a new trace in the same session (same `thread_id`). +3. Sub-agents: give each a unique `name=` and add it to `AGENT_NAMES`. Agent-as-tool pattern: + +```py +weather_worker = create_agent(model, tools=[get_weather], name="weather_worker") + + +@tool +def ask_weather_worker(city: str) -> str: + """Ask the weather worker for the current weather in a city.""" + result = weather_worker.invoke({"messages": [{"role": "user", "content": f"Weather in {city}?"}]}) + return result["messages"][-1].content +``` + + The nested run inherits the caller's trace and `thread_id`. For `StateGraph` workers as nodes, add the node names to `AGENT_NAMES`. +4. Never put "agent" in a tool name (`ask_weather_agent`): OpenInference then marks the span AGENT, and it stops counting as a tool call. +5. Python 3.10 + async: pass the node's `config` to nested `ainvoke()` calls. Own thread pools: `from langchain_core.runnables.config import ContextThreadPoolExecutor`. + +## Step 6: flush + +- `BatchSpanProcessor` exports every 5 s; the SDK flushes at normal interpreter exit. Long-running servers (incl. LangGraph Server) need nothing. +- AWS Lambda / Cloud Functions / Cloud Run jobs: `provider.force_flush()` in a `finally` in the handler. +- Scripts, CLIs, one-shot jobs: `provider.shutdown()` at the end (`finally`). +- Celery/RQ workers, notebooks: `provider.force_flush()` after each task/cell that runs an agent. + +## Step 7: verify + +Run one real conversation: 2+ messages with the same id, at least one tool call, one streamed message if the app streams, a sub-agent call if the app delegates. Then check (Maple → Agent Sessions, filter by service name; wait up to ~1 min): + +- [ ] Exactly one session per conversation, id = the `thread_id` you passed. A second conversation is a different session. +- [ ] Framework shows **Unidentified** (expected for this path). +- [ ] One turn per `invoke()`; transcript shows user messages as turn labels, assistant replies, tool calls (not a raw JSON blob). +- [ ] Each turn trace starts at the agent span (`name=`), with `model`/`tools` node spans, `ChatOpenAI` (or other chat model class) spans and tool spans named after the tools, all in one trace. +- [ ] Tool call count = the tools the model actually called (a higher count means a tool node is missing from `STEP_NAMES`). +- [ ] Every chat model span has input and output tokens, including the streamed one. +- [ ] A failing tool is marked failed with its message; successful tools and interrupts are not. +- [ ] Sub-agents appear as separate lanes, all in the caller's session. +- [ ] Cost shows "unpriced" (expected: nothing records cost). +- [ ] No attribute contains an API key or `Bearer ` token. + +Known gaps (not setup bugs, don't try to fix): tool spans have no `gen_ai.tool.call.id` or arguments (arguments are in the model's tool call in the transcript); chat model spans have no `gen_ai.response.id`. + +Local check without Maple: add `SimpleSpanProcessor(ConsoleSpanExporter())` temporarily and confirm `gen_ai.conversation.id` is identical on every span of every turn of one conversation. + +## Do not + +- Do not skip `enable_genai_semconv=True`: without it Maple ignores the session id and shows a raw JSON transcript. +- Do not also enable LangSmith's OTel export (`LANGSMITH_OTEL_ENABLED`, `LANGSMITH_OTEL_ONLY`, `LANGSMITH_TRACING_MODE=otel`) or a provider-level instrumentor (`openinference-instrumentation-openai`/`-anthropic`, OpenLLMetry): duplicate spans and doubled tokens. +- Do not create a second `TracerProvider` when one exists. +- Do not generate a new `thread_id` per request, and do not forget it on `stream()` and `Command(resume=...)` calls. +- Do not rely on `configurable` for plain chains; use `metadata`. +- Do not forget `stream_usage=True` on `ChatOpenAI` with a custom `base_url`. +- Do not name tools with "agent" in them. +- Do not add `maple_ai.session.id` attributes (they re-vendor the span). +- Do not use `gen_ai.operation.name=agent_step` for step spans: Maple fingerprints it as the Vercel AI SDK. Use `invoke_workflow` as above. +- Do not promise cost or the "LangChain" framework label in Maple; do not add token pricing code. +- Do not print or commit real keys beyond the repo's convention. diff --git a/skills/maple-agent-tracing-litellm/SKILL.md b/skills/maple-agent-tracing-litellm/SKILL.md new file mode 100644 index 0000000000..ef7d07673a --- /dev/null +++ b/skills/maple-agent-tracing-litellm/SKILL.md @@ -0,0 +1,252 @@ +--- +name: maple-agent-tracing-litellm +description: "Trace LiteLLM agents with Maple: export LiteLLM's v2 OpenTelemetry spans (Python SDK or self-hosted LiteLLM Proxy) plus your own agent/tool spans so each conversation is one Maple Agent Session with transcript, tool calls, tokens and optional cost. Triggers on 'trace my litellm agent', 'add Maple to litellm', 'agent sessions for litellm', 'OpenTelemetry for litellm', 'trace the litellm proxy'." +--- + +# Maple agent tracing: LiteLLM + +Goal: every conversation = one Maple Agent Session. Each agent run = one turn (one trace) with transcript, LiteLLM `chat ` spans with tokens, `execute_tool` spans with args/results, failed tools marked failed, sub-agents in their own lanes. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/litellm + +Mechanism: +- LiteLLM traces model calls only (scope `litellm`). It has no agent loop, no tool executor, no conversation concept. The app MUST emit `invoke_agent` and `execute_tool` spans itself. +- Use LiteLLM's **v2** OTel logger (`OpenTelemetryV2`). It writes `gen_ai.operation.name=chat`, provider, model, usage, `gen_ai.response.id`, TTFT, `error.type`, and `gen_ai.conversation.id` from `litellm_session_id`. Maple reads `gen_ai.conversation.id` as the session key for LiteLLM. +- The default v1 logger (`litellm.callbacks=["otel"]` without v2) is wrong for Maple: op `acompletion`, no conversation id ever, and under an open parent span it writes onto the (ended) parent and the data is dropped. + +## Step 0: detect + +1. Versions: `python -c "import importlib.metadata as m; print(m.version('litellm'), m.version('opentelemetry-api'))"` (or read `pyproject.toml` / lock files). + - LiteLLM 1.103.x (tested 1.103.0): OpenTelemetry must be **<= 1.43.0**. On >= 1.44 the v2 logger fails to import (`No module named 'opentelemetry._events'`), LiteLLM logs one line and exports nothing. Fixed in LiteLLM >= 1.104 (then any OTel version). + - `python -c "from litellm.integrations.otel.logger import OpenTelemetryV2"` must succeed; if the module is missing, upgrade LiteLLM to >= 1.103. +2. Which path: + - App calls `litellm.acompletion(` / `litellm.completion(` / `Router(` → **SDK path** (Steps 2a-6). + - App calls a LiteLLM Proxy (OpenAI client with `base_url` pointing at the proxy, a `config.yaml` with `model_list`, docker `ghcr.io/berriai/litellm`) → **proxy path** (Step 2b). Only if the user controls the proxy; otherwise tell them and trace in-app with an OpenAI client instrumentor (out of scope here). +3. Sync vs async: grep for `litellm.completion(`, `litellm.text_completion(`, `router.completion(`. The v2 logger traces **async calls only** (`acompletion`, `Router.acompletion`); sync calls produce no span. Convert the agent loop to async where feasible. If the project is sync-only and can't change, use the v1 fallback in Step 2c. +4. Existing OTel: search `TracerProvider(`, `set_tracer_provider`, `opentelemetry-instrument`, `logfire.configure`, `sentry_sdk.init`, `litellm.callbacks`, `success_callback`, `"otel"`, `LITELLM_OTEL_V2`, `OpenAIInstrumentor`, `Traceloop.init`. + - A `TracerProvider` exists → reuse it (pass it as `tracer_provider=`), add a Maple `BatchSpanProcessor` to it; do NOT create a second one. + - `"otel"` already in `litellm.callbacks`/`success_callback` → remove it when adding the v2 instance (two loggers = duplicate spans). + - An OpenAI/LiteLLM client instrumentor (OpenInference `LiteLLMInstrumentor`/`OpenAIInstrumentor`, OpenLLMetry, openai-v2) → duplicates LiteLLM's spans. Keep one; ask before removing. +5. Find: the agent loop (where `acompletion` results' `tool_calls` are executed), every tool function, where the chat/thread/conversation id lives, and any multi-agent orchestration. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention if it has one. Otherwise inline is acceptable: ingest keys are write-only. + +```bash +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +The Python OTLP exporter appends `/v1/traces` to the base endpoint. If you pass `endpoint=` to `OTLPSpanExporter(...)` in code instead, pass the full `https://ingest.maple.dev/v1/traces`. + +## Step 2a: SDK install + init + +```bash +pip install "litellm==1.103.0" "opentelemetry-sdk==1.43.0" "opentelemetry-exporter-otlp-proto-http==1.43.0" +``` + +(LiteLLM >= 1.104: OTel pin can be lifted.) Use the repo's package manager; keep existing litellm extras. + +`tracing.py` (adapt service name / environment): + +```py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +import litellm +from litellm.integrations.otel.logger import OpenTelemetryV2 +from litellm.integrations.otel.model.config import OpenTelemetryV2Config + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +litellm.callbacks = [ + OpenTelemetryV2( + config=OpenTelemetryV2Config(capture_message_content="span_only"), + tracer_provider=provider, + ) +] + +tracer = trace.get_tracer("support-agent") +``` + +- Import it first in the entry point, before the first model call. +- Passing the instance: no `LITELLM_OTEL_V2` env needed, LiteLLM builds no provider/exporter of its own. +- If the project appends to `litellm.callbacks` elsewhere (other loggers), append the instance instead of overwriting the list. +- Set a real `service.name`. + +## Step 2b: proxy path (self-hosted LiteLLM Proxy) + +Proxy `config.yaml`: add `litellm_settings: { callbacks: ["otel"] }` (merge with existing callbacks). Proxy environment: + +```bash +LITELLM_OTEL_V2=true +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer " +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +OTEL_SERVICE_NAME=litellm-proxy +OTEL_ENVIRONMENT_NAME=production +OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only +``` + +- Docker image `ghcr.io/berriai/litellm` ships OTel 1.28 + FastAPI instrumentation: fine. pip-installed proxy on 1.103: add `opentelemetry-sdk==1.43.0 opentelemetry-exporter-otlp-proto-http==1.43.0 opentelemetry-instrumentation-fastapi==0.64b0`. +- App side: keep Step 3-5 spans (`agent_span`, `run_tool`), do NOT register the LiteLLM logger in the app, and on every request to the proxy send `traceparent` (so proxy spans join the app trace under `invoke_agent`) and `x-litellm-session-id` (becomes `gen_ai.conversation.id`): + +```py +import os + +from openai import AsyncOpenAI +from opentelemetry import propagate + +client = AsyncOpenAI(base_url="http://localhost:4000", api_key=os.environ["LITELLM_API_KEY"]) + + +async def call_model(conversation_id: str, messages: list, tools: list | None): + headers = {"x-litellm-session-id": conversation_id} + propagate.inject(headers) # must run inside the agent span + return await client.chat.completions.create( + model="gpt-4o-mini", messages=messages, tools=tools, extra_headers=headers + ) +``` + +- Alternative session carrier: body `metadata: {"session_id": ...}` (`extra_body={"metadata": {...}}`). +- Never also instrument the app's OpenAI client when the proxy traces: double LLM calls and tokens. Pick gateway OR in-app. +- Do not set `OTEL_IGNORE_CONTEXT_PROPAGATION=true` on the proxy. + +## Step 2c: sync-only fallback (v1, only if Step 0.3 says so) + +`litellm.callbacks = ["otel"]` after `trace.set_tracer_provider(provider)` (v1 reuses the global SDK provider), plus env `USE_OTEL_LITELLM_REQUEST_SPAN=true` and `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental`. v1 has no conversation id: set `gen_ai.conversation.id` on each `invoke_agent` span. Consequence: framework shows **Unidentified** in Maple; tell the user. v1 content is on by default. + +## Step 3: agent loop spans + session id (required) + +Wrap each agent run in `invoke_agent` and pass `litellm_session_id=` on EVERY `acompletion`: + +```py +@contextmanager +def agent_span(name: str): + with tracer.start_as_current_span(f"invoke_agent {name}") as span: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", name) + yield span + + +async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: + with agent_span(agent.name): + while True: + response = await litellm.acompletion( + model=agent.model, + messages=[{"role": "system", "content": agent.instructions}, *messages], + tools=agent.schemas or None, + litellm_session_id=conversation_id, + ) + message = response.choices[0].message + messages.append(message.model_dump(exclude_none=True)) + if not message.tool_calls: + return message.content or "" + for call in message.tool_calls: + messages.append({"role": "tool", "tool_call_id": call.id, "content": run_tool(agent, call)}) +``` + +- Adapt to the project's existing loop; don't rewrite it into this shape if it already has one. What matters: one `invoke_agent` span per agent run, all model calls and tool calls inside it, `litellm_session_id=` on every call (`metadata={"session_id": ...}` also works when the project already passes metadata). +- Id: the app's chat/thread/conversation id. Stable per conversation, unique across conversations. No process-wide constants, no `uuid4()` per request. Single-shot script: one uuid per conversation, reused. +- Do NOT put `gen_ai.conversation.id` or `maple_ai.session.id` on your own spans in the v2 setup: Maple labels the session with the vendor of the earliest session-bearing span, so the session would show **Unidentified**. LiteLLM's `chat` spans carrying the id is enough (Maple groups whole traces). +- Streaming: `stream=True, stream_options={"include_usage": True}`, and consume the stream inside `agent_span`; LiteLLM closes its span when the stream ends. + +## Step 4: content + +- v2 default is `no_content`. `capture_message_content="span_only"` (or env `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only`) puts `gen_ai.input.messages` / `gen_ai.output.messages` JSON on `chat` spans. Maple's transcript needs them. +- Never `event_only` / `span_and_event` for Maple: events are not read. +- User wants content off → `no_content` + drop the tool args/result attributes in `run_tool`. `litellm.turn_off_message_logging = True` keeps structure but replaces text with `redacted-by-litellm`. + +## Step 5: tools, errors, sub-agents + +```py +def run_tool(agent: Agent, call) -> str: + name = call.function.name + with tracer.start_as_current_span(f"execute_tool {name}") as span: + span.set_attribute("gen_ai.operation.name", "execute_tool") + span.set_attribute("gen_ai.tool.name", name) + span.set_attribute("gen_ai.tool.call.id", call.id) + span.set_attribute("gen_ai.tool.call.arguments", call.function.arguments) + try: + result = agent.tools[name](**json.loads(call.function.arguments or "{}")) + except Exception as exc: + span.set_status(StatusCode.ERROR, str(exc)) + span.set_attribute("error.type", type(exc).__name__) + result = {"error": str(exc)} + output = json.dumps(result) + span.set_attribute("gen_ai.tool.call.result", output) + return output +``` + +- Real tool name in `gen_ai.tool.name`, the model's `call.id` in `gen_ai.tool.call.id`. Async tools: `await` them inside the span. +- Tool failure: ERROR status + `error.type` on the tool span; the model still gets an error payload. Where the existing loop swallows exceptions into strings, add the status/`error.type` there (don't change what the model sees unless asked). +- LiteLLM marks failed model calls ERROR + `error.type` itself. +- Multi-agent: each agent its own `name` (→ `gen_ai.agent.name`, one lane each). Run workers inside an outer `agent_span("orchestrator")` so the whole run is one trace; `asyncio.gather` keeps context. Same `conversation_id` to every `run_agent`. Delegation via a tool: call `run_agent` inside that tool's function (an `execute_tool` whose only child is `invoke_agent` renders as a delegation). +- Thread pools (`run_in_executor`, `ThreadPoolExecutor`) lose OTel context: wrap with `contextvars.copy_context().run`. + +## Step 6: cost (optional) and flush + +Cost: Maple reads only `gen_ai.usage.cost`; LiteLLM writes `litellm.cost.total` (v2) / `hidden_params` (v1) → unpriced by default. If the user wants cost, sum `response._hidden_params.get("response_cost") or 0.0` per `run_agent` and `span.set_attribute("gen_ai.usage.cost", total)` on the `invoke_agent` span before returning (stream: last chunk's `usage.cost`). Session total becomes correct; per-model breakdown stays unpriced. Never add token pricing tables. + +Flush (required for scripts, CLIs, Lambda/Cloud Run jobs, notebooks, workers; web servers only at shutdown). LiteLLM creates its span after the call via a background queue that drains at interpreter exit, after the provider shut down, so the last call is lost without this: + +```py +from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER + + +async def flush_tracing() -> None: + await asyncio.sleep(0) # LiteLLM queues its log event on the next loop tick + await GLOBAL_LOGGING_WORKER.flush() + provider.force_flush() +``` + +Await it in a `finally` inside the event loop (end of `main()`, end of each handler invocation, FastAPI lifespan shutdown); scripts then call `provider.shutdown()` after `asyncio.run(...)`. + +## Step 7: verify + +Run one real conversation: 2+ messages with the same id, one tool call, one streamed reply if the app streams, a failing tool if one exists, a second conversation with a different id, and a multi-agent run if the app has one. Check (Maple → Agent Sessions, filter by service; wait ~30 s): + +- [ ] Exactly one session per conversation, session id = the id passed; the second conversation is a separate session; no `trace:` sessions. +- [ ] Framework shows **LiteLLM** (not Unidentified). +- [ ] One turn per top-level `invoke_agent`; transcript shows user prompts, assistant replies and tool calls. +- [ ] Spans: `invoke_agent ` (app scope), `chat ` (scope `litellm`) and `execute_tool ` inside it, same trace. No `litellm_request` / `raw_gen_ai_request` spans (those mean v1). +- [ ] Every `chat` span has input and output tokens, including the streamed one; no call appears twice. +- [ ] Tool spans have real names, arguments, results and call ids. +- [ ] The failing tool is marked failed with its message; successful tools and model calls are not. +- [ ] Sub-agents appear as separate lanes, all in the caller's session. +- [ ] Cost: unpriced, or the per-run total if Step 6 cost was added. +- [ ] Last call of a script run present (flush worked). +- [ ] No attribute contains an API key, `Bearer ` or `sk-`. +- [ ] Proxy path: `chat` spans (service `litellm-proxy`) sit in the app's trace under `invoke_agent`, carrying the session id. + +Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` to the provider and confirm `gen_ai.conversation.id` on every `chat` span and the parent ids. + +## Do not + +- Do not use the v1 logger (`litellm.callbacks=["otel"]` alone) when async is possible: no session key, op `acompletion`, attributes lost under parent spans. +- Do not run LiteLLM 1.103 with OpenTelemetry >= 1.44 (v2 silently disabled). +- Do not register two loggers (v2 instance plus `"otel"` / `LITELLM_OTEL_V2` factory) in one process. +- Do not trace the same call at the proxy and in the app (client instrumentor): double counting. +- Do not rely on sync `litellm.completion()` under v2: no span. +- Do not put `gen_ai.conversation.id` / `maple_ai.session.id` on your own spans in the v2 setup. +- Do not use `event_only` content capture. +- Do not skip the flush in short-lived processes. +- Do not create a second `TracerProvider` when one exists. +- Do not promise cost without the Step 6 recipe; do not add token pricing code. +- Do not print or commit real keys beyond the repo's convention. diff --git a/skills/maple-agent-tracing-llamaindex/SKILL.md b/skills/maple-agent-tracing-llamaindex/SKILL.md new file mode 100644 index 0000000000..2405253de3 --- /dev/null +++ b/skills/maple-agent-tracing-llamaindex/SKILL.md @@ -0,0 +1,227 @@ +--- +name: maple-agent-tracing-llamaindex +description: "Trace LlamaIndex agents with Maple: OpenInference LlamaIndex instrumentor with GenAI output, a conversation id per chat, agent names and one span per model call, so each conversation is one Maple Agent Session with transcript, tools and tokens. Triggers on 'trace my llamaindex agent', 'add Maple to llamaindex', 'agent sessions for llamaindex', 'OpenTelemetry for llamaindex', 'llama-index observability'." +--- + +# Maple agent tracing: LlamaIndex (Python) + +Goal: every conversation = one Maple Agent Session. Each `agent.run()` / `workflow.run()` = one turn (one trace) with transcript, one model span per model call with tokens, `FunctionTool.acall` tool spans with results, failed tools marked failed, sub-agents in their own lanes. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/llamaindex + +Mechanism: `openinference-instrumentation-llama-index` (scope `openinference.instrumentation.llama_index`) with `TraceConfig(enable_genai_semconv=True)`, which dual-writes `gen_ai.*` on span attributes. Maple reads `gen_ai.conversation.id` as the session key for these spans. Maple files them under "Unidentified" framework (generic GenAI/OpenInference buckets); that is expected. + +## Step 0: detect + +1. Versions: `python -c "import llama_index.core as c; print(c.__version__)"` (or `pyproject.toml` / `uv.lock` / `requirements*.txt`). + - Need llama-index-core >= 0.14.19 (tested 0.14.25). Older: the instrumentor logs `DependencyConflict` and does nothing. Tell the user to upgrade. + - Python >= 3.10. +2. Existing tracing. Search for `LlamaIndexOpenTelemetry`, `llama_index.observability.otel`, `LlamaIndexInstrumentor`, `set_global_handler`, `TracerProvider(`, `set_tracer_provider`, `opentelemetry-instrument`, `logfire.configure`, `sentry_sdk.init`, `langfuse`, `phoenix.otel.register`. + - `LlamaIndexOpenTelemetry` (native `llama-index-observability-otel`) present → replace it with Step 2 (it puts content in span events, loses reply and usage on streamed calls, ignores `OTEL_EXPORTER_OTLP_*`). Never run both: every span doubles. Ask before removing if it also feeds another backend. + - `LlamaIndexInstrumentor` already used (Phoenix/Langfuse/Arize) → keep it, reuse its provider, add `config=TraceConfig(enable_genai_semconv=True)` and the Maple processor to that provider. + - Another `TracerProvider` exists → add the Maple processor chain to it; do NOT create a second provider. +3. Other instrumentors on the model client (`openinference-instrumentation-openai`, `-litellm`, OpenLLMetry `Traceloop.init`, `logfire.instrument_openai`) → duplicate model spans with their own usage. Remove for LlamaIndex models (ask if they serve other code). +4. Find: every `agent.run(` / `workflow.run(` / `AgentWorkflow(` call, where the chat/thread id lives in the request, every `FunctionAgent(`/`ReActAgent(`/`CodeActAgent(` construction, every tool that runs another agent, every `ctx.wait_for_event(` (HITL). +5. LLM class: `OpenAILike`/`OpenRouter` need `is_function_calling_model=True` or tools silently never run (no tool spans). Check it's set. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention (`.env`, settings module, secret manager) if it has one. Otherwise inline is acceptable: ingest keys are write-only. + +Env vars (`OTLPSpanExporter()` reads them and appends `/v1/traces`): + +```bash +OTEL_SERVICE_NAME=support-agent +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +If you pass `OTLPSpanExporter(endpoint=...)` in code, it must end in `/v1/traces` (no auto-append). + +## Step 2: install + init + +Add with the repo's package manager (uv/poetry/pip): + +```bash +pip install "llama-index-core>=0.14.25" "openinference-instrumentation-llama-index>=4.5.2" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Create `tracing.py` verbatim (adapt nothing except where noted): + +```py +# tracing.py +from llama_index.core.instrumentation.dispatcher import active_instrument_tags +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.llama_index import LlamaIndexInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +LLM_METHODS = (".chat", ".achat", ".stream_chat", ".astream_chat", + ".complete", ".acomplete", ".stream_complete", ".astream_complete") + + +class LlamaIndexForMaple(SpanProcessor): + """Sits in front of the exporter: one span per model call, agent names, no false HITL failures.""" + + def __init__(self, exporter_processor: SpanProcessor): + self._next = exporter_processor + self._open_llm_spans = {} + + def on_start(self, span, parent_context=None): + # instrument_tags({"gen_ai.agent.name": ...}) becomes an attribute, so sub-agents get lanes + agent_name = active_instrument_tags.get().get("gen_ai.agent.name") + if agent_name: + span.set_attribute("gen_ai.agent.name", agent_name) + if span.name.endswith(LLM_METHODS): + self._open_llm_spans[span.context.span_id] = span + self._next.on_start(span, parent_context) + + def on_end(self, span): + self._open_llm_spans.pop(span.context.span_id, None) + if span.name.endswith("._prepare_chat_with_tools"): + return # builds the request; never calls the model + if (span.status.description or "").startswith("WaitingForEvent"): + return # ctx.wait_for_event() suspends the tool and replays it later; not a failure + outer = self._open_llm_spans.get(span.parent.span_id) if span.parent else None + if outer is not None and outer.name == span.name: + outer.set_attributes(span.attributes) # the inner twin holds the messages and usage + return + self._next.on_end(span) + + def shutdown(self): + self._next.shutdown() + + def force_flush(self, timeout_millis=30000): + return self._next.force_flush(timeout_millis) + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))) +trace.set_tracer_provider(provider) + +LlamaIndexInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +- Import `tracing` first in the entry point (app module, `main.py`, worker). `instrument()` must run before the first `agent.run()`. +- Existing provider: skip `TracerProvider()`/`set_tracer_provider`; call `existing.add_span_processor(LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter())))` and pass `tracer_provider=existing`. +- The exporter MUST be added through `LlamaIndexForMaple`, never directly: without it every model call counts 2-3x in Maple (`_prepare_chat_with_tools` + nested same-name `astream_chat`/`achat` spans, all OpenInference kind LLM). +- No `service.name` default is acceptable: set `OTEL_SERVICE_NAME` (never `unknown_service`). + +## Step 3: session id (required) + +Nothing in LlamaIndex sets a conversation id; `Context` and `llamaindex.run_id` never reach Maple as one. Wrap every `agent.run(...)` / `workflow.run(...)` CALL in `using_session()`; add `instrument_tags` with the agent name in the same `with`: + +```py +from llama_index.core.agent.workflow import AgentStream +from llama_index.core.instrumentation.dispatcher import instrument_tags +from llama_index.core.workflow import Context +from openinference.instrumentation import using_session + +contexts: dict[str, Context] = {} + + +async def handle_message(conversation_id: str, text: str): + if conversation_id not in contexts: + contexts[conversation_id] = Context(agent) + ctx = contexts[conversation_id] + with using_session(conversation_id), instrument_tags({"gen_ai.agent.name": agent.name}): + handler = agent.run(user_msg=text, ctx=ctx) + async for event in handler.stream_events(): + if isinstance(event, AgentStream): + yield event.delta + await handler +``` + +- Only the `run()` call must be inside the `with` (tasks start there and inherit the contextvars). Consume the stream / `await handler` outside it; do not `yield` inside the `with` in async generators. +- The id must be stable per conversation and unique across conversations: the app's chat/thread id. No `uuid4()` per request, no constant, no module-level default. No id available → ask the user where the conversation boundary is; for a one-shot script, one uuid per conversation reused across its turns. +- One `Context` per conversation (a shared `Context` shares memory across users). +- HITL: keep `handler.ctx.send_event(HumanResponseEvent(...))` on the same handler; the resumed step stays in the same trace and session. +- Do not use `session.id`/`maple_ai.session.id` attributes of your own; `using_session` + `enable_genai_semconv` already writes `session.id` and `gen_ai.conversation.id`. + +## Step 4: content + +- On by default: `gen_ai.input.messages` (system + history + tool results), `gen_ai.output.messages` (incl. tool_call parts), tool results. Maple's transcript needs them. +- User wants content off → `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` (or `hide_input_text`/`hide_output_text` to keep structure). Env equivalents `OPENINFERENCE_HIDE_*`. Tell them the transcript will be empty. +- Pattern redaction (emails, cards): recommend an OTel Collector `redaction` processor. + +## Step 5: tools, errors, sub-agents + +1. Give every agent a `name=` and wrap each agent's `run()` in `instrument_tags({"gen_ai.agent.name": agent.name})` (the processor copies it onto every span). No tag = no agent facet, no lanes. +2. Multi-agent via custom `Workflow`: run the whole workflow inside `using_session(...)`; inside steps call sub-agents like this: + +```py +async def run_agent(agent: FunctionAgent, message: str) -> str: + with instrument_tags({"gen_ai.agent.name": agent.name}): + handler = agent.run(user_msg=message) + return str(await handler) +``` + + Parallel fan-out (`ctx.send_event` + `@step(num_workers=N)` + `ctx.collect_events`) stays in one trace with the session and agent tags. +3. Agents-as-tools: put the same `instrument_tags` block inside the tool function around the sub-agent's `run()`. +4. `AgentWorkflow` handoffs: one `AgentWorkflow.run` span, no per-agent spans, so no lanes. Tag the call with the root agent's name. If the user needs lanes, suggest running agents via workflow steps or tools (ask first; it changes app behavior). +5. Tool failures: a tool that raises → `FunctionTool.acall` status ERROR with the exception message (Maple counts it). Tools that `return "Error: ..."` look successful: convert to `raise` only where the user agrees. +6. HITL (`ctx.wait_for_event`): the first, suspended `FunctionTool.acall` ends ERROR `WaitingForEvent: ...`; the processor drops it. Nothing to add. +7. Known, unfixable here: `gen_ai.tool.call.arguments` = the tool's parameter schema (OpenInference GenAI mapping bug); no `gen_ai.tool.call.id` on tool spans. Real args are in the transcript's tool_call parts. + +## Step 6: tokens, cost, streaming + +- Tokens come from the provider response on the model span. `FunctionAgent` streams by default; OpenAI-API streams only include usage when asked. For `OpenAI`/`OpenAILike` models not on OpenRouter add: + +```py +llm = OpenAI(model="gpt-4o-mini", additional_kwargs={"stream_options": {"include_usage": True}}) +``` + + (LlamaIndex strips it from non-streaming requests.) OpenRouter sends usage without it. +- Streamed model spans end when the stream is handed back (~1 ms), so their duration is not model latency. If the app never streams tokens to users, `FunctionAgent(..., streaming=False)` gives real model-span durations. Ask before changing it. +- Cost: never recorded. Sessions show "unpriced". Do not add pricing code. OpenRouter users can add OpenRouter Broadcast for cost. + +## Step 7: flush + +- `BatchSpanProcessor` exports every 5 s; the provider flushes at normal interpreter exit. +- Scripts/CLIs/one-shot jobs: `provider.shutdown()` in a `finally` at the end. +- Serverless handlers, Celery/RQ tasks, notebooks: `provider.force_flush()` in a `finally` after each run (`from tracing import provider`). +- FastAPI/long-running servers: nothing extra; optionally `provider.shutdown()` in the lifespan shutdown. + +## Step 8: verify + +Run one real conversation: 2+ messages with the same id, at least one tool call, a failing tool if one exists, and a sub-agent run if the app delegates. Then check (Maple → Agent Sessions, filter by service name; wait ~30-60 s): + +- [ ] Exactly one session per conversation, id = the id you passed. A second conversation is a second session. Not `trace:` sessions. +- [ ] One turn per `run()`; each trace roots at `FunctionAgent.run` / `AgentWorkflow.run` / `.run`. +- [ ] Framework shows "Unidentified" (expected for OpenInference LlamaIndex spans). +- [ ] Transcript shows user messages, assistant replies and tool calls. +- [ ] LLM call count ≈ real number of model calls (one `.astream_chat`/`.achat` span per call; no `_prepare_chat_with_tools` spans exported). +- [ ] Every model span has a model and input/output tokens, including streamed calls. +- [ ] Tool spans `FunctionTool.acall` carry `gen_ai.tool.name` (real tool name) and a result; model and tool spans sit under the turn's agent span in the same trace. +- [ ] A failing tool is failed with its message; successful and HITL-approved tools are not. +- [ ] Sub-agents appear as separate lanes with their names, under the caller's session. +- [ ] Cost shows "unpriced". +- [ ] No attribute contains an API key or `Bearer ` token. + +Local check without Maple: temporarily pass `SimpleSpanProcessor(ConsoleSpanExporter())` into `LlamaIndexForMaple` and confirm `gen_ai.conversation.id` is identical on every span of a conversation and `gen_ai.agent.name` is set. + +## Do not + +- Do not use or keep `LlamaIndexOpenTelemetry` (`llama-index-observability-otel`) for Maple, and never alongside the OpenInference instrumentor. +- Do not add the exporter to the provider directly; always through `LlamaIndexForMaple`. +- Do not forget `enable_genai_semconv=True` (tokens in the list, empty session page). +- Do not treat `Context` or `llamaindex.run_id` as a session id; use `using_session` around `run()`. +- Do not generate a new conversation id per request, and do not share one `Context` across conversations. +- Do not wrap `yield` statements in `using_session`/`instrument_tags` blocks. +- Do not add provider instrumentors (OpenAI/LiteLLM OpenInference, OpenLLMetry) on top: duplicate model spans and tokens. +- Do not add `maple_ai.session.id` (re-vendors spans) or custom pricing attributes. +- Do not return error strings from failing tools where a raise is acceptable. +- Do not print or commit real keys beyond the repo's convention. diff --git a/skills/maple-agent-tracing-mastra/SKILL.md b/skills/maple-agent-tracing-mastra/SKILL.md new file mode 100644 index 0000000000..c2c4517dd7 --- /dev/null +++ b/skills/maple-agent-tracing-mastra/SKILL.md @@ -0,0 +1,181 @@ +--- +name: maple-agent-tracing-mastra +description: "Trace Mastra agents and workflows with Maple: export Mastra's built-in GenAI spans through @mastra/otel-exporter so each conversation is one Maple Agent Session with transcript, tool calls, sub-agent lanes and tokens. Triggers on 'trace my mastra agent', 'add Maple to mastra', 'agent sessions for mastra', 'OpenTelemetry for mastra'." +--- + +# Maple agent tracing: Mastra + +Goal: every conversation = one Maple Agent Session. Each `agent.generate()` / `agent.stream()` / workflow `run.start()` = one turn (one trace) with transcript, `chat ` spans with tokens, `execute_tool ` spans with args/results, failed tools marked failed, sub-agents in their own lanes. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/mastra + +Mechanism: Mastra's own tracing (`@mastra/observability`) converted to OTel GenAI semconv v1.38 by `@mastra/otel-exporter`, which runs its own BatchSpanProcessor. No OTel SDK or instrumentation package needed. Maple detects the vendor from resource `telemetry.sdk.name=@mastra/otel-exporter` and reads `gen_ai.conversation.id` as the session key. The exporter writes that key from span `metadata.threadId`, which Mastra sets from `memory.thread`. No thread = no session key. + +## Step 0: detect + +1. Versions: read `package.json` / lockfile for `@mastra/core`, `@mastra/observability`, `@mastra/otel-exporter`, `@mastra/memory`. + - Need `@mastra/core` 1.x (written against 1.71) and Node >= 22.13. Mastra 0.x uses a different telemetry API: tell the user to upgrade; do not work around it. + - `@mastra/core`, `@mastra/observability`, `@mastra/otel-exporter` must be from the same release train. Update all three together (`@latest`) if you add or bump one. +2. Find the `new Mastra({...})` instance (usually `src/mastra/index.ts`) and its current `observability` value. + - Already `new Observability({ configs: {...} })` → add the Maple exporter to the existing config's `exporters`; keep existing exporters (MastraStorageExporter, MastraPlatformExporter, Langfuse...). + - `observability: { default: { enabled: true } }` or any plain object → replace with `new Observability(...)` (a plain object silently installs a no-op). Keep `MastraStorageExporter` if the project uses Mastra Studio traces. + - `@mastra/otel-bridge` already configured with an OTel SDK → do NOT add OtelExporter (double export). Point the existing OTel exporter at Maple instead. +3. Existing OTel NodeSDK / TracerProvider elsewhere in the app: leave it alone. OtelExporter does not use the global provider; Mastra spans become their own traces. +4. Find: every `generate(` / `stream(` / `network(` call and where the chat/thread id lives in the request; every `new Agent(` (need `id` + `name`); every Agent with an `agents:` property (supervisor); every `createWorkflow` / `run.start(` / `createStep` that calls an agent; tools created with `createTool`. +5. Other tracers of the same model calls (OpenLLMetry `Traceloop.init`, OpenInference AI SDK instrumentation, Vercel AI SDK `experimental_telemetry` on the underlying model) → they double-trace. Keep Mastra's; ask before removing the others if they serve something else. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization: Bearer ` (passed as a headers object in code; no `%20` encoding). +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention (`.env`, config module, secret manager) if it has one, e.g. `process.env.MAPLE_INGEST_KEY`. Otherwise inline is acceptable: ingest keys are write-only. +- OtelExporter's `custom` provider reads NO env vars (`OTEL_EXPORTER_OTLP_*` are ignored). Endpoint, protocol and headers must be passed in code. + +## Step 2: install + init + +```bash +npm install @mastra/observability@latest @mastra/otel-exporter@latest @opentelemetry/exporter-trace-otlp-proto +``` + +Use the repo's package manager (pnpm/yarn/bun). Bump `@mastra/core` to latest in the same command if it is older than the other two. + +In the file that creates the Mastra instance: + +```ts +import { Mastra } from "@mastra/core/mastra" +import { SpanType } from "@mastra/core/observability" +import { Observability } from "@mastra/observability" +import { OtelExporter } from "@mastra/otel-exporter" + +export const mapleExporter = new OtelExporter({ + provider: { + custom: { + endpoint: "https://ingest.maple.dev", + protocol: "http/protobuf", + headers: { Authorization: `Bearer ${process.env.MAPLE_INGEST_KEY}` }, + }, + }, + resourceAttributes: { "deployment.environment.name": "production" }, +}) + +export const mastra = new Mastra({ + // ...existing agents, workflows, storage + observability: new Observability({ + configs: { + maple: { + serviceName: "support-agent", + exporters: [mapleExporter], + excludeSpanTypes: [SpanType.MODEL_CHUNK], + spanOutputProcessors: [conversationIdFromRoot], // only if Step 5 applies + }, + }, + }), +}) +``` + +- `protocol: "http/protobuf"` is mandatory: the custom provider defaults to `http/json`, and a missing protocol package disables tracing with one console error. +- Endpoint is the base URL; the exporter appends `/v1/traces` and `/v1/logs` (and strips them if present). +- `serviceName` is required; use the project/service name, never leave `mastra-service`. Set `deployment.environment.name` from the app's env (e.g. `process.env.NODE_ENV`). +- Logs: the exporter also sends Mastra log records (warn+) to `/v1/logs` by default. Keep that unless the user wants traces only: `signals: { logs: false }` on OtelExporter. +- Do not set `includeInternalSpans: true` (5x spans, ~25x bytes; Mastra's internal agent-loop workflows). +- Agents and workflows must be registered on this `Mastra` instance and called through it (`mastra.getAgent(...)`, `mastra.getWorkflow(...)`), or called with a `tracingContext` from a traced parent. An `Agent` used standalone has no observability. +- Debugging delivery: `logLevel: "debug"` on OtelExporter prints `Export completed: N spans sent successfully` / `Export FAILED: ...`. Remove it afterwards. + +## Step 3: session id (conversation id) + +Agents: pass the app's conversation id as the memory thread on EVERY call of a conversation: + +```ts +const result = await agent.generate(text, { memory: { thread: chatId, resource: userId } }) + +const stream = await agent.stream(text, { memory: { thread: chatId, resource: userId } }) +for await (const chunk of stream.textStream) res.write(chunk) // consume fully; spans end with the stream +``` + +- `thread` = stable per conversation, different between conversations. Never a constant, never a fresh UUID per request. +- `resource` = the user/tenant id. +- Works without a `Memory` on the agent (Mastra only logs a warning), but prefer the project's existing memory setup. +- Server routes (`@mastra/server`, `mastra dev`): the client's `threadId` or middleware's `mastra__threadId` request-context key is used; verify the frontend sends a stable thread id. + +Workflows, and agents that must not use memory: set the thread in root tracing metadata; Mastra copies root metadata to every span: + +```ts +const run = await mastra.getWorkflow("briefingWorkflow").createRun() +await run.start({ inputData, tracingOptions: { metadata: { threadId: conversationId } } }) +``` + +`agent.generate(text, { tracingOptions: { metadata: { threadId: conversationId } } })` works the same for memory-less agents. + +HITL resumes (`approveToolCallGenerate` / `declineToolCallGenerate` / `approveToolCall` / `declineToolCall`): pass the same `memory` again. + +## Step 4: content + +- On by default: `gen_ai.input.messages` / `gen_ai.output.messages` on `chat` spans, `gen_ai.system_instructions` on `invoke_agent`, tool args/results on `execute_tool`. Change nothing to keep it. +- Opt-out per request: `tracingOptions: { hideInput: true, hideOutput: true }` (whole trace). +- `SensitiveDataFilter` is auto-applied (redacts values under keys like password/token/apiKey/authorization/secret). Do not disable it (`sensitiveDataFilter: false`) unless the user asks. +- Serialization caps: 128 KiB/string, 50 items/array, 50 keys/object, depth 8. If agents keep > ~40 messages of history (`lastMessages` > 40 or custom history), add `serializationOptions: { maxArrayLength: 200 }` to the config, or the newest messages are cut from the transcript. + +## Step 5: tools, errors, sub-agents + +- Tools: `createTool({ id, description, inputSchema, execute })`. Span name `execute_tool `, `gen_ai.tool.name` = id. Give tools real ids. +- Failures must throw (`throw new Error("...")`). Thrown → span ERROR + `error.type=TOOL_EXECUTION_FAILED` + message. Returning `{ error }` = counted as success in Maple. If a tool swallows errors into a return value and the user wants failures visible, rethrow. +- Every `new Agent({ id, name, ... })` needs a distinct `name`: it is `gen_ai.agent.name`, which Maple uses for lanes. +- Supervisor agents (`agents: {...}` on an Agent), `agent.network()`, or workflow steps that call agents which have their own `memory`: sub-agents get their own thread ids, so one trace carries several `gen_ai.conversation.id` values and Maple may pick the wrong one or split the turn. Add this processor and list it in `spanOutputProcessors`: + +```ts +import type { SpanOutputProcessor } from "@mastra/core/observability" + +export const conversationIdFromRoot: SpanOutputProcessor = { + name: "conversation-id-from-root", + process(span) { + let root = span + while (root?.parent) root = root.parent + const threadId = root?.metadata?.threadId + if (span && threadId) span.metadata = { ...span.metadata, threadId } + return span + }, + async shutdown() {}, +} +``` + +- Workflow steps that call an agent: pass `tracingContext` from the step's `execute` args into `agent.generate(prompt, { tracingContext })`, or the agent starts a separate trace. + +## Step 6: flush + +Batch interval is 5 s. Anything that can exit sooner must flush. + +- Scripts, CLIs, tests, one-shot jobs: `await mastra.shutdown()` in a `finally` before exit (flushes all exporters). Never `process.exit()` before it. +- Serverless / per-request handlers (Next.js route handlers, Vercel, Lambda, Workers): `await mastra.observability.flush()` in a `finally` at the end of each request; keep the instance. For streamed responses, flush after the stream completes (`after()`, `ctx.waitUntil()`, stream `onFinish`), not when the handler returns. +- Long-running servers: nothing needed; optionally call `mastra.shutdown()` on SIGTERM. + +## Step 7: verify + +Run one conversation: 2+ user messages with the same thread id (one streamed), at least one tool call, and a second conversation with a different thread id. If the project has a supervisor or workflow, run it once. Flush. Then check (Maple UI Agent Sessions, or debug log + Maple MCP `list_agent_sessions` filtered by service): + +- Startup log has no `[OtelExporter]` errors and no no-op observability warning; with `logLevel: "debug"`, `Export completed` lines appear. +- Framework shows **Mastra** (not Unidentified). +- Exactly one session per conversation, id = the thread id; the second conversation is a separate session; no `trace:` sessions for chat turns. +- One turn per `generate()` / `stream()` / workflow run, labeled with the user message. +- Transcript shows user prompts, assistant replies and tool calls. +- `chat ` spans have model, provider and input/output tokens, including the streamed turn. +- `execute_tool ` spans have the real tool name, arguments and result; a throwing tool is marked failed with its message; successful tools are not. +- Supervisor/workflow: one session, one turn per run, one lane per sub-agent `name`, all spans in one trace. +- Cost shows as unpriced (expected: Mastra emits no cost attribute). + +## Do not + +- Do not pass a plain object as `observability`; use `new Observability(...)`. +- Do not rely on `OTEL_EXPORTER_OTLP_ENDPOINT` / `OTEL_EXPORTER_OTLP_HEADERS` / `OTEL_EXPORTER_OTLP_PROTOCOL`; the custom provider ignores them. +- Do not omit `protocol: "http/protobuf"` (defaults to `http/json`, needs a different package). +- Do not add an OTel NodeSDK, `@vercel/otel`, OpenLLMetry or OpenInference just for Mastra; they are unnecessary and double-trace model calls. +- Do not use both OtelExporter and `@mastra/otel-bridge` to reach Maple. +- Do not call `generate()` / `stream()` without a stable `memory.thread` (or `tracingOptions.metadata.threadId`) in a multi-turn chat. +- Do not enable `includeInternalSpans`. +- Do not stamp `maple_ai.session.id` on Mastra spans; it re-vendors them away from Mastra decoding. Use the thread id. +- Do not mix `@mastra/observability` / `@mastra/otel-exporter` versions from different releases (spans arrive with no model calls or tokens). +- Do not return error objects from tools you want counted as failures; throw. +- Do not exit a script without `await mastra.shutdown()`. +- Do not promise cost or time-to-first-token in Maple: Mastra emits neither under keys Maple reads. Reasoning tokens are in the output total but not broken out. diff --git a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md new file mode 100644 index 0000000000..12a196757d --- /dev/null +++ b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md @@ -0,0 +1,244 @@ +--- +name: maple-agent-tracing-microsoft-agent-framework +description: "Trace Microsoft Agent Framework and Semantic Kernel agents (Python and .NET) with Maple: native GenAI spans exported over OTLP/HTTP, plus a span processor that adds the conversation id so each chat is one Agent Session with transcript, tools and tokens. Triggers on 'trace my agent framework agent', 'add Maple to Microsoft Agent Framework', 'add Maple to Semantic Kernel', 'agent sessions for agent-framework', 'OpenTelemetry for Semantic Kernel'." +--- + +# Maple agent tracing: Microsoft Agent Framework and Semantic Kernel + +Goal: one user conversation = one Maple Agent Session, with a transcript, every model call, every tool call (arguments, results, failures) and tokens. Human guide with the reasoning: https://maple.dev/docs/agent-tracing/microsoft-agent-framework + +The framework emits the spans itself. You add: an OTLP/HTTP exporter, content capture, a `gen_ai.conversation.id` span processor (the framework never sets one for local-history sessions), and a flush. + +## Step 0: detect + +- Python MAF: `agent-framework`, `agent-framework-core` in `pyproject.toml` / `requirements*.txt` / `uv.lock`. Check the installed version (`python -c "import agent_framework; print(agent_framework.__version__)"`). Target ≥ 1.19.0; upgrade if older (1.13 lacks the `otlp_*` arguments used below). +- .NET MAF: `Microsoft.Agents.AI` in `*.csproj`. Target ≥ 1.22.0. +- Semantic Kernel: `semantic-kernel` (Python, target ≥ 1.44.1) or `Microsoft.SemanticKernel` (.NET). Use the Semantic Kernel section. +- Existing OpenTelemetry: search for `TracerProvider(`, `set_tracer_provider`, `configure_azure_monitor`, `logfire.configure`, `configure_otel_providers`, `AddOpenTelemetry(`, `Sdk.CreateTracerProviderBuilder`. If a provider exists, add Maple's exporter and the conversation processor to it; do not create a second provider and do not call `configure_otel_providers()`. +- Find every place a user message is handled (HTTP route, queue consumer, CLI loop) and what identifies the conversation there (chat id, thread id, `AgentSession`). You need it in Step 3. + +## Step 1: key and region + +- US endpoint `https://ingest.maple.dev`, EU `https://ingest.eu.maple.dev`. Header `Authorization=Bearer `. Protocol `http/protobuf`. +- Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's existing secret/env convention (`.env`, settings class, user-secrets). If there is none, inlining the ingest key is acceptable: ingest keys are write-only. + +## Step 2: install and init + +### Python MAF + +```bash +pip install "agent-framework-core>=1.19.0" "agent-framework-openai>=1.14.4" opentelemetry-exporter-otlp-proto-http +``` + +(Keep the `agent-framework` meta-package if the project already uses it; still add `opentelemetry-exporter-otlp-proto-http`, MAF installs no exporter.) + +Create `maple_tracing.py` next to the app entry point: + +```py +from contextlib import contextmanager +from contextvars import ContextVar + +from opentelemetry.sdk.trace import SpanProcessor + +_conversation_id: ContextVar[str | None] = ContextVar("conversation_id", default=None) + + +class ConversationIdProcessor(SpanProcessor): + """Puts gen_ai.conversation.id on every span started inside `conversation()`.""" + + def on_start(self, span, parent_context=None): + if (conversation_id := _conversation_id.get()) is not None: + span.set_attribute("gen_ai.conversation.id", conversation_id) + + +@contextmanager +def conversation(conversation_id: str): + token = _conversation_id.set(conversation_id) + try: + yield + finally: + _conversation_id.reset(token) +``` + +At startup, once, before agents are created: + +```py +import os + +from agent_framework.observability import configure_otel_providers +from opentelemetry import trace + +from maple_tracing import ConversationIdProcessor + +configure_otel_providers( + service_name="", + resource_attributes={"deployment.environment.name": ""}, + otlp_endpoint="https://ingest.maple.dev", + otlp_protocol="http/protobuf", + otlp_headers={"Authorization": f"Bearer {os.environ['MAPLE_INGEST_KEY']}"}, + enable_sensitive_data=True, + enable_message_events=False, +) +trace.get_tracer_provider().add_span_processor(ConversationIdProcessor()) +``` + +- `otlp_protocol` is mandatory: MAF defaults to gRPC (spec says http/protobuf) and fails silently or with an ImportError. +- Existing provider instead: add `BatchSpanProcessor(OTLPSpanExporter(endpoint="https://ingest.maple.dev/v1/traces", headers={...}))` and `ConversationIdProcessor()` to it, then `from agent_framework.observability import enable_instrumentation; enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)`. +- For OpenRouter or any Chat Completions endpoint use `OpenAIChatCompletionClient(model=..., api_key=..., base_url=...)`, not `OpenAIChatClient` (Responses API; OpenRouter rejects its `previous_response_id` on turn 2). + +### .NET MAF + +```bash +dotnet add package Microsoft.Agents.AI --version 1.22.0 +dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol --version 1.19.1 +``` + +```csharp +using var tracerProvider = Sdk.CreateTracerProviderBuilder() + .ConfigureResource(r => r.AddService("")) + .AddSource("*Microsoft.Agents.AI*") + .AddSource("*Microsoft.Extensions.AI") + .AddProcessor(new ConversationIdProcessor()) + .AddOtlpExporter(o => + { + o.Endpoint = new Uri("https://ingest.maple.dev/v1/traces"); + o.Protocol = OtlpExportProtocol.HttpProtobuf; + o.Headers = "Authorization=Bearer "; + }) + .Build(); + +AIAgent agent = chatClient + .AsAIAgent(instructions: "...", name: "", tools: [...]) + .AsBuilder() + .UseOpenTelemetry(configure: a => a.EnableSensitiveData = true) + .Build(); + +sealed class ConversationIdProcessor : BaseProcessor +{ + public static readonly AsyncLocal Current = new(); + + public override void OnStart(Activity activity) + { + if (Current.Value is { } id) activity.SetTag("gen_ai.conversation.id", id); + } +} +``` + +- Default source names carry an `Experimental.` prefix; `AddSource("Microsoft.Agents.AI.*")` matches nothing. Keep the leading `*`. +- `UseOpenTelemetry()` on the agent (1.22) also instruments its chat client. Do not add a second `UseOpenTelemetry()` on the `IChatClient`. +- Workflows: `.WithOpenTelemetry()` on the `WorkflowBuilder` (source `Microsoft.Agents.AI.Workflows`, matched by the wildcard). +- In a hosted app, put the same sources, processor and exporter in `builder.Services.AddOpenTelemetry().WithTracing(...)`. + +## Step 3: conversation id + +Wrap every request/turn so all spans of that turn start inside `conversation()`: + +```py +async def handle_message(session: AgentSession, text: str) -> str: + with conversation(session.session_id): + response = await agent.run(text, session=session) + return response.text +``` + +- Id: the app's chat/thread id, or `AgentSession.session_id` (create sessions with `agent.create_session(session_id=chat_id)` when the app has an id). Must be stable across all turns of one conversation and differ between conversations. +- Streaming: the whole `async for update in agent.run(..., stream=True)` loop goes inside the `with`. +- Approval resumes (`request.to_function_approval_response(...)` passed back to `agent.run`) and workflow runs go inside the same `with`. +- Build workflows (`WorkflowBuilder(...).build()`, `SequentialBuilder`, `ConcurrentBuilder`, ...) inside the `with`; a build outside it emits a stray `workflow.build` trace with no id. +- .NET: `ConversationIdProcessor.Current.Value = chatId;` in the request handler before `RunAsync`/`RunStreamingAsync`. + +## Step 4: content + +- Python: `enable_sensitive_data=True` (or `ENABLE_SENSITIVE_DATA=true`). .NET: `EnableSensitiveData = true` (or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=true`). +- If the process sets `OTEL_SEMCONV_STABILITY_OPT_IN`, it must include `gen_ai_latest_experimental` (e.g. `http,gen_ai_latest_experimental`); otherwise MAF moves content off the spans into log events Maple doesn't read. +- `enable_message_events=False`: otherwise every message is also exported as OTLP log records (duplicate payload). +- If the user wants no prompt content stored, leave sensitive data off and tell them the transcript and tool arguments/results will be empty. + +## Step 5: tools, errors, sub-agents + +- Give every `Agent` a distinct `name` (Maple lanes key on `gen_ai.agent.name`; unnamed agents get a UUID). +- Tool failures: raising from the tool function is enough; MAF sets ERROR + `error.type` on `execute_tool`. Do not catch and return an error string from the tool body (that hides the failure). +- Sub-agents: prefer `worker.as_tool()` in the orchestrator's `tools=[...]`, or MAF workflows/orchestrations. Do not also instrument the provider SDK (e.g. OpenInference/OpenLLMetry OpenAI instrumentors): that double-counts every call. + +## Step 6: flush + +Scripts, CLIs, notebooks, tests, serverless: + +```py +from opentelemetry import _logs, metrics, trace + +try: + asyncio.run(main()) +finally: + for provider in (trace.get_tracer_provider(), metrics.get_meter_provider(), _logs.get_logger_provider()): + provider.shutdown() +``` + +Serverless handler kept warm: `trace.get_tracer_provider().force_flush()` before returning. Long-running servers: nothing per request. .NET: `using var tracerProvider` disposes at exit; hosted apps flush on graceful shutdown. + +## Semantic Kernel + +Python (≥ 1.44.1; `uv` needs `--prerelease=allow` for its `azure-ai-agents` dependency): + +```bash +pip install "semantic-kernel>=1.44.1" opentelemetry-sdk opentelemetry-exporter-otlp-proto-http +``` + +In a module imported before anything that imports `semantic_kernel`: + +```py +import os + +os.environ["SEMANTICKERNEL_EXPERIMENTAL_GENAI_ENABLE_OTEL_DIAGNOSTICS"] = "true" +os.environ["SEMANTICKERNEL_EXPERIMENTAL_GENAI_ENABLE_OTEL_DIAGNOSTICS_SENSITIVE"] = "true" + +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +from maple_tracing import ConversationIdProcessor + +provider = TracerProvider(resource=Resource.create({"service.name": ""})) +provider.add_span_processor(ConversationIdProcessor()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter( + endpoint="https://ingest.maple.dev/v1/traces", + headers={"Authorization": f"Bearer {os.environ['MAPLE_INGEST_KEY']}"}, +))) +trace.set_tracer_provider(provider) +``` + +- The env vars are read once at import; set later, SK emits no GenAI spans and no warning. +- Wrap each turn in `with conversation(thread.id):` (`ChatHistoryAgentThread().id` is stable; SK never exports it). +- Call agents with positional messages: `await agent.get_response(text, thread=thread)`, `agent.invoke_stream(text, thread=thread)`. The keyword form `messages=` records empty input. +- Transcript comes only from `invoke_agent` spans (`ChatCompletionAgent`). Model-call content is Python logging, which Maple doesn't read; code that uses the kernel without an agent has no transcript. Tell the user. +- Orchestrations on `InProcessRuntime`: wrap in `with conversation(id), tracer.start_as_current_span(""):` and call `runtime.start()` inside it, or the run splits into many traces. +- Flush with `provider.shutdown()` in `finally`. +- .NET SK: same env vars (or `AppContext` switches `Microsoft.SemanticKernel.Experimental.GenAI.EnableOTelDiagnostics[Sensitive]`), `AddSource("Microsoft.SemanticKernel*")`, same `ConversationIdProcessor`. Shows as "Unidentified" framework in Maple. + +## Step 7: verify + +Run one real conversation (2-3 turns, one tool call; a second conversation if cheap). With a real key, open Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`) after ~1 minute. With `MAPLE_TEST`, check the spans locally instead (add `ConsoleSpanExporter` temporarily, or point the endpoint at a local collector). Check: + +- Spans arrive and the process exits cleanly (flush ran); `service.name` is yours, not `agent_framework` / `unknown_service`. +- Every span of every turn in one conversation has the same `gen_ai.conversation.id`; a second conversation has a different one. In Maple: one session per conversation, not `trace:` sessions. +- Each turn: `invoke_agent ` root with `chat ` and `execute_tool ` descendants in the same trace (streamed turn included). +- `chat` spans: `gen_ai.request.model`, `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` (also on the streamed turn), `gen_ai.input.messages` / `gen_ai.output.messages` as JSON arrays of `{role, parts}` (Python MAF; SK: on `invoke_agent` only). +- `execute_tool`: real tool name, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`; a raising tool has ERROR status + `error.type`, successful ones don't. +- Sub-agents have distinct `gen_ai.agent.name`s. +- No attribute contains the model API key or `Bearer `. +- Cost: none is emitted; Maple shows sessions as unpriced. Framework label: "Microsoft Agent Framework" / "Semantic Kernel" (Python); .NET shows "Unidentified". + +## Do not + +- Do not leave the MAF protocol at its gRPC default. +- Do not use `gen_ai.agent.id`, `workflow.id`, `service.instance.id` or a module-level constant as the conversation id. +- Do not pass `conversation_id` as an agent/run option to label traces: it disables in-memory history and becomes `previous_response_id` on the Responses client. +- Do not call `configure_otel_providers()` when the app already has a TracerProvider, and do not create a second provider. +- Do not set `OTEL_SEMCONV_STABILITY_OPT_IN` without `gen_ai_latest_experimental`. +- Do not add a provider-SDK instrumentor (OpenAI, httpx-based GenAI instrumentors) on top; MAF already traces model calls. +- Do not import `semantic_kernel` before setting its diagnostics env vars. +- Do not pass a bare string as `Message("user", text)` contents; use `[text]`. +- Do not filter out non-AI spans (HTTP, workflow) in a collector; dropping parents orphans the AI spans. diff --git a/skills/maple-agent-tracing-openai-agents/SKILL.md b/skills/maple-agent-tracing-openai-agents/SKILL.md new file mode 100644 index 0000000000..04d6d9c1f1 --- /dev/null +++ b/skills/maple-agent-tracing-openai-agents/SKILL.md @@ -0,0 +1,233 @@ +--- +name: maple-agent-tracing-openai-agents +description: "Trace OpenAI Agents SDK agents with Maple: bridges the SDK's tracing to OpenTelemetry with OpenInference (GenAI attributes on), wraps each run in using_session so each chat is one Maple Agent Session with transcript, tool calls, handoffs, sub-agent lanes and tokens. Triggers on 'trace my openai agents sdk agent', 'add Maple to openai-agents', 'agent sessions for OpenAI Agents SDK', 'OpenTelemetry for openai agents'." +--- + +# Maple agent tracing for the OpenAI Agents SDK + +Goal: every conversation with the app shows up in Maple **Agent Sessions** as exactly one session, one turn per `Runner.run`, with transcript, model calls, tool calls (failures marked), agent lanes for sub-agents and handoffs, and tokens (streamed turns included). + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/openai-agents + +Mechanism: the SDK has its own tracing pipeline (not OpenTelemetry) whose default processor uploads to the OpenAI dashboard. `openinference-instrumentation-openai-agents` registers a processor on that pipeline that converts each SDK span into an OTel span; an OTel SDK `TracerProvider` + OTLP/HTTP exporter sends them to Maple. Maple fingerprints the scope `openinference.instrumentation.openai_agents` as "OpenAI Agents SDK" and reads `session.id` for the session. + +Python is the primary path. TypeScript: see Step 2e. + +## Step 0: Detect versions and existing setup + +1. Find the project file (`pyproject.toml`, `requirements*.txt`, `uv.lock`, `poetry.lock`) and the installed `openai-agents` version. Target `openai-agents>=0.22` (verified 0.22.3) and `openinference-instrumentation-openai-agents>=2.5` (verified 2.5.0; 1.x does not record agent names). Python 3.10 to 3.14. +2. Grep for existing tracing: `TracerProvider(`, `set_tracer_provider`, `OpenAIAgentsInstrumentor`, `set_trace_processors`, `add_trace_processor`, `set_tracing_disabled`, `OPENAI_AGENTS_DISABLE_TRACING`, `tracing_disabled=`, `logfire.configure`, `instrument_openai_agents`, `OpenAIInstrumentor`, `phoenix.otel.register`, `langfuse`, `openlit.init`, `Traceloop.init`. + - Existing `TracerProvider` of the app's own: reuse it. Add a `BatchSpanProcessor(OTLPSpanExporter(...))` for Maple to it. Do not create a second provider. + - Existing `OpenAIAgentsInstrumentor().instrument(...)`: edit that call; never call `instrument()` twice. + - Tracing disabled anywhere (`set_tracing_disabled(True)`, `OPENAI_AGENTS_DISABLE_TRACING=1`, `RunConfig(tracing_disabled=True)`): remove it. It kills the pipeline the bridge reads; zero spans. + - Any other instrumentation on the same calls (OpenInference `OpenAIInstrumentor`, Logfire `instrument_openai_agents`/`instrument_openai`, Langfuse/Traceloop/OpenLIT Agents or OpenAI instrumentors, `opentelemetry-instrumentation-genai-openai-agents`): remove it or every model call is recorded twice. +3. Find every `Runner.run(`, `Runner.run_sync(`, `Runner.run_streamed(` and where the conversation/thread id lives in the request. Note `RunConfig(group_id=...)` and `SQLiteSession(...)`/other `Session` ids: reuse that id in Step 3. +4. Find the model setup: default OpenAI (Responses API, `OPENAI_API_KEY`) vs a custom base URL (`AsyncOpenAI(base_url=...)`, `set_default_openai_client`, `OpenAIChatCompletionsModel`, `set_default_openai_api("chat_completions")`, LiteLLM extension). + +## Step 1: Key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. Protocol `http/protobuf`. +- Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from **Settings → Ingestion**. +- Private `maple_sk_` keys never go in browser code. This runs server-side; a `maple_pk_` ingest key is write-only. +- Follow the repo's existing secret/env convention (`.env`, settings module, secret manager). If there is none, inline is acceptable because ingest keys are write-only. + +## Step 2: Install and initialize + +### 2a. Packages + +Add with the project's package manager: + +``` +openai-agents>=0.22 +openinference-instrumentation-openai-agents>=2.5 +opentelemetry-sdk>=1.45 +opentelemetry-exporter-otlp-proto-http>=1.45 +``` + +### 2b. Environment + +```bash +OTEL_SERVICE_NAME= +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name= +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev # EU: https://ingest.eu.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +`OTLPSpanExporter()` appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. `OTLPSpanExporter(endpoint=...)` in code does NOT append; give the full `.../v1/traces` URL there. + +### 2c. Tracing module + +Create `tracing.py` (or add to the app's existing observability module), imported at the top of the entry point before any `Runner.run`: + +```py +from agents import set_trace_processors +from agents.tracing import TracingProcessor +from agents.tracing.span_data import FunctionSpanData, GenerationSpanData, HandoffSpanData +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.openai_agents import OpenAIAgentsInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +def _chat_message(response: dict) -> dict: + """The assistant message inside a Responses-shaped dict, as a Chat Completions message.""" + text, calls = "", [] + for item in response.get("output") or []: + if item.get("type") == "message": + text += "".join(c.get("text", "") for c in item.get("content") or [] if c.get("type") == "output_text") + elif item.get("type") == "function_call": + calls.append({"id": item["call_id"], "type": "function", + "function": {"name": item["name"], "arguments": item["arguments"]}}) + return {"role": "assistant", "content": text or None, "tool_calls": calls or None} + + +class MapleSpanFixes(TracingProcessor): + """Fills three gaps in what OpenInference exports. Must run before the OpenInference processor.""" + + def on_span_end(self, span): + data = span.span_data + current = trace.get_current_span() # the matching OpenTelemetry span, still open here + if isinstance(data, FunctionSpanData) and data.input and getattr(current, "name", None) == data.name: + # Real arguments; OpenInference would copy the tool's JSON schema. + current.set_attribute("gen_ai.tool.call.arguments", data.input) + elif isinstance(data, HandoffSpanData) and data.to_agent: + # Handoff spans carry no tool name. + current.set_attribute("gen_ai.tool.name", f"transfer_to_{data.to_agent}") + elif isinstance(data, GenerationSpanData) and data.output and data.output[0].get("object") == "response": + # Streamed Chat Completions calls record a Responses object OpenInference can't read. + data.output = [_chat_message(data.output[0])] + + def on_trace_start(self, t): pass + def on_trace_end(self, t): pass + def on_span_start(self, span): pass + def shutdown(self): pass + def force_flush(self): pass + + +provider = TracerProvider() # resource from OTEL_SERVICE_NAME / OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +set_trace_processors([MapleSpanFixes()]) # drops the SDK's upload to OpenAI +OpenAIAgentsInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), + exclusive_processor=False, # append after MapleSpanFixes +) +``` + +- `enable_genai_semconv=True` is REQUIRED. It dual-writes `gen_ai.*` (operation name, `{role, parts}` messages, usage, agent name, tool name/description/result, finish reasons, `gen_ai.conversation.id`). Without it Maple's session page has no transcript for this framework. Env equivalent `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` only works if set before `TraceConfig()` is built. +- `MapleSpanFixes` is required. It runs on each SDK span before the OpenInference processor and fixes three gaps: (1) the dual-write fills `gen_ai.tool.call.arguments` from `tool.parameters`, the tool's JSON schema, so it sets the real arguments first (the dual-write never overwrites a set key); (2) handoff spans get no tool name, so it sets `gen_ai.tool.name=transfer_to_`; (3) streamed Chat Completions calls (`run_streamed` on `OpenAIChatCompletionsModel`, LiteLLM, any-llm) record their output as a Responses object that OpenInference can't parse, so the streamed reply is missing from the transcript; it rewrites that output into a Chat Completions message. It must be FIRST in the SDK's processor list, hence `set_trace_processors([...])` + `exclusive_processor=False`. Keep the OpenAI dashboard upload too only if the user asks: `set_trace_processors([MapleSpanFixes(), default_processor()])` (`from agents.tracing.processors import default_processor`; needs a valid OpenAI key). + +### 2d. Non-OpenAI models (OpenRouter, LiteLLM proxy, vLLM, Ollama, Azure-compatible) + +If any agent uses a Chat Completions model on a base URL other than `api.openai.com`, set `ModelSettings(include_usage=True)` on those agents (or in `RunConfig(model_settings=...)`): + +```py +agent = Agent( + name="assistant", + instructions="...", + model=OpenAIChatCompletionsModel(model="openai/gpt-4o-mini", openai_client=client), + model_settings=ModelSettings(include_usage=True), +) +``` + +Without it the SDK sends no `stream_options` for non-OpenAI clients and every `run_streamed` turn has zero tokens. Non-streamed calls are unaffected. Responses API models on OpenAI need nothing. + +### 2e. TypeScript (`@openai/agents`) only + +Verified with `@openai/agents` 0.18.0, `@arizeai/openinference-instrumentation-openai-agents` 0.2.15, `@opentelemetry/sdk-trace-node` 2.11. Use a `NodeTracerProvider({ resource: detectResources({ detectors: [envDetector] }), spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())] })` (`@opentelemetry/resources`, `@opentelemetry/exporter-trace-otlp-proto`; the 2.x provider does NOT read `OTEL_SERVICE_NAME` without `envDetector`, you get `unknown_service:node`), `provider.register()`, then `new OpenAIAgentsInstrumentation({ tracerProvider: provider }).manuallyInstrument(agents)`. Session: wrap each run in `context.with(setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }), () => agents.run(agent, text))` (`setAttributes` from `@arizeai/openinference-core`). Tell the user the limits: framework shows as Unidentified, model replies render as the raw API response JSON (inputs render as messages), no tool arguments/results fields, no agent lanes. `await provider.forceFlush()` before a script exits. Maple ignores `session.id`/`setSession` for this scope. + +## Step 3: Session id (one conversation = one session) + +Maple reads `session.id` on this framework's spans. Only OpenInference's `using_session` sets it. `RunConfig(group_id=...)`, `trace_metadata` and SDK `Session` ids are NOT exported. + +```py +from openinference.instrumentation import using_session + +async def handle_message(conversation_id: str, text: str) -> str: + with using_session(conversation_id): + result = await Runner.run( + agent, text, + session=SQLiteSession(conversation_id, "chats.db"), # or the app's existing Session + run_config=RunConfig(workflow_name=" workflow"), + ) + return result.final_output +``` + +- Wrap EVERY `Runner.run` / `run_sync` / `run_streamed` call in `using_session()`. Use the id the app already stores the chat under (the same one it passes as `group_id` or to its `Session`). Never mint a new id per request; never a constant. +- `run_streamed`: call it inside the `with` block (its background task inherits the context there); consuming `stream_events()` may continue inside or after. +- Human-in-the-loop resumes (`Runner.run(agent, state)` after `state.approve(...)`): wrap in the same `using_session` id. The resume is its own trace (a second turn with the same user message as label) whose root is the workflow-named span, not an `invoke_agent` root. +- Set `RunConfig(workflow_name=...)` to name the trace root; the default `Agent workflow` is the same for every run. The name must NOT contain `chat`, `completion` or `tool` (any case): the per-run CHAIN span carries it with no operation, and Maple's name fallback then counts every run as an extra LLM call (`chat`, `completion`) or tool call (`tool`). End it in `workflow` or `agent`. +- Multi-agent fan-out with several `Runner.run` calls: wrap them in one `with trace("")` (from `agents`) inside `using_session`, or each run becomes its own trace/turn. +- Do not set `gen_ai.conversation.id` or `maple_ai.session.id` by hand; the dual-write copies `session.id` to `gen_ai.conversation.id` already. + +## Step 4: Content + +- Content is ON by default (SDK `trace_include_sensitive_data=True`, OpenInference copies it). With Step 2c, messages land in `gen_ai.input.messages` / `gen_ai.output.messages`. Nothing to enable. +- Content is sent three times per model span (flattened `llm.input_messages.*`, `input.value`, `gen_ai.input.messages`). Expected; don't strip the OpenInference keys, the GenAI ones are derived from them. +- User wants no content / PII-sensitive: `OPENAI_AGENTS_TRACE_INCLUDE_SENSITIVE_DATA=false` (or `RunConfig(trace_include_sensitive_data=False)`). Tell them: empty transcript, tool error details redacted, tokens/tools/failures remain. Alternative at the bridge: `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)`. + +## Step 5: Tools, errors, sub-agents + +- Function tools: span named after the tool, `execute_tool`, `gen_ai.tool.name`, description, arguments (via `MapleSpanFixes`), result. A tool that RAISES: the SDK catches it and the span gets status ERROR with message `Error running tool (non-fatal): {...}`; Maple counts it failed. A tool that RETURNS an error string counts as success; point it out, don't change behavior unasked. +- Give every `Agent` a distinct `name=`. Agent spans carry `gen_ai.agent.name`; Maple draws one lane per name. +- Agents as tools (`agent.as_tool(...)`): nested run appears inside the calling tool span. Nothing to add. +- Handoffs: a `handoff to ` span (counted as a tool call named `transfer_to_` via `MapleSpanFixes`) and the target agent span as a sibling of the source agent's. Nothing to add. +- `needs_approval=True` tools: the paused run records a tool span without a result, and the resumed run records the executed call again, so Maple shows the tool twice for one approved call. Expected; tell the user. +- Known gaps, don't try to fix: no `gen_ai.tool.call.id` on tool spans; no `gen_ai.response.id` on Chat Completions model spans; no cost. + +## Step 6: Flush + +- Long-running server: nothing to add; the provider flushes at normal exit. +- Script / CLI / worker: + + ```py + from tracing import provider + try: + asyncio.run(main()) + finally: + provider.force_flush() + provider.shutdown() + ``` + +- Serverless handler: `provider.force_flush()` before returning from EVERY invocation; never `shutdown()`. +- Notebook: `provider.force_flush()` after the cell that runs the agent. +- `agents.flush_traces()` is not enough: the bridge's `force_flush` is a no-op; spans sit in the OTel batch processor. + +## Step 7: Verify + +Run one real conversation: 2-3 turns with the same conversation id including one tool call and one streamed turn, plus a second conversation with a different id. Flush. Wait ~1 minute. In Maple **Agent Sessions**, filtered by the service name (or via the Maple MCP `list_agent_sessions` + `get_agent_session`), check: + +- [ ] Exactly one session per conversation id; none named `trace:` (that means a run was outside `using_session`). +- [ ] The two conversations are two different sessions. +- [ ] Framework shows **OpenAI Agents SDK** (Python). Unidentified = wrong/old bridge or TypeScript. +- [ ] One turn per `Runner.run`; turn labels are the user messages; root span named after `workflow_name`. +- [ ] Transcript shows instructions, user and assistant messages, tool calls, including the streamed turn's reply (missing there = `MapleSpanFixes` absent or not first). Empty transcript with non-zero tokens = `enable_genai_semconv` not active. +- [ ] Model calls (`generation` for Chat Completions, `response` for Responses API) show the model and non-zero input/output tokens, INCLUDING the streamed turn (zero there = Step 2d missing). LLM call count equals real model calls (higher = `chat`/`completion` in `workflow_name`). +- [ ] Each tool call appears once, with its real name and real arguments (not a JSON schema); a raised tool error is marked failed and nothing else is. +- [ ] Multi-agent: one lane per agent name; all sub-agent spans in one trace with the same session. +- [ ] Each model call appears once (no second `ChatCompletion`-style span from another instrumentor). +- [ ] Cost shows "unpriced" (expected; nothing emits cost). +- [ ] No span attribute contains the model provider API key, `Bearer ` or `sk-`. + +Raw span check (optional, e.g. an `InMemorySpanExporter` in a scratch run): model spans have `gen_ai.operation.name=chat`, `gen_ai.input.messages`, `gen_ai.usage.input_tokens`, `session.id`; tool spans have `gen_ai.operation.name=execute_tool`, `gen_ai.tool.call.arguments` equal to the call's arguments; agent spans have `gen_ai.agent.name`. + +## Do not + +- Do not disable the SDK's tracing (`set_tracing_disabled(True)`, `OPENAI_AGENTS_DISABLE_TRACING`, `tracing_disabled=True`) to stop the OpenAI upload; `set_trace_processors` already removes it. +- Do not omit `TraceConfig(enable_genai_semconv=True)`. +- Do not register `MapleSpanFixes` with `add_trace_processor` or after `instrument()`; it must precede the OpenInference processor. +- Do not put `chat`, `completion` or `tool` in `workflow_name`. +- Do not pass OpenRouter-style model strings (`Agent(model="openai/gpt-4o-mini")`): the SDK strips `openai/` and rejects other prefixes (`Unknown prefix: anthropic`). Use `OpenAIChatCompletionsModel(model=..., openai_client=...)`. +- Do not rely on `group_id`, `trace_metadata` or SDK `Session` ids for the Maple session; use `using_session`. +- Do not generate a fresh session id per request or use a constant. +- Do not stack `openinference-instrumentation-openai`, Logfire, Langfuse, Traceloop or the OTel contrib Agents instrumentation on the same process. +- Do not skip `include_usage=True` for streamed non-OpenAI Chat Completions models. +- Do not create a second `TracerProvider` or call `instrument()` twice. +- Do not use `SimpleSpanProcessor` or a console exporter in servers. +- Do not skip the flush in scripts, notebooks, CLIs and serverless. diff --git a/skills/maple-agent-tracing-openrouter/SKILL.md b/skills/maple-agent-tracing-openrouter/SKILL.md new file mode 100644 index 0000000000..6816d1d2f0 --- /dev/null +++ b/skills/maple-agent-tracing-openrouter/SKILL.md @@ -0,0 +1,166 @@ +--- +name: maple-agent-tracing-openrouter +description: "Trace OpenRouter calls with Maple: route OpenRouter Broadcast (OTLP) to Maple and edit the app's OpenRouter requests to send session_id and trace ids, so each conversation is one Maple Agent Session with model calls, tokens and real cost. Triggers on 'trace my openrouter calls', 'add Maple to openrouter', 'agent sessions for openrouter', 'OpenTelemetry for openrouter', 'openrouter broadcast to maple'." +--- + +# Maple agent tracing: OpenRouter Broadcast + +Goal: every conversation = one Maple Agent Session with each OpenRouter model call, its tokens, cost (`gen_ai.usage.total_cost`, OpenRouter's real charge) and prompt/completion. If the app already exports its own traces to Maple, the Broadcast spans nest inside them and each call is counted once. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/openrouter + +Mechanism: OpenRouter Broadcast, configured in the OpenRouter dashboard, exports one OTLP/HTTP JSON trace per request (scope and `service.name` = `openrouter`, root span `LLM Generation`, children `provider attempt N: `). Maple detects vendor `openrouter` from the scope and reads **`session.id`** as the session key. OpenRouter sets `session.id` from the request's `session_id` body field or `x-session-id` header, and uses `trace.trace_id` / `trace.parent_span_id` from the body verbatim as the OTLP trace id / parent span id. + +You cannot change the OpenRouter dashboard. Your job: (1) edit the app's OpenRouter calls, (2) hand the user the exact destination settings. + +What Broadcast cannot give (tell the user, don't try to fix it here): tool spans, tool failures, agent names / sub-agent lanes. Those need the app's framework instrumentation (router skill `maple-agent-tracing`). Content arrives as wrapped JSON (`{"messages":[...]}`, `{"completion":...}`), which Maple shows as a raw JSON block, without turn labels. + +## Step 0: detect + +1. Find every OpenRouter call site: `openrouter.ai` base URLs (`https://openrouter.ai/api/v1`, `https://eu.openrouter.ai/api/v1`), `OPENROUTER_API_KEY`, `@openrouter/ai-sdk-provider`, `@openrouter/sdk`, the `openrouter` PyPI package, `openai` clients with an OpenRouter `baseURL` / `base_url`, LiteLLM `openrouter/` models, framework model classes pointed at OpenRouter. +2. Find where the conversation id lives per request: chat thread id, conversation id, session id, agent run id. It must be per conversation, never a process constant or a per-client default. +3. Existing OTel exporting to Maple? Search for `TracerProvider`, `NodeSDK`, `registerOTel`, `registerTelemetry`, `OTEL_EXPORTER_OTLP_ENDPOINT`, `ingest.maple.dev`, `logfire.configure`, OpenInference / OpenLLMetry instrumentors. + - Yes → do Step 2 and Step 3 (nesting). + - No → Step 2 only (Broadcast-only). Mention that the app's framework skill adds tool spans and agent structure. +4. Which endpoint region the app calls (`openrouter.ai` vs `eu.openrouter.ai`), for the destination's data regions. + +## Step 1: key and region (for the destination headers) + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. Test Connection passes with it, but nothing lands. +- Never put a private `maple_sk_` key in browser code. The key here goes into OpenRouter's dashboard, not the repo. +- If the app also exports its own OTel to Maple, follow the repo's existing secret/env convention for that exporter. Ingest keys are write-only, so inline is acceptable if there is none. + +## Step 2: send `session_id` on every OpenRouter request + +Same id for every request of one conversation, new id per conversation, max 256 characters. If the app also has framework instrumentation, use the SAME value as the framework's conversation/session id (a trace with two different ids is assigned to the lexically larger one, silently). + +`openai` npm (>= 7; field is untyped, sent as-is): + +```ts +const completion = await client.chat.completions.create({ + model, + messages, + // @ts-expect-error OpenRouter-only field + session_id: conversationId, +}) +``` + +Or per-request header, no type workaround: + +```ts +await client.chat.completions.create({ model, messages }, { headers: { "x-session-id": conversationId } }) +``` + +`openai` PyPI (>= 3): `extra_body={"session_id": conversation_id}` on `chat.completions.create(...)` (and on `responses.create(...)` if used). + +Vercel AI SDK + `@openrouter/ai-sdk-provider` (>= 3.1): `providerOptions: { openrouter: { session_id: conversationId } }` on `generateText` / `streamText` / agent calls. Everything under `providerOptions.openrouter` is merged into the body. + +`@openrouter/sdk`: `openRouter.chat.send({ model, messages, sessionId: conversationId })`. `openrouter` PyPI: `session_id=conversation_id`. + +Other clients: find the framework's extra-body or per-request-headers option and set `session_id` / `x-session-id`. Check the outgoing request (log the body once, or a test that captures `fetch`) to confirm the field is actually sent; some wrappers drop unknown fields. + +Optional: `user` (<= 128 chars) is forwarded as `user.id`. Maple does not use it for sessions. Never put emails/names in `user`, `session_id` or `trace` metadata: Privacy Mode does not strip them. + +## Step 3: nest Broadcast under the app's own traces (only if the app exports OTel to Maple) + +Without this, every model call is recorded twice (app span + Broadcast trace) and Broadcast-only turns split one per model call. Put the active span's W3C ids in `trace.trace_id` (32 hex) and `trace.parent_span_id` (16 hex). + +TypeScript: wrap `fetch` and pass it to the client (`new OpenAI({ baseURL, apiKey, fetch: openRouterFetch })` or `createOpenRouter({ apiKey, fetch: openRouterFetch })`). Reads the span active when the SDK sends the request, which is the framework's model-call span if it has one: + +```ts +import { trace } from "@opentelemetry/api" + +// Nests each OpenRouter Broadcast trace under the span that made the request. +export const openRouterFetch: typeof fetch = (input, init) => { + const span = trace.getActiveSpan()?.spanContext() + if (span && typeof init?.body === "string") { + const body = JSON.parse(init.body) + body.trace = { ...body.trace, trace_id: span.traceId, parent_span_id: span.spanId } + init = { ...init, body: JSON.stringify(body) } + } + return fetch(input, init) +} +``` + +- Add `@opentelemetry/api` only if it isn't already a dependency (it is, wherever OTel is set up). +- If the app has a single custom `fetch` already, compose with it instead of replacing it. + +Python (`openai` or any client with `extra_body`): build the body at the call site: + +```py +from opentelemetry import trace + + +def openrouter_extra_body(conversation_id: str) -> dict: + body = {"session_id": conversation_id} + ctx = trace.get_current_span().get_span_context() + if ctx.is_valid: + body["trace"] = { + "trace_id": format(ctx.trace_id, "032x"), + "parent_span_id": format(ctx.span_id, "016x"), + } + return body +``` + +This parents to the span current at the call site (turn/agent span), so the app's model span and Broadcast's `LLM Generation` are siblings. Maple dedupes siblings only by `gen_ai.response.id` (OpenRouter's `gen-…` id). Check that the app's instrumentation records the response id (Vercel AI SDK `ai.response.id`, OTel `openai-v2` instrumentor `gen_ai.response.id`). If it doesn't, tell the user tokens will double-count for that service, and offer to exclude that service's OpenRouter API key from the destination instead. + +Don't set `trace_name` / `span_name` / `generation_name` unless the user asks: `span_name` creates an extra intermediate span. + +## Step 4: content + +Broadcast includes prompts and completions by default (`gen_ai.prompt`, `gen_ai.completion` on `LLM Generation`). The completion object has also been seen echoing the request body (tool definitions, `user`, `session_id`, `trace`). Nothing to change in code. If the user must keep content out of Maple: tell them to enable **Privacy Mode** on the destination (tokens, cost, timing and metadata still arrive). + +## Step 5: tools, errors, sub-agents + +- Broadcast has no tool spans and no agent names. Don't add fake tool spans. Point the user to their framework's skill for tools/lanes. +- Provider fallbacks show as `provider attempt N: ` children; a failed attempt followed by a successful one is a retry and not counted as a failure. +- A call where every provider failed: `LLM Generation` status Error, message `Provider returned error`, counted as `provider_error`. On this path OpenRouter drops the `trace` object, so it's its own trace (still in the right session via `session.id`). + +## Step 6: flush + +Broadcast needs none: OpenRouter exports from its servers after each request. Delivery lag is about a minute. If the app exports its own spans, keep/ensure its normal shutdown flush (`sdk.shutdown()`, `provider.force_flush()`), otherwise nested Broadcast spans point at a missing parent. + +## Step 7: hand the user the dashboard settings + +Print these, filled in (you cannot apply them): + +1. OpenRouter → Settings → Observability (`https://openrouter.ai/settings/observability`) → **Enable Broadcast** (org accounts: org admin only). +2. Edit **OpenTelemetry Collector**: + - Endpoint: `https://ingest.maple.dev/v1/traces` (EU: `https://ingest.eu.maple.dev/v1/traces`). Full path; OpenRouter does not append `/v1/traces`. + - Headers: `{ "Authorization": "Bearer " }` + - Sampling rate: `1` (sampling is per session; lower values drop whole conversations). + - API keys: empty, or include the key(s) the app uses. Excluded keys always win. + - Data regions: include Europe if the app calls `eu.openrouter.ai`. + - Privacy Mode: off unless content must not leave OpenRouter. + - Leave **Additional generation metadata → Cost** off; Maple reads `gen_ai.usage.total_cost`, which is sent anyway. +3. Click **Test Connection**; it only saves if the test passes. The test creates a sessionless `openrouter-connection-test` trace (`trace:0000…0001` in Maple); ignore it. + +## Step 8: verify + +Run one conversation (3+ turns, one with a tool call) with a fixed `session_id`, then a second conversation. Wait about a minute. In Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), filter service `openrouter` (plus the app's service if nested). Check: + +- Exactly one session per conversation, id = your `session_id`, vendor OpenRouter. No `trace:` sessions except the connection test. +- Each model call = one `LLM Generation` span with `gen_ai.operation.name=chat`, `gen_ai.request.model` (OpenRouter slug), `gen_ai.response.id` (`gen-…`), input/output tokens, `gen_ai.usage.total_cost`. +- Session tokens, LLM-call count and cost ≈ what the app made (not ~2x). If ~2x: Step 3 nesting or response-id match is missing. +- Nested case: `LLM Generation` has a parent in the app's trace; turns = the app's turns, not one per model call. +- Transcript shows each call's prompt/completion JSON (absent if Privacy Mode is on). +- Streamed calls have tokens too (OpenRouter accounts server-side). +- Tool counts from Broadcast are 0 (expected). Tool spans only come from app instrumentation. +- No attribute contains `sk-or-`, `Bearer ` or `maple_sk_`. + +Tell the user: cost is shown (OpenRouter's charge); cache writes, TTFT, environment, tool calls and agent lanes are not available from Broadcast; transcript is raw JSON. + +## Do not + +- Do not use a constant or per-client `session_id`; it merges every user into one session. +- Do not use non-hex or wrong-length `trace_id` / `parent_span_id`; use the active span's W3C ids. +- Do not use a different id for `session_id` than the framework's conversation id when both exist. +- Do not stack Broadcast on in-app instrumentation without nesting or a response-id match; tokens and calls double. +- Do not claim Broadcast captures tools, tool errors or sub-agents. +- Do not point the destination at `https://ingest.maple.dev` without `/v1/traces`, and do not put the Maple key anywhere in the repo for Broadcast. +- Do not set sampling below 1 to "save volume" without telling the user whole sessions disappear. +- Do not create or edit OpenRouter destinations via the management API unless the user explicitly gives a management key and asks. +- Do not put PII in `user`, `session_id` or `trace` metadata. diff --git a/skills/maple-agent-tracing-opentelemetry/SKILL.md b/skills/maple-agent-tracing-opentelemetry/SKILL.md new file mode 100644 index 0000000000..07f38dd923 --- /dev/null +++ b/skills/maple-agent-tracing-opentelemetry/SKILL.md @@ -0,0 +1,144 @@ +--- +name: maple-agent-tracing-opentelemetry +description: "Trace a hand-rolled or unsupported AI agent with Maple by emitting the OpenTelemetry GenAI conventions yourself (invoke_agent, chat, execute_tool spans) in any language: TypeScript, Python, Go, Rust, Ruby, Elixir, Java, .NET. Triggers on 'trace my custom agent', 'add Maple to my agent loop', 'agent sessions without a framework', 'OpenTelemetry GenAI spans by hand', 'OpenTelemetry for my AI agent in Go/Rust/Ruby'." +--- + +# Maple agent tracing: OpenTelemetry GenAI conventions (any language) + +Goal: every conversation = one Maple Agent Session. Each user message = one trace rooted at an `invoke_agent` span, with a `chat` span per model call (model, tokens, transcript) and an `execute_tool` span per tool call (args, result, failures). + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/opentelemetry + +Mechanism: you write the spans. Maple classifies a span only by `gen_ai.operation.name`, groups a trace by `gen_ai.conversation.id`, and reads content only from span attributes holding JSON strings. Hand-written spans show as framework "Unidentified" (vendor `unknown:genai`); that is expected. + +## Step 0: detect + +1. Language and entry points (web server, workers, scripts, serverless handlers). +2. Is a supported framework the real agent runtime? (`@mastra/core`, `ai`, `@openai/agents`/`openai-agents`, `langchain`/`langgraph`, `pydantic-ai`, `crewai`, `google-adk`, `llama-index`, `strands-agents`, `smolagents`, `agno`, `dspy`, `haystack-ai`, `agent-framework`, Spring AI, `litellm`, Claude Agent SDK). If yes, stop and use `maple-agent-tracing-` instead; use this skill only for the parts that framework doesn't cover, or for the `maple_ai.session.id` wrapper (Step 4). +3. Existing OTel setup. Search for `TracerProvider`, `NodeTracerProvider`, `NodeSDK`, `registerOTel`, `set_tracer_provider`, `opentelemetry-instrument`, `logfire.configure`, `sentry_sdk.init`/`Sentry.init`, `otel.SetTracerProvider`. Exists → add Maple's exporter/processor to it; never create a second provider. +4. Existing GenAI auto-instrumentation on the model client (`@opentelemetry/instrumentation-openai`, `opentelemetry-instrumentation-openai-v2`, OpenLLMetry `Traceloop.init`, OpenInference `OpenAIInstrumentor`, `logfire.instrument_openai`). Pick one source of `chat` spans: either keep that instrumentation (then see `maple-agent-tracing-provider-sdks`) or remove it and write `chat` spans here. Both = every model call twice. +5. Find in the code: the agent loop (where one user message is handled), every model call site, every tool dispatch, sub-agent calls, and where the conversation/chat/thread id lives in the request. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention if it has one. Otherwise inline is acceptable: ingest keys are write-only. + +```bash +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +The exporters append `/v1/traces`. If an SDK rejects the space in the header, use `Bearer%20`. + +## Step 2: install + init + +Read the reference for the language and adapt it: +- TypeScript/Node: `references/typescript.md` +- Python: `references/python.md` +- Go, Rust, Ruby, Elixir, Java, .NET: `references/go.md` (Go code + notes for the others) + +Rules: +- Init module is imported first in every entry point. Set a real `service.name` and `deployment.environment.name`. +- Name the tracer after the app (e.g. `support-agent`). Never `openrouter`, `langsmith`, `litellm`, `haystack`, `ai`, `gen_ai`: Maple fingerprints frameworks by scope name and would read a different session key. +- Keep the project's loop structure; add spans around its existing calls. Use the reference's complete loop only when there is no loop yet. + +## Step 3: the three spans (exact keys) + +`invoke_agent` (kind INTERNAL, name `invoke_agent `), around one agent run; for a user turn it is the trace root: +- `gen_ai.operation.name`=`invoke_agent`, `gen_ai.agent.name`, `gen_ai.conversation.id` (Step 4) +- optional: `gen_ai.input.messages` (the user message), `gen_ai.output.messages` (final answer) + +`chat` (kind CLIENT, name `chat `), around each model call: +- at start: `gen_ai.operation.name`=`chat` (or `generate_content`/`text_completion`), `gen_ai.provider.name`, `gen_ai.request.model`, `gen_ai.system_instructions`, `gen_ai.input.messages` +- at end: `gen_ai.response.id`, `gen_ai.response.model`, `gen_ai.response.finish_reasons` (string array), `gen_ai.output.messages`, usage (Step 7), `gen_ai.response.time_to_first_chunk` (double, SECONDS, streamed calls) + +`execute_tool` (kind INTERNAL, name `execute_tool `), around each tool call: +- `gen_ai.operation.name`=`execute_tool`, `gen_ai.tool.name` (real name), `gen_ai.tool.call.id` (the model's call id), `gen_ai.tool.type`=`function` +- `gen_ai.tool.call.arguments`: JSON string of an object +- `gen_ai.tool.call.result`: JSON string of an object/array. Wrap scalars: `{"result": }` (a bare JSON string is dropped). + +Message JSON (`input.messages`/`output.messages`): array of `{role, parts}`; parts `{type:"text",content}`, `{type:"tool_call",id,name,arguments:}`, `{type:"tool_call_response",id,response}`, `{type:"reasoning",content}`. Output messages add `finish_reason`. `system_instructions` = array of parts, no role: `[{"type":"text","content":"..."}]`. Always a JSON **string** attribute; plain text and structured (non-string) attribute values don't render. + +`provider.name` = the API actually called: `openai`, `anthropic`, `gcp.gemini`, `gcp.vertex_ai`, `aws.bedrock`, `azure.ai.openai`, `mistral_ai`, `groq`, `x_ai`, `deepseek`, or `openrouter` for OpenRouter. + +## Step 4: session id (required) + +- Set `gen_ai.conversation.id` on the turn's `invoke_agent` span from the app's conversation/chat/thread id. Same value for every message of a conversation; different across conversations. One classified span per trace is enough; every span in the trace joins. +- Never: `uuid4()`/`randomUUID()` per request, the trace id, a module-level constant, a per-process default. No real id (single-shot script) → generate one per conversation, not per message, and reuse it. +- Sub-agents in the same trace: no id (they inherit the trace's session) or the same id. Two different ids in one trace → the lexically larger silently wins. +- Chat backends: one message list per conversation (keyed by that id, persisted), never one global list. + +Escape hatch, only when a framework's spans carry a session key Maple ignores for that framework (OpenInference `session.id` from LangChain/LlamaIndex instrumentors, Vercel AI SDK `runtimeContext`, LiteLLM, Haystack): wrap each turn in your own span with `gen_ai.operation.name`=`invoke_agent`, `gen_ai.agent.name`, and `maple_ai.session.id`=, and run the framework inside it. Rules: +- Only on your own wrapper span, never on framework spans (it re-vendors the span to `maple`: framework decoding lost, usage read as inclusive of cache). +- Use the same value the framework would use for the session. +- Not needed for hand-written spans: use `gen_ai.conversation.id`. + +## Step 5: content + +- Content = `gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages` (chat), `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result` (execute_tool). On span attributes only: span events, log records, and indexed keys (`gen_ai.prompt.0.content`, `llm.input_messages.0.*`) are not read. +- Do not set `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` / `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT`; if the platform sets one, unset it. Truncated JSON is dropped whole. To cap size, drop the oldest messages whole before serializing. +- Replace base64 images/files in history with a placeholder part (ingest rejects requests > 20 MiB with 413). +- User wants no content → skip those five attributes (everything else still works; transcript empty). Wants redaction → redact inside `toSemconv`/`to_semconv` before serializing, or an OTel Collector `redaction`/`transform` processor. +- Never put API keys or `Authorization` headers in any attribute. + +## Step 6: tools, errors, sub-agents + +- Tool failure: set status ERROR with the error message as description, set `error.type` (exception class or error code), no `gen_ai.tool.call.result`, then return the error to the model as the tool result so the loop continues. Keep the message specific: Maple groups failures by it. +- Tools that return `{"error": ...}` instead of raising: mark the span failed the same way when you detect it. +- Model call failure: status ERROR + `error.type` (HTTP status or exception class), rethrow. +- Agent run failure: same on `invoke_agent`. +- Sub-agent: call its loop inside the delegating tool's `execute_tool` span, so `execute_tool ask_x` → `invoke_agent x` → its `chat`/`execute_tool`. Distinct `gen_ai.agent.name` per agent (lanes need it). +- Parallel tools: start each `execute_tool` span inside the turn's context (Node `Promise.all` keeps it; Python threads need `contextvars.copy_context().run`). + +## Step 7: tokens and cost + +- Usage on `chat` spans only, never cumulative totals on `invoke_agent`. +- Keys: `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` (ints). Not read: `total_tokens`, `reasoning_tokens`, `cache_read_input_tokens`. +- Copy the provider's raw numbers. Maple interprets by `gen_ai.provider.name`: `anthropic` → input EXCLUDES cache (send Anthropic's raw `input_tokens`, NOT input+cache as the spec says, or cache is double counted); `gcp.gemini`/`gcp.vertex_ai` → input includes cache, output excludes thoughts (raw Gemini counts); `openai`/`openrouter`/other → input includes cache, output includes reasoning (OpenAI shape). +- Streaming OpenAI-compatible: `stream_options: {include_usage: true}`; usage is on the last chunk (empty `choices`). OpenRouter always sends usage. +- Cost: `gen_ai.usage.cost` (double, USD) on `chat` spans. OpenRouter returns `usage.cost` → copy it. Other providers return none → compute only if the project has a price table; otherwise leave it (Maple shows "unpriced"; it never prices tokens). +- `gen_ai.response.id` always (dedupes against gateway mirrors such as OpenRouter Broadcast). + +## Step 8: flush + +- Node script/CLI: `await provider.shutdown()` in `finally`. Serverless: `await provider.forceFlush()` before returning (inside `waitUntil`/`after()` if available). +- Python script: `provider.shutdown()` in `finally`. Lambda: `force_flush()` in `finally`. Notebooks/workers: `force_flush()` per cell/task. +- Go: `defer tp.Shutdown(context.Background())` in `main`. + +## Step 9: verify + +Run one real conversation: 2+ messages with the same id, one streamed reply if the app streams, one tool call, one failing tool if one exists, one sub-agent call if the app delegates; then a second conversation. Wait ~30 s; Maple → Agent Sessions, filter by service name. Check: + +- [ ] One session per conversation; session id = the conversation id; second conversation = different session; no `trace:` sessions. +- [ ] One turn per user message, labeled with it; transcript shows system instructions, user messages, assistant replies, tool calls and results. +- [ ] Spans `invoke_agent ` → `chat ` / `execute_tool `, all in the turn's trace (no orphan roots). +- [ ] Every `chat` span: model, provider, response id, input+output tokens (streamed ones too), finish reasons; TTFT in seconds on streamed calls. +- [ ] Tool spans: real names, call ids matching the model's tool calls, JSON args and results. +- [ ] The failing tool is failed with its message; successful tools and model calls are not. +- [ ] Sub-agents: own lane with their `gen_ai.agent.name`, same session. +- [ ] Cost present only if spans carry `gen_ai.usage.cost`; else "unpriced". +- [ ] No attribute contains an API key, `Bearer `, `sk-or-`, `maple_sk_`. + +Local check without Maple: temporarily add a console exporter (`ConsoleSpanExporter` + `SimpleSpanProcessor`) and confirm the tree, the conversation id, and that every messages attribute parses with `JSON.parse`/`json.loads`. + +## Do not + +- Do not emit spans without `gen_ai.operation.name` and expect them in Agent Sessions. +- Do not send messages as plain text, structured attribute values, span events or logs. +- Do not generate a conversation id per request or use the trace id. +- Do not give sub-agents their own conversation ids. +- Do not set attribute length limits. +- Do not stack a provider auto-instrumentor on top of hand-written `chat` spans. +- Do not put `maple_ai.session.id` on framework spans or on `chat` spans. +- Do not sum Anthropic cache tokens into `input_tokens`. +- Do not put usage on `invoke_agent` spans. +- Do not use `gen_ai.system` in new code (read as a fallback only); use `gen_ai.provider.name`. +- Do not name the tracer after a framework or gateway. +- Do not create a second TracerProvider. +- Do not print or commit real keys beyond the repo's convention. diff --git a/skills/maple-agent-tracing-opentelemetry/references/go.md b/skills/maple-agent-tracing-opentelemetry/references/go.md new file mode 100644 index 0000000000..1c1eb4eb8d --- /dev/null +++ b/skills/maple-agent-tracing-opentelemetry/references/go.md @@ -0,0 +1,153 @@ +# Go reference + +Pattern for `go.opentelemetry.io/otel` v1.46 (`sdk`, `exporters/otlp/otlptrace/otlptracehttp` v1.46). Not compiled in authoring; run `go vet ./...` after adding it. + +```bash +go get go.opentelemetry.io/otel go.opentelemetry.io/otel/sdk go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp +``` + +Wrap the project's existing client call in `Chat` and tool dispatch in `Tool`. Pass the `ctx` returned by `StartTurn` into both, or the spans become separate traces. Set `gen_ai.provider.name` to the API actually called. Copy cache/reasoning counts when the provider returns them (`gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.reasoning.output_tokens`). + +```go +// genai.go +package agent + +import ( + "context" + "encoding/json" + "fmt" + + "go.opentelemetry.io/otel" + "go.opentelemetry.io/otel/attribute" + "go.opentelemetry.io/otel/codes" + "go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp" + "go.opentelemetry.io/otel/sdk/resource" + sdktrace "go.opentelemetry.io/otel/sdk/trace" + "go.opentelemetry.io/otel/trace" +) + +var tracer = otel.Tracer("support-agent") + +// SetupTracing reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. +// Call Shutdown on the returned provider before the process exits. +func SetupTracing(ctx context.Context) (*sdktrace.TracerProvider, error) { + exporter, err := otlptracehttp.New(ctx) + if err != nil { + return nil, err + } + tp := sdktrace.NewTracerProvider( + sdktrace.WithBatcher(exporter), + sdktrace.WithResource(resource.NewSchemaless( + attribute.String("service.name", "support-agent"), + attribute.String("deployment.environment.name", "production"), + )), + ) + otel.SetTracerProvider(tp) + return tp, nil +} + +// Message is the GenAI semconv shape: {role, parts}. +type Message struct { + Role string `json:"role"` + Parts []map[string]any `json:"parts"` + FinishReason string `json:"finish_reason,omitempty"` +} + +// ChatResult holds what your provider returned, copied verbatim. +type ChatResult struct { + ID, Model, FinishReason string + Output Message + InputTokens, OutputTokens int64 + CostUSD float64 // 0 when the provider doesn't return a cost +} + +func jsonAttr(key string, value any) attribute.KeyValue { + b, _ := json.Marshal(value) + return attribute.String(key, string(b)) +} + +func fail(span trace.Span, err error) { + span.SetStatus(codes.Error, err.Error()) + span.SetAttributes(attribute.String("error.type", fmt.Sprintf("%T", err))) +} + +// StartTurn opens the invoke_agent span for one user message. End it when the turn is done. +func StartTurn(ctx context.Context, agentName, conversationID string) (context.Context, trace.Span) { + return tracer.Start(ctx, "invoke_agent "+agentName, trace.WithAttributes( + attribute.String("gen_ai.operation.name", "invoke_agent"), + attribute.String("gen_ai.agent.name", agentName), + attribute.String("gen_ai.conversation.id", conversationID), + )) +} + +// Chat wraps one model call (your existing client code goes in call). +func Chat(ctx context.Context, model string, input []Message, call func(context.Context) (ChatResult, error)) (ChatResult, error) { + ctx, span := tracer.Start(ctx, "chat "+model, trace.WithSpanKind(trace.SpanKindClient), trace.WithAttributes( + attribute.String("gen_ai.operation.name", "chat"), + attribute.String("gen_ai.provider.name", "openai"), + attribute.String("gen_ai.request.model", model), + jsonAttr("gen_ai.input.messages", input), + )) + defer span.End() + res, err := call(ctx) + if err != nil { + fail(span, err) + return res, err + } + res.Output.FinishReason = res.FinishReason + span.SetAttributes( + attribute.String("gen_ai.response.id", res.ID), + attribute.String("gen_ai.response.model", res.Model), + attribute.StringSlice("gen_ai.response.finish_reasons", []string{res.FinishReason}), + attribute.Int64("gen_ai.usage.input_tokens", res.InputTokens), + attribute.Int64("gen_ai.usage.output_tokens", res.OutputTokens), + jsonAttr("gen_ai.output.messages", []Message{res.Output}), + ) + if res.CostUSD > 0 { + span.SetAttributes(attribute.Float64("gen_ai.usage.cost", res.CostUSD)) + } + return res, nil +} + +// Tool wraps one tool call. result must marshal to a JSON object or array. +func Tool(ctx context.Context, name, callID, arguments string, run func(context.Context) (any, error)) (string, error) { + ctx, span := tracer.Start(ctx, "execute_tool "+name, trace.WithAttributes( + attribute.String("gen_ai.operation.name", "execute_tool"), + attribute.String("gen_ai.tool.name", name), + attribute.String("gen_ai.tool.call.id", callID), + attribute.String("gen_ai.tool.call.arguments", arguments), + )) + defer span.End() + result, err := run(ctx) + if err != nil { + fail(span, err) + return "", err + } + b, _ := json.Marshal(result) + span.SetAttributes(attribute.String("gen_ai.tool.call.result", string(b))) + return string(b), nil +} +``` + +Usage: + +```go +tp, err := agent.SetupTracing(ctx) +if err != nil { + log.Fatal(err) +} +defer tp.Shutdown(context.Background()) + +ctx, turn := agent.StartTurn(ctx, "support", chatID) +defer turn.End() +res, err := agent.Chat(ctx, model, input, func(ctx context.Context) (agent.ChatResult, error) { + return callModel(ctx, model, input) // your existing client code +}) +``` + +Failure on the turn span: call the same `fail(turn, err)` pattern before returning an error. + +## Other languages (Rust, Ruby, Elixir, Java, .NET) + +Same three spans, same attribute keys and value types (string, int64, double, string array). Serialize every messages/tool payload to a JSON string before setting it. Register the SDK's context propagation so child spans nest (Rust: `Context::current_with_span` / `tracing-opentelemetry`; Ruby: `in_span`; Elixir: `OpenTelemetry.Tracer.with_span`). + diff --git a/skills/maple-agent-tracing-opentelemetry/references/python.md b/skills/maple-agent-tracing-opentelemetry/references/python.md new file mode 100644 index 0000000000..17e330b58b --- /dev/null +++ b/skills/maple-agent-tracing-opentelemetry/references/python.md @@ -0,0 +1,275 @@ +# Python reference (3.10+) + +Tested pattern: `opentelemetry-sdk` 1.45, `opentelemetry-exporter-otlp-proto-http` 1.45, `openai` 3.20 against an OpenAI-compatible Chat Completions API (OpenRouter here). + +This is a complete loop. If the project already has a loop, keep its structure and copy only the span code: `invoke_agent` around one agent run, `chat` around each model call, `execute_tool` around each tool call, `to_semconv` for messages. + +```bash +pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Existing provider (`opentelemetry-instrument`, Sentry, Logfire, Datadog): do not create `tracing.py`; add `BatchSpanProcessor(OTLPSpanExporter())` to it with `add_span_processor`. + +## tracing.py + +```py +# tracing.py: import this first in every entry point +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) +``` + +## agent.py + +```py +# agent.py +import json +import os +import time +from dataclasses import dataclass, field +from typing import Any, Callable + +from openai import OpenAI, omit +from opentelemetry import trace +from opentelemetry.trace import SpanKind, Status, StatusCode + +tracer = trace.get_tracer("support-agent") +client = OpenAI(base_url="https://openrouter.ai/api/v1", api_key=os.environ["OPENROUTER_API_KEY"]) +PROVIDER = "openrouter" # gen_ai.provider.name: who you send the request to + + +@dataclass +class Tool: + definition: dict[str, Any] + run: Callable[..., Any] + + +@dataclass +class Agent: + name: str + model: str + instructions: str + tools: dict[str, Tool] = field(default_factory=dict) + + +def to_semconv(message: dict[str, Any]) -> dict[str, Any]: + """OpenAI message -> GenAI semconv message: {role, parts: [...]}.""" + if message["role"] == "tool": + part = {"type": "tool_call_response", "id": message["tool_call_id"], "response": message["content"]} + return {"role": "tool", "parts": [part]} + parts: list[dict[str, Any]] = [] + if message.get("content"): + parts.append({"type": "text", "content": message["content"]}) + for call in message.get("tool_calls") or []: + fn = call["function"] + parts.append({"type": "tool_call", "id": call["id"], "name": fn["name"], "arguments": json.loads(fn["arguments"] or "{}")}) + return {"role": message["role"], "parts": parts} + + +def mark_failed(span: trace.Span, error: Exception) -> None: + span.set_status(Status(StatusCode.ERROR, str(error))) + span.set_attribute("error.type", type(error).__qualname__) + + +def chat(agent: Agent, messages: list[dict[str, Any]], on_text: Callable[[str], None] | None = None) -> dict[str, Any]: + """One model call = one `chat` span. Streams, so time to first chunk is recorded too.""" + attributes = { + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": PROVIDER, + "gen_ai.request.model": agent.model, + "gen_ai.system_instructions": json.dumps([{"type": "text", "content": agent.instructions}]), + "gen_ai.input.messages": json.dumps([to_semconv(m) for m in messages]), + } + with tracer.start_as_current_span(f"chat {agent.model}", kind=SpanKind.CLIENT, attributes=attributes) as span: + try: + started = time.perf_counter() + stream = client.chat.completions.create( + model=agent.model, + messages=[{"role": "system", "content": agent.instructions}, *messages], + tools=[tool.definition for tool in agent.tools.values()] or omit, + stream=True, + stream_options={"include_usage": True}, # without it, streamed calls report no tokens + ) + response_id, model, text, finish_reason, usage = "", agent.model, "", "stop", None + calls: dict[int, dict[str, Any]] = {} + for chunk in stream: + if not response_id: + span.set_attribute("gen_ai.response.time_to_first_chunk", time.perf_counter() - started) + response_id, model = chunk.id, chunk.model + usage = chunk.usage or usage + if not chunk.choices: + continue + choice = chunk.choices[0] + finish_reason = choice.finish_reason or finish_reason + if choice.delta.content: + text += choice.delta.content + if on_text: + on_text(choice.delta.content) + for delta in choice.delta.tool_calls or []: + call = calls.setdefault(delta.index, {"id": "", "type": "function", "function": {"name": "", "arguments": ""}}) + call["id"] = delta.id or call["id"] + if delta.function: + call["function"]["name"] += delta.function.name or "" + call["function"]["arguments"] += delta.function.arguments or "" + reply: dict[str, Any] = {"role": "assistant", "content": text} + if calls: + reply["tool_calls"] = [calls[i] for i in sorted(calls)] + span.set_attributes({ + "gen_ai.response.id": response_id, + "gen_ai.response.model": model, + "gen_ai.response.finish_reasons": [finish_reason], + "gen_ai.output.messages": json.dumps([{**to_semconv(reply), "finish_reason": finish_reason}]), + }) + if usage: + span.set_attributes({ + "gen_ai.usage.input_tokens": usage.prompt_tokens, + "gen_ai.usage.output_tokens": usage.completion_tokens, + "gen_ai.usage.cache_read.input_tokens": getattr(usage.prompt_tokens_details, "cached_tokens", None) or 0, + "gen_ai.usage.reasoning.output_tokens": getattr(usage.completion_tokens_details, "reasoning_tokens", None) or 0, + }) + cost = getattr(usage, "cost", None) # OpenRouter returns USD cost + if cost is not None: + span.set_attribute("gen_ai.usage.cost", cost) + return reply + except Exception as error: + mark_failed(span, error) + raise + + +def run_tool(agent: Agent, call: dict[str, Any]) -> str: + """One tool call = one `execute_tool` span. A failure is marked on the span and returned to the model.""" + name, arguments = call["function"]["name"], call["function"]["arguments"] or "{}" + attributes = { + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.type": "function", + "gen_ai.tool.call.id": call["id"], + "gen_ai.tool.call.arguments": arguments, + } + with tracer.start_as_current_span(f"execute_tool {name}", kind=SpanKind.INTERNAL, attributes=attributes) as span: + try: + result = agent.tools[name].run(**json.loads(arguments)) + # Maple reads a JSON object or array here; a bare string is dropped + output = json.dumps(result if isinstance(result, (dict, list)) else {"result": result}) + span.set_attribute("gen_ai.tool.call.result", output) + return output + except Exception as error: + mark_failed(span, error) + return json.dumps({"error": str(error)}) + + +def run_agent( + agent: Agent, + messages: list[dict[str, Any]], + conversation_id: str | None = None, + on_text: Callable[[str], None] | None = None, +) -> str: + """One agent run = one `invoke_agent` span. For a user turn it is the root of the trace.""" + attributes = {"gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agent.name} + if conversation_id: + attributes["gen_ai.conversation.id"] = conversation_id + if messages: + attributes["gen_ai.input.messages"] = json.dumps([to_semconv(messages[-1])]) + with tracer.start_as_current_span(f"invoke_agent {agent.name}", kind=SpanKind.INTERNAL, attributes=attributes) as span: + try: + for _ in range(10): + reply = chat(agent, messages, on_text) + messages.append(reply) + if not reply.get("tool_calls"): + span.set_attribute("gen_ai.output.messages", json.dumps([to_semconv(reply)])) + return reply["content"] + for call in reply["tool_calls"]: + messages.append({"role": "tool", "tool_call_id": call["id"], "content": run_tool(agent, call)}) + raise RuntimeError("agent exceeded 10 steps") + except Exception as error: + mark_failed(span, error) + raise +``` + +## main.py (chat backend entry point) + +```py +# main.py +from tracing import provider # first import: sets up the provider + +from agent import Agent, Tool, run_agent + +get_weather = Tool( + definition={ + "type": "function", + "function": { + "name": "get_weather", + "description": "Current weather for a city", + "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}, + }, + }, + run=lambda city: {"city": city, "temperature_c": 21, "condition": "partly cloudy"}, +) +assistant = Agent("support", "openai/gpt-4o-mini", "You are a concise assistant.", {"get_weather": get_weather}) + +# One history per conversation. Store it in your database in a real backend. +histories: dict[str, list[dict]] = {} + + +def handle_message(chat_id: str, text: str, on_text=None) -> str: + history = histories.setdefault(chat_id, []) + history.append({"role": "user", "content": text}) + return run_agent(assistant, history, conversation_id=chat_id, on_text=on_text) + + +if __name__ == "__main__": + # In a script: two messages of one conversation, then flush + try: + handle_message("chat_42", "Hi! Briefly introduce yourself.") + handle_message("chat_42", "What's the weather in Berlin?", on_text=lambda d: print(d, end="", flush=True)) + finally: + provider.shutdown() +``` + +## Sub-agent (delegation through a tool) + +```py +weather_worker = Agent("weather_worker", "openai/gpt-4o-mini", "Answer weather questions with get_weather.", {"get_weather": get_weather}) +orchestrator = Agent( + "orchestrator", + "openai/gpt-4o-mini", + "Delegate weather questions to the weather worker.", + { + "ask_weather_worker": Tool( + definition={ + "type": "function", + "function": { + "name": "ask_weather_worker", + "description": "Ask the weather worker a question", + "parameters": {"type": "object", "properties": {"question": {"type": "string"}}, "required": ["question"]}, + }, + }, + # No conversation_id: the sub-agent's spans are in this trace, so it inherits the session + run=lambda question: run_agent(weather_worker, [{"role": "user", "content": question}]), + ) + }, +) +``` + +## Context across threads and async + +- asyncio: `start_as_current_span` context survives `await`; an async variant only needs `AsyncOpenAI` and `async for`. +- Threads (`ThreadPoolExecutor`, `run_in_executor`): context is empty in the worker. Submit `contextvars.copy_context().run, fn, *args` so tool spans stay in the turn's trace. + +## Flush + +- Script/CLI: `provider.shutdown()` in `finally`. +- Lambda/Cloud Functions: `trace.get_tracer_provider().force_flush()` in `finally` of the handler. +- Celery/RQ workers, notebooks: `force_flush()` after each task/cell. + diff --git a/skills/maple-agent-tracing-opentelemetry/references/typescript.md b/skills/maple-agent-tracing-opentelemetry/references/typescript.md new file mode 100644 index 0000000000..b578c53030 --- /dev/null +++ b/skills/maple-agent-tracing-opentelemetry/references/typescript.md @@ -0,0 +1,308 @@ +# TypeScript reference (Node.js 20+) + +Tested pattern: OpenTelemetry JS SDK 2.11 (`@opentelemetry/sdk-trace-node`, `sdk-trace-base`, `resources` 2.11; `exporter-trace-otlp-proto` 0.222; `api` 1.9), `openai` 7.23 against an OpenAI-compatible Chat Completions API (OpenRouter here). + +This is a complete loop. If the project already has a loop, keep its structure and copy only the span code: `invoke_agent` around one agent run, `chat` around each model call, `execute_tool` around each tool call, `toSemconv` for messages. + +```bash +npm install @opentelemetry/api @opentelemetry/sdk-trace-node @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources +``` + +Existing provider (`NodeSDK`, `@vercel/otel`, Sentry): do not create `tracing.ts`; add `new BatchSpanProcessor(new OTLPTraceExporter())` to it. The provider must register an async context manager (`provider.register()` / `NodeSDK.start()` do), or child spans become separate traces. + +## tracing.ts + +```ts +// tracing.ts: import this first in every entry point +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { resourceFromAttributes } from "@opentelemetry/resources" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { NodeTracerProvider } from "@opentelemetry/sdk-trace-node" + +export const provider = new NodeTracerProvider({ + resource: resourceFromAttributes({ + "service.name": "support-agent", + "deployment.environment.name": "production", + }), + // Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS + spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())], +}) +// Registers the global provider and the async context manager, so spans nest across awaits +provider.register() +``` + +## agent.ts + +```ts +// agent.ts +import { type Span, SpanKind, SpanStatusCode, trace } from "@opentelemetry/api" +import OpenAI from "openai" +import type { + ChatCompletionAssistantMessageParam, + ChatCompletionMessageFunctionToolCall, + ChatCompletionMessageParam, + ChatCompletionTool, +} from "openai/resources/chat/completions" + +const tracer = trace.getTracer("support-agent") +const client = new OpenAI({ baseURL: "https://openrouter.ai/api/v1", apiKey: process.env.OPENROUTER_API_KEY }) +const PROVIDER = "openrouter" // gen_ai.provider.name: who you send the request to + +type Message = ChatCompletionMessageParam +type ToolCall = ChatCompletionMessageFunctionToolCall +type Tool = { definition: ChatCompletionTool; run: (args: Record) => unknown } +export type Agent = { name: string; model: string; instructions: string; tools: Record } + +const json = (value: unknown) => JSON.stringify(value) + +// OpenAI message -> GenAI semconv message: { role, parts: [...] } +function toSemconv(message: Message) { + if (message.role === "tool") { + return { role: "tool", parts: [{ type: "tool_call_response", id: message.tool_call_id, response: message.content }] } + } + const parts: object[] = typeof message.content === "string" && message.content ? [{ type: "text", content: message.content }] : [] + if (message.role === "assistant") { + for (const call of message.tool_calls ?? []) { + if (call.type !== "function") continue + parts.push({ type: "tool_call", id: call.id, name: call.function.name, arguments: JSON.parse(call.function.arguments || "{}") }) + } + } + return { role: message.role, parts } +} + +function markFailed(span: Span, error: unknown) { + const err = error instanceof Error ? error : new Error(String(error)) + span.setStatus({ code: SpanStatusCode.ERROR, message: err.message }) + span.setAttribute("error.type", err.name) +} + +// One model call = one `chat` span. Streams, so time to first chunk is recorded too. +async function chat(agent: Agent, messages: Message[], onText?: (delta: string) => void) { + return tracer.startActiveSpan( + `chat ${agent.model}`, + { + kind: SpanKind.CLIENT, + attributes: { + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": PROVIDER, + "gen_ai.request.model": agent.model, + "gen_ai.system_instructions": json([{ type: "text", content: agent.instructions }]), + "gen_ai.input.messages": json(messages.map(toSemconv)), + }, + }, + async (span) => { + try { + const started = performance.now() + const stream = await client.chat.completions.create({ + model: agent.model, + messages: [{ role: "system", content: agent.instructions }, ...messages], + tools: Object.values(agent.tools).map((tool) => tool.definition), + stream: true, + stream_options: { include_usage: true }, // without it, streamed calls report no tokens + }) + let id = "" + let model = agent.model + let text = "" + let finishReason = "stop" + let usage: (OpenAI.CompletionUsage & { cost?: number }) | undefined + const calls: ToolCall[] = [] + for await (const chunk of stream) { + if (!id) span.setAttribute("gen_ai.response.time_to_first_chunk", (performance.now() - started) / 1000) + id = chunk.id + model = chunk.model + if (chunk.usage) usage = chunk.usage + const choice = chunk.choices[0] + if (!choice) continue + if (choice.finish_reason) finishReason = choice.finish_reason + if (choice.delta.content) { + text += choice.delta.content + onText?.(choice.delta.content) + } + for (const delta of choice.delta.tool_calls ?? []) { + const call = (calls[delta.index] ??= { id: "", type: "function", function: { name: "", arguments: "" } }) + if (delta.id) call.id = delta.id + call.function.name += delta.function?.name ?? "" + call.function.arguments += delta.function?.arguments ?? "" + } + } + const reply: ChatCompletionAssistantMessageParam = { role: "assistant", content: text, ...(calls.length ? { tool_calls: calls } : {}) } + span.setAttributes({ + "gen_ai.response.id": id, + "gen_ai.response.model": model, + "gen_ai.response.finish_reasons": [finishReason], + "gen_ai.output.messages": json([{ ...toSemconv(reply), finish_reason: finishReason }]), + }) + if (usage) { + span.setAttributes({ + "gen_ai.usage.input_tokens": usage.prompt_tokens, + "gen_ai.usage.output_tokens": usage.completion_tokens, + "gen_ai.usage.cache_read.input_tokens": usage.prompt_tokens_details?.cached_tokens ?? 0, + "gen_ai.usage.reasoning.output_tokens": usage.completion_tokens_details?.reasoning_tokens ?? 0, + }) + if (usage.cost !== undefined) span.setAttribute("gen_ai.usage.cost", usage.cost) // OpenRouter returns USD cost + } + return reply + } catch (error) { + markFailed(span, error) + throw error + } finally { + span.end() + } + }, + ) +} + +// One tool call = one `execute_tool` span. A failure is marked on the span and returned to the model. +async function runTool(agent: Agent, call: ToolCall) { + const name = call.function.name + return tracer.startActiveSpan( + `execute_tool ${name}`, + { + kind: SpanKind.INTERNAL, + attributes: { + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.type": "function", + "gen_ai.tool.call.id": call.id, + "gen_ai.tool.call.arguments": call.function.arguments || "{}", + }, + }, + async (span) => { + try { + const result = await agent.tools[name]!.run(JSON.parse(call.function.arguments || "{}")) + // Maple reads a JSON object or array here; a bare string is dropped + const output = json(typeof result === "object" && result !== null ? result : { result }) + span.setAttribute("gen_ai.tool.call.result", output) + return output + } catch (error) { + markFailed(span, error) + return json({ error: error instanceof Error ? error.message : String(error) }) + } finally { + span.end() + } + }, + ) +} + +// One agent run = one `invoke_agent` span. For a user turn it is the root of the trace. +export async function runAgent( + agent: Agent, + messages: Message[], + options: { conversationId?: string; onText?: (delta: string) => void } = {}, +): Promise { + const input = messages.at(-1) + return tracer.startActiveSpan( + `invoke_agent ${agent.name}`, + { + kind: SpanKind.INTERNAL, + attributes: { + "gen_ai.operation.name": "invoke_agent", + "gen_ai.agent.name": agent.name, + ...(options.conversationId ? { "gen_ai.conversation.id": options.conversationId } : {}), + ...(input ? { "gen_ai.input.messages": json([toSemconv(input)]) } : {}), + }, + }, + async (span) => { + try { + for (let step = 0; step < 10; step++) { + const reply = await chat(agent, messages, options.onText) + messages.push(reply) + if (!reply.tool_calls?.length) { + span.setAttribute("gen_ai.output.messages", json([toSemconv(reply)])) + return typeof reply.content === "string" ? reply.content : "" + } + for (const call of reply.tool_calls) { + if (call.type !== "function") continue + messages.push({ role: "tool", tool_call_id: call.id, content: await runTool(agent, call) }) + } + } + throw new Error("agent exceeded 10 steps") + } catch (error) { + markFailed(span, error) + throw error + } finally { + span.end() + } + }, + ) +} +``` + +## main.ts (chat backend entry point) + +```ts +// main.ts +import { provider } from "./tracing" // first import: sets up the provider +import type { ChatCompletionMessageParam } from "openai/resources/chat/completions" +import { type Agent, runAgent } from "./agent" + +const assistant: Agent = { + name: "support", + model: "openai/gpt-4o-mini", + instructions: "You are a concise assistant.", + tools: { + get_weather: { + definition: { + type: "function", + function: { + name: "get_weather", + description: "Current weather for a city", + parameters: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, + }, + }, + run: ({ city }) => ({ city, temperature_c: 21, condition: "partly cloudy" }), + }, + }, +} + +// One history per conversation. Store it in your database in a real backend. +const histories = new Map() + +export async function handleMessage(chatId: string, text: string, onText?: (delta: string) => void) { + const history = histories.get(chatId) ?? [] + histories.set(chatId, history) + history.push({ role: "user", content: text }) + return runAgent(assistant, history, { conversationId: chatId, onText }) +} + +// In a script: two messages of one conversation, then flush +try { + await handleMessage("chat_42", "Hi! Briefly introduce yourself.") + await handleMessage("chat_42", "What's the weather in Berlin?", (delta) => process.stdout.write(delta)) +} finally { + await provider.shutdown() +} +``` + +## Sub-agent (delegation through a tool) + +```ts +// weatherWorker is an Agent like `assistant` in main.ts, with get_weather +const orchestrator: Agent = { + name: "orchestrator", + model: "openai/gpt-4o-mini", + instructions: "Delegate weather questions to the weather worker.", + tools: { + ask_weather_worker: { + definition: { + type: "function", + function: { + name: "ask_weather_worker", + description: "Ask the weather worker a question", + parameters: { type: "object", properties: { question: { type: "string" } }, required: ["question"] }, + }, + }, + // No conversationId: the sub-agent's spans are in this trace, so it inherits the session + run: ({ question }) => runAgent(weatherWorker, [{ role: "user", content: String(question) }]), + }, + }, +} +``` + +## Serverless flush + +```ts +// after the handler's work, before returning (inside waitUntil/after() where the platform has one) +await provider.forceFlush() +``` + diff --git a/skills/maple-agent-tracing-provider-sdks/SKILL.md b/skills/maple-agent-tracing-provider-sdks/SKILL.md new file mode 100644 index 0000000000..d0c3ef2fc9 --- /dev/null +++ b/skills/maple-agent-tracing-provider-sdks/SKILL.md @@ -0,0 +1,130 @@ +--- +name: maple-agent-tracing-provider-sdks +description: "Trace agents built directly on the OpenAI, Anthropic or Google Gen AI SDKs (Python or TypeScript, no agent framework) with Maple: official OTel GenAI instrumentations in Python, a small span helper in TypeScript, plus invoke_agent/execute_tool spans and gen_ai.conversation.id so each conversation is one Agent Session. Triggers on 'trace my openai agent', 'add Maple to my anthropic agent', 'agent sessions for gemini', 'OpenTelemetry for the openai sdk'." +--- + +# Maple agent tracing: OpenAI, Anthropic and Gemini SDKs + +Goal: every conversation = one Maple Agent Session. Each user message = one turn = one trace rooted at an `invoke_agent ` span, containing a `chat ` / `generate_content ` span per model call (transcript + tokens) and an `execute_tool ` span per tool call (args, result, failures). + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/provider-sdks + +Mechanism: +- Python: OpenTelemetry GenAI instrumentations (`opentelemetry-instrumentation-genai-openai`, `-genai-anthropic`, `opentelemetry-instrumentation-google-genai`, all >= 1.2b0) write `chat` spans with GenAI semconv on span attributes. +- TypeScript: no usable instrumentation (`@opentelemetry/instrumentation-openai` only patches `openai` < 7 and puts messages in log events; nothing official for `@anthropic-ai/sdk` / `@google/genai`). Record the model call with the helper in `references/typescript.md`. +- Both: YOU add the `invoke_agent` span (with `gen_ai.conversation.id`) and `execute_tool` spans. Instrumentations can't see turns, conversations or your tools. +- Maple files these spans under vendor `unknown:genai` (UI: "Unidentified") and reads `gen_ai.conversation.id` as the session key. Everything else is read in full. + +## Step 0: detect + +1. Confirm there is NO agent framework: `openai-agents`/`@openai/agents`, `langchain*`, `langgraph`, `pydantic-ai*`, `crewai`, `llama-index*`, `ai` (Vercel), `@mastra/core`, `google-adk`, `strands-agents`, `smolagents`, `agno`, `dspy`, `haystack-ai`, `agent-framework`, `litellm`. If one is present, stop and use that framework's skill (`npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing -y` routes). Instrumenting the provider SDK under a framework double-records every call. +2. Language and SDK versions: + - Python >= 3.10. `openai` < 4 (tested 3.20.0), `anthropic` < 2 (tested 1.8.0), `google-genai` < 3 (tested 2.25.0). Outside these ranges the instrumentation won't patch: tell the user. + - TypeScript: `openai` 7.x (tested 7.23.0). Anthropic / Gemini: adapt the helper (see reference). +3. Existing OTel setup. Search for `TracerProvider(`, `set_tracer_provider`, `NodeSDK(`, `NodeTracerProvider`, `registerOTel`, `opentelemetry-instrument`, `logfire.configure`, `sentry_sdk.init` / `Sentry.init`, `Traceloop.init`. + - A provider exists → add a `BatchSpanProcessor(OTLPSpanExporter())` to it; do NOT create a second provider. +4. Other instrumentations of the same SDK → duplicate model-call spans. Look for: `opentelemetry-instrumentation-openai` / `-anthropic` (OpenLLMetry, NOT the official ones), `opentelemetry-instrumentation-openai-v2` (deprecated), `openinference-instrumentation-*`, `@arizeai/openinference-*`, `@traceloop/*`, `logfire.instrument_openai/anthropic`, Sentry OpenAI/Anthropic integrations, `langfuse.openai`. Keep exactly one; ask before removing one that serves something else. +5. Find: every model call site, the agent loop(s), the tool dispatch, where the conversation/thread id lives per request, any agent that calls another agent (sub-agent), any streaming call. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention (`.env`, settings module, secret manager) if it has one. Otherwise inline is acceptable: ingest keys are write-only. + +Env vars (the exporter reads them and appends `/v1/traces`): + +```bash +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY +``` + +## Step 2: install + init + helper + +Read the reference for the service's language and apply it exactly: + +- Python → `references/python.md` (install, `tracing.py`, `agent_tracing.py` helper, loop). +- TypeScript → `references/typescript.md` (install, `instrumentation.ts`, `agent-tracing.ts` helper incl. `tracedChat`, loop, Anthropic/Gemini mapping). + +Rules for both: +- Init runs once at process start, before the first model call (Python: before the first request so the SDK classes are patched). +- Set a real `service.name` (never `unknown_service`). +- Do not set `OTEL_SEMCONV_STABILITY_OPT_IN`; the 1.x GenAI packages don't need it. + +## Step 3: session id (required) + +- Wrap each user turn (the whole model/tool loop) in `agent_span(, conversation_id)` / `agentSpan(...)`. It sets `gen_ai.operation.name=invoke_agent`, `gen_ai.agent.name`, `gen_ai.conversation.id`. +- `conversation_id` = the app's stable conversation/thread/ticket id from the request. Same for every turn of one conversation, different across conversations. No `uuid4()` per request, no process-wide constant, no module-level default. +- No id in the app → ask the user where the conversation boundary is. Single-shot script: one uuid per conversation, reused for all its turns. +- Model and tool calls MUST happen inside the span (same trace). A streamed reply: keep the span open until the stream is consumed (in a web handler, inside the generator that writes the response). +- OpenAI Responses API with `conversation=`: the instrumentation also stamps `gen_ai.conversation.id=conv_...` on the model span. Pass that same id to `agent_span`. +- Do not add `session.id` or `maple_ai.session.id`: Maple reads `gen_ai.conversation.id` for these spans. + +## Step 4: content + +- `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` puts `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.definitions` on span attributes. The helpers read the same variable for tool args/results (and, in TS, messages). +- Values: `NO_CONTENT` (default, empty transcript), `SPAN_ONLY` (use), `EVENT_ONLY` (logs only, Maple shows nothing), `SPAN_AND_EVENT` (also works; duplicates into logs). Legacy `true` = events: wrong. +- User wants content off → leave it unset in that environment; tell them the transcript will be empty but sessions, turns, tools, tokens and failures remain. +- Redaction → in app code before the call, or an OTel Collector `redaction`/`transform` processor. Never lower `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` / `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` (truncated JSON is dropped). + +## Step 5: tools, errors, sub-agents + +1. Route every tool execution through `run_tool(call_id, name, arguments_json, fn)` / `runTool(...)`, passing the model's tool call id. It catches the exception, marks the span failed (status ERROR + `error.type` + recorded exception) and returns `{"error": ...}` to the model. + - If the existing loop already catches tool exceptions, move the catch into `run_tool` (or set status + `error.type` where it catches). A caught, unmarked failure shows as success. + - Anthropic: for each `tool_use` block → `run_tool(block.id, block.name, json.dumps(block.input), fn)`; return `tool_result` blocks. + - Gemini with automatic function calling: the instrumentation records `execute_tool` spans itself; do NOT also wrap tools. With AFC disabled, use `run_tool` with `function_call.id` (or name if id is empty). +2. Sub-agent (an agent run inside a tool): call it via `run_tool`, and inside wrap its loop in `agent_span("")` WITHOUT a conversation id (it's in the caller's trace). Every agent gets a unique `gen_ai.agent.name`; same names merge lanes. +3. Parallel tools: `asyncio.gather` / `Promise.all` keep context. `ThreadPoolExecutor`: submit `contextvars.copy_context().run(fn, ...)`, or tool spans become orphan traces. + +## Step 6: tokens, streaming, cost + +- OpenAI Chat Completions streaming: ALWAYS pass `stream_options={"include_usage": True}` (Python) or use the helper's `onText` path (TS sets it). Otherwise the streamed call has 0 tokens. The final usage chunk has empty `choices`; skip it when reading text (`if chunk.choices`). +- Anthropic / Gemini streams include usage; nothing to add. +- No instrumentation emits cost; Maple doesn't price tokens → sessions show "unpriced". Only if the user asks and the gateway returns cost (OpenRouter `usage.cost`): in the TS helper set `gen_ai.usage.cost` (USD) on the chat span. Don't add pricing tables. + +## Step 7: flush + +- Python: the provider flushes at normal interpreter exit. Add `provider.force_flush()` after each turn in Lambda/Cloud Functions/Cloud Run jobs, notebooks, Celery/RQ tasks, and anything ending in `os._exit`. Scripts/CLIs: `provider.shutdown()` in `finally`. +- TypeScript: `await sdk.shutdown()` before a CLI/script exits. Serverless: `await spanProcessor.forceFlush()` before returning (or in `waitUntil` after a streamed response). + +## Step 8: verify + +Run one real conversation: 3+ user messages with the same id, one tool call, one streamed message if the app streams, one failing tool if you can trigger it, one sub-agent call if the app delegates; then a second conversation with a different id. Flush. In Maple → Agent Sessions (filter by service; allow ~30 s): + +- [ ] Exactly one session per conversation, session id = the id you passed. The second conversation is a separate session. No `trace:` sessions. +- [ ] Framework shows **Unidentified** (expected for this setup). +- [ ] One turn per user message; each turn's trace is rooted at `invoke_agent `, with all `chat`/`generate_content` and `execute_tool` spans inside it. +- [ ] Transcript shows user messages, assistant replies and tool calls (needs `SPAN_ONLY`). +- [ ] Every model call has input and output tokens, INCLUDING the streamed one. +- [ ] Each model call appears once (no duplicate sibling spans from a second instrumentation). +- [ ] Tool spans have the real tool name, call id, arguments and result. +- [ ] The failing tool is marked failed with its message; successful tools are not. +- [ ] Sub-agents appear as separate lanes under their own names, inside the caller's session. +- [ ] Cost shows "unpriced" unless you set `gen_ai.usage.cost`. +- [ ] No attribute contains an API key, `Bearer `, `sk-` or `maple_sk_`. + +Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` and confirm the `invoke_agent` span has `gen_ai.conversation.id` and the model spans have `gen_ai.input.messages` and `gen_ai.usage.*`. + +## Known limitations (tell the user, don't work around) + +- Framework facet shows "Unidentified". +- Python Anthropic with prompt caching: the instrumentation reports `input_tokens` including cache, Maple treats Anthropic input as excluding cache → cache-read tokens counted twice in totals. +- No cost. + +## Do not + +- Do not install `opentelemetry-instrumentation-openai` or `opentelemetry-instrumentation-anthropic` (OpenLLMetry) or the deprecated `opentelemetry-instrumentation-openai-v2`; install the `-genai-` packages. +- Do not use `@opentelemetry/instrumentation-openai` in TS (no `openai` 7 support; content only in logs). +- Do not run two instrumentations on the same SDK, and do not add provider instrumentation under an agent framework. +- Do not create a second `TracerProvider`. +- Do not set content capture to `EVENT_ONLY` or `true`, or rely on log export for content. +- Do not generate a conversation id per request or share one across conversations. +- Do not give a sub-agent the orchestrator's agent name, or a different conversation id. +- Do not catch tool exceptions outside `run_tool` without marking the span failed. +- Do not stream OpenAI Chat Completions without `stream_options.include_usage`. +- Do not print or commit real keys beyond the repo's convention. diff --git a/skills/maple-agent-tracing-provider-sdks/references/python.md b/skills/maple-agent-tracing-provider-sdks/references/python.md new file mode 100644 index 0000000000..31f9216575 --- /dev/null +++ b/skills/maple-agent-tracing-provider-sdks/references/python.md @@ -0,0 +1,147 @@ +# Python: install, init, helper, loop + +Tested: Python 3.10+, `opentelemetry-sdk` 1.45.0, `opentelemetry-exporter-otlp-proto-http` 1.45.0, `opentelemetry-instrumentation-genai-openai` / `-genai-anthropic` / `opentelemetry-instrumentation-google-genai` 1.2b0 (pull `opentelemetry-util-genai` 1.2b0), `openai` 3.20.0, `anthropic` 1.8.0, `google-genai` 2.25.0. + +## Install + +Add with the repo's package manager (uv/poetry/pip). Only the instrumentation(s) for SDKs the service imports: + +```bash +pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ + "opentelemetry-instrumentation-genai-openai>=1.2b0" # openai +# "opentelemetry-instrumentation-genai-anthropic>=1.2b0" # anthropic +# "opentelemetry-instrumentation-google-genai>=1.2b0" # google-genai +``` + +Package names matter: `opentelemetry-instrumentation-openai` / `-anthropic` are OpenLLMetry (Traceloop), not these. + +## tracing.py + +```py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.instrumentation.genai.openai import OpenAIInstrumentor +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + +provider = TracerProvider(resource=Resource.create({"service.name": "support-agent"})) +# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +OpenAIInstrumentor().instrument() +# from opentelemetry.instrumentation.genai.anthropic import AnthropicInstrumentor; AnthropicInstrumentor().instrument() +# from opentelemetry.instrumentation.google_genai import GoogleGenAiSdkInstrumentor; GoogleGenAiSdkInstrumentor().instrument() +``` + +- Import `tracing` first in the entry point (app module, `main.py`, worker). +- Existing provider → don't create one; add the processor to it and just call `.instrument()`. +- Using `opentelemetry-instrument` (zero-code)? It already calls every installed instrumentor; don't call `.instrument()` again, and make sure no other GenAI instrumentation package is installed. + +## agent_tracing.py (copy verbatim, rename the tracer) + +```py +import json +import os +from contextlib import contextmanager + +from opentelemetry import trace +from opentelemetry.trace import Status, StatusCode + +tracer = trace.get_tracer("support-agent") + +# Same switch the instrumentors read, so one env var controls content everywhere. +CAPTURE_CONTENT = os.environ.get( + "OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT", "" +).upper() in ("SPAN_ONLY", "SPAN_AND_EVENT") + + +@contextmanager +def agent_span(agent_name: str, conversation_id: str | None = None): + """One agent run. With a conversation id, it's the turn Maple files under that session.""" + attributes = {"gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agent_name} + if conversation_id: + attributes["gen_ai.conversation.id"] = conversation_id + with tracer.start_as_current_span(f"invoke_agent {agent_name}", attributes=attributes) as span: + yield span + + +def run_tool(call_id: str, name: str, arguments: str, tool) -> str: + """One tool call. A failing tool returns its error to the model, and the span still says it failed.""" + with tracer.start_as_current_span( + f"execute_tool {name}", + attributes={ + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.call.id": call_id, + }, + ) as span: + if CAPTURE_CONTENT: + span.set_attribute("gen_ai.tool.call.arguments", arguments) + try: + result = json.dumps(tool(**json.loads(arguments))) + except Exception as exc: + # The exception never leaves this block, so mark the span failed by hand. + span.record_exception(exc) + span.set_status(Status(StatusCode.ERROR, str(exc))) + span.set_attribute("error.type", type(exc).__qualname__) + result = json.dumps({"error": str(exc)}) + if CAPTURE_CONTENT: + span.set_attribute("gen_ai.tool.call.result", result) + return result +``` + +Async tools: make an `async def run_tool_async` with the same body and `await tool(...)`. + +## Loop (OpenAI Chat Completions) + +```py +def chat_turn(conversation_id: str, history: list, user_text: str) -> str: + with agent_span("support_agent", conversation_id): + history.append({"role": "user", "content": user_text}) + while True: + response = client.chat.completions.create(model=MODEL, messages=history, tools=TOOL_SCHEMAS) + message = response.choices[0].message + history.append(message.model_dump(include={"role", "content", "tool_calls"}, exclude_none=True)) + if not message.tool_calls: + return message.content + for call in message.tool_calls: + result = run_tool(call.id, call.function.name, call.function.arguments, TOOLS[call.function.name]) + history.append({"role": "tool", "tool_call_id": call.id, "content": result}) +``` + +Adapt the existing loop; don't rewrite the app's logic. The tracing is the `with agent_span(...)` and the `run_tool(...)` call. + +Streaming (OpenAI): + +```py +stream = client.chat.completions.create( + model=MODEL, messages=history, stream=True, stream_options={"include_usage": True} +) +reply = "".join(chunk.choices[0].delta.content or "" for chunk in stream if chunk.choices) +``` + +## Anthropic loop shape + +```py +with agent_span("support_agent", conversation_id): + history.append({"role": "user", "content": user_text}) + while True: + response = client.messages.create(model=MODEL, max_tokens=1024, system=SYSTEM, messages=history, tools=TOOL_SCHEMAS) + history.append({"role": "assistant", "content": response.content}) + calls = [b for b in response.content if b.type == "tool_use"] + if not calls: + return "".join(b.text for b in response.content if b.type == "text") + history.append({"role": "user", "content": [ + {"type": "tool_result", "tool_use_id": b.id, "content": run_tool(b.id, b.name, json.dumps(b.input), TOOLS[b.name])} + for b in calls + ]}) +``` + +`client.messages.stream(...)` is instrumented too; usage arrives without extra options. + +## Gemini + +- Automatic function calling (Python functions passed in `tools=`): wrap the turn in `agent_span` only; the instrumentation emits `execute_tool` spans for the SDK-run functions. +- Model spans are named `generate_content `. diff --git a/skills/maple-agent-tracing-provider-sdks/references/typescript.md b/skills/maple-agent-tracing-provider-sdks/references/typescript.md new file mode 100644 index 0000000000..b1065cd426 --- /dev/null +++ b/skills/maple-agent-tracing-provider-sdks/references/typescript.md @@ -0,0 +1,260 @@ +# TypeScript: install, init, helper, loop + +Tested: Node.js 22+ (ESM), `openai` 7.23.0, `@opentelemetry/api` 1.9.1, `@opentelemetry/sdk-node` 0.222.0, `@opentelemetry/exporter-trace-otlp-proto` 0.222.0. + +Why a helper: `@opentelemetry/instrumentation-openai` 0.20 only patches `openai` >=4.19 <7 and writes message content to log events (Maple reads span attributes only). OpenInference JS (`@arizeai/openinference-instrumentation-openai`) supports v7 but Maple shows its transcript as one raw JSON blob. `tracedChat` writes the GenAI attributes Maple reads. + +## Install + +```bash +npm install @opentelemetry/api @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-proto +``` + +## instrumentation.ts (import first in the entry point) + +```ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { NodeSDK, tracing } from "@opentelemetry/sdk-node" + +// The exporter reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. +export const spanProcessor = new tracing.BatchSpanProcessor(new OTLPTraceExporter()) + +export const sdk = new NodeSDK({ serviceName: "support-agent", spanProcessors: [spanProcessor] }) +sdk.start() +``` + +Existing `NodeSDK` / `NodeTracerProvider` / `registerOTel` → don't add another; add a `BatchSpanProcessor(new OTLPTraceExporter())` to it (keep a reference for `forceFlush`). + +## agent-tracing.ts (copy verbatim, rename the tracer) + +```ts +import { type Attributes, type Span, SpanKind, SpanStatusCode, trace } from "@opentelemetry/api" +import type OpenAI from "openai" + +const tracer = trace.getTracer("support-agent") + +// Same switch as the Python instrumentors, so one env var controls content everywhere. +const captureContent = ["SPAN_ONLY", "SPAN_AND_EVENT"].includes( + (process.env.OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT ?? "").toUpperCase(), +) + +/** One agent run. With a conversation id, it's the turn Maple files under that session. */ +export function agentSpan(agentName: string, conversationId: string | undefined, fn: () => Promise) { + const attributes: Attributes = { "gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agentName } + if (conversationId) attributes["gen_ai.conversation.id"] = conversationId + return withSpan(`invoke_agent ${agentName}`, SpanKind.INTERNAL, attributes, () => fn()) +} + +/** One tool call. A failing tool returns its error to the model, and the span still says it failed. */ +export function runTool( + callId: string, + name: string, + args: string, + tool: (args: any) => unknown, +): Promise { + return withSpan( + `execute_tool ${name}`, + SpanKind.INTERNAL, + { + "gen_ai.operation.name": "execute_tool", + "gen_ai.tool.name": name, + "gen_ai.tool.call.id": callId, + }, + async (span) => { + if (captureContent) span.setAttribute("gen_ai.tool.call.arguments", args) + let result: string + try { + result = JSON.stringify(await tool(JSON.parse(args))) + } catch (error) { + markFailed(span, error) + result = JSON.stringify({ error: String(error) }) + } + if (captureContent) span.setAttribute("gen_ai.tool.call.result", result) + return result + }, + ) +} + +type ChatParams = Omit + +/** One model call. Pass `onText` to stream; usage still arrives. */ +export function tracedChat(client: OpenAI, params: ChatParams, onText?: (delta: string) => void) { + return withSpan( + `chat ${params.model}`, + SpanKind.CLIENT, + { + "gen_ai.operation.name": "chat", + "gen_ai.provider.name": "openai", + "gen_ai.request.model": params.model, + }, + async (span) => { + if (captureContent) { + span.setAttribute("gen_ai.input.messages", JSON.stringify(params.messages.map(toGenAiMessage))) + } + let completion: OpenAI.Chat.ChatCompletion + if (onText) { + const started = performance.now() + let firstChunkAt: number | undefined + const stream = client.chat.completions.stream({ + ...params, + // Without this, a streamed call reports no tokens at all. + stream_options: { include_usage: true }, + }) + stream.on("content", (delta) => { + firstChunkAt ??= performance.now() + onText(delta) + }) + completion = await stream.finalChatCompletion() + if (firstChunkAt !== undefined) { + span.setAttribute("gen_ai.response.time_to_first_chunk", (firstChunkAt - started) / 1000) + } + } else { + completion = await client.chat.completions.create(params) + } + + span.setAttributes({ + "gen_ai.response.id": completion.id, + "gen_ai.response.model": completion.model, + "gen_ai.response.finish_reasons": completion.choices.map((c) => c.finish_reason), + }) + if (completion.usage) { + span.setAttributes({ + "gen_ai.usage.input_tokens": completion.usage.prompt_tokens, + "gen_ai.usage.output_tokens": completion.usage.completion_tokens, + "gen_ai.usage.cache_read.input_tokens": completion.usage.prompt_tokens_details?.cached_tokens ?? 0, + "gen_ai.usage.reasoning.output_tokens": + completion.usage.completion_tokens_details?.reasoning_tokens ?? 0, + }) + } + if (captureContent) { + const output = completion.choices.map((c) => ({ + ...toGenAiMessage(c.message), + finish_reason: c.finish_reason, + })) + span.setAttribute("gen_ai.output.messages", JSON.stringify(output)) + } + return completion + }, + ) +} + +// OpenAI chat messages -> the {role, parts} shape Maple renders as a transcript. Text and tool calls only. +function toGenAiMessage(message: OpenAI.Chat.ChatCompletionMessageParam | OpenAI.Chat.ChatCompletionMessage) { + if (message.role === "tool") { + return { + role: "tool", + parts: [{ type: "tool_call_response", id: message.tool_call_id, response: message.content }], + } + } + const parts: object[] = [] + if (typeof message.content === "string" && message.content) { + parts.push({ type: "text", content: message.content }) + } + if (message.role === "assistant") { + for (const call of message.tool_calls ?? []) { + if (call.type === "function") { + parts.push({ type: "tool_call", id: call.id, name: call.function.name, arguments: call.function.arguments }) + } + } + } + return { role: message.role, parts } +} + +function withSpan(name: string, kind: SpanKind, attributes: Attributes, fn: (span: Span) => Promise) { + return tracer.startActiveSpan(name, { kind, attributes }, async (span) => { + try { + return await fn(span) + } catch (error) { + markFailed(span, error) + throw error + } finally { + span.end() + } + }) +} + +function markFailed(span: Span, error: unknown) { + const err = error instanceof Error ? error : new Error(String(error)) + span.recordException(err) + span.setStatus({ code: SpanStatusCode.ERROR, message: err.message }) + span.setAttribute("error.type", err.name) +} +``` + +## Loop (OpenAI Chat Completions) + +```ts +import "./instrumentation.ts" +import OpenAI from "openai" +import { agentSpan, runTool, tracedChat } from "./agent-tracing.ts" + +const client = new OpenAI() +const model = "gpt-4o-mini" + +const tools: Record unknown> = { + get_weather: ({ city }: { city: string }) => ({ city, temperature_c: 21, condition: "partly cloudy" }), +} +const toolSchemas: OpenAI.Chat.ChatCompletionTool[] = [ + { + type: "function", + function: { + name: "get_weather", + parameters: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, + }, + }, +] + +/** One user message in, one reply out. `history` is this conversation's stored messages. */ +export function chatTurn( + conversationId: string, + history: OpenAI.Chat.ChatCompletionMessageParam[], + userText: string, + onText?: (delta: string) => void, +) { + return agentSpan("support_agent", conversationId, async () => { + history.push({ role: "user", content: userText }) + while (true) { + const completion = await tracedChat(client, { model, messages: history, tools: toolSchemas }, onText) + const message = completion.choices[0].message + history.push(message) + if (!message.tool_calls?.length) return message.content ?? "" + for (const call of message.tool_calls) { + if (call.type !== "function") continue + const result = await runTool(call.id, call.function.name, call.function.arguments, tools[call.function.name]) + history.push({ role: "tool", tool_call_id: call.id, content: result }) + } + } + }) +} +``` + +- `onText` switches `tracedChat` to streaming with `stream_options.include_usage` and records time to first chunk. +- Replace every `client.chat.completions.create(...)` in the agent with `tracedChat(client, params)`. Don't wrap `create` twice. +- Sub-agent inside a tool: `runTool(id, "ask_weather_worker", args, (a) => agentSpan("weather_worker", undefined, () => workerLoop(a)))`. + +## Anthropic (`@anthropic-ai/sdk`) or Gemini (`@google/genai`) + +Copy `tracedChat` to `tracedAnthropic` / `tracedGemini`, call the SDK, and set these attributes (raw provider figures; Maple applies the provider's token arithmetic from `gen_ai.provider.name`): + +| Attribute | Anthropic Messages | Gemini `generateContent` | +| --- | --- | --- | +| span name | `chat ` | `generate_content ` | +| `gen_ai.operation.name` | `chat` | `generate_content` | +| `gen_ai.provider.name` | `anthropic` | `gcp.gemini` (`gcp.vertex_ai` on Vertex) | +| `gen_ai.request.model` | request `model` | request `model` | +| `gen_ai.response.id` / `gen_ai.response.model` | `id` / `model` | `responseId` / `modelVersion` | +| `gen_ai.response.finish_reasons` | `[stop_reason]` | `candidates.map(c => c.finishReason)` | +| `gen_ai.usage.input_tokens` | `usage.input_tokens` | `usageMetadata.promptTokenCount` | +| `gen_ai.usage.output_tokens` | `usage.output_tokens` | `usageMetadata.candidatesTokenCount` | +| `gen_ai.usage.cache_read.input_tokens` | `usage.cache_read_input_tokens ?? 0` | `usageMetadata.cachedContentTokenCount ?? 0` | +| `gen_ai.usage.cache_write.input_tokens` | `usage.cache_creation_input_tokens ?? 0` | (omit) | +| `gen_ai.usage.reasoning.output_tokens` | (omit) | `usageMetadata.thoughtsTokenCount ?? 0` | +| `gen_ai.system_instructions` (content on) | `JSON.stringify([{type:"text",content:system}])` | same, from `config.systemInstruction` | + +Messages (content on) as JSON `[{role, parts}]`: +- text block / text part → `{type:"text", content}` +- `tool_use` / `functionCall` → `{type:"tool_call", id, name, arguments}` (`input` / `args`) +- `tool_result` / `functionResponse` → `{type:"tool_call_response", id, response}` +- Gemini role `model` → `assistant`. Output messages also get `finish_reason`. + +Streaming: Anthropic `client.messages.stream(...)` → `await stream.finalMessage()` has usage. Gemini `generateContentStream` → take `usageMetadata` from the last chunk. diff --git a/skills/maple-agent-tracing-pydantic-ai/SKILL.md b/skills/maple-agent-tracing-pydantic-ai/SKILL.md new file mode 100644 index 0000000000..c4cef2ae53 --- /dev/null +++ b/skills/maple-agent-tracing-pydantic-ai/SKILL.md @@ -0,0 +1,202 @@ +--- +name: maple-agent-tracing-pydantic-ai +description: "Trace Pydantic AI agents with Maple: export Pydantic AI's built-in OpenTelemetry spans (with or without Logfire) so each conversation is one Maple Agent Session with transcript, tool calls, sub-agent lanes and tokens. Triggers on 'trace my pydantic ai agent', 'add Maple to pydantic ai', 'agent sessions for pydantic ai', 'OpenTelemetry for pydantic ai'." +--- + +# Maple agent tracing: Pydantic AI + +Goal: every conversation = one Maple Agent Session. Each `agent.run()` = one turn (one trace) with transcript, `chat` spans with tokens, `execute_tool` spans with args/results, failed tools marked failed, sub-agents in their own lanes. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/pydantic-ai + +Mechanism: Pydantic AI's native OTel instrumentation (scope `pydantic-ai`, GenAI semconv on span attributes). No extra instrumentation package. Maple reads `gen_ai.conversation.id` as the session key for this framework. + +## Step 0: detect + +1. Pydantic AI version: `python -c "import pydantic_ai; print(pydantic_ai.__version__)"` (or read `pyproject.toml` / `uv.lock` / `requirements*.txt`). + - Need 2.x (tested 2.51.0). 1.x has no `conversation_id=`: tell the user to upgrade; do not work around it. + - `ToolFailed` needs >= 2.16. +2. Existing OTel setup. Search for `TracerProvider(`, `set_tracer_provider`, `logfire.configure`, `opentelemetry-instrument`, `sentry_sdk.init`, `instrument_all`, `Instrumentation(`, `.instrument =`. + - Logfire already configured → use the Logfire path (Step 2b). + - Another `TracerProvider` exists → add a `BatchSpanProcessor(OTLPSpanExporter())` to it; do NOT create a second provider. + - Nothing → Step 2a. +3. Find: every `agent.run(` / `run_sync(` / `run_stream(` / `iter(` call, where the chat/thread id lives in the request, every `Agent(` construction, and every tool that calls another agent's `run()`. +4. Other instrumentors on the same model client (`logfire.instrument_openai`, `OpenAIInstrumentor`, OpenLLMetry `Traceloop.init`) → they double-trace model calls. Keep Pydantic AI's; ask before removing the others if they serve something else. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention (`.env`, settings module, secret manager) if it has one. Otherwise inline is acceptable: ingest keys are write-only. + +Env vars (the exporter reads them; it appends `/v1/traces`): + +```bash +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +## Step 2a: install + init (plain OpenTelemetry, default) + +Add with the repo's package manager (uv/poetry/pip): + +```bash +pip install "pydantic-ai-slim[openai]>=2.51" "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Keep the project's existing pydantic-ai extras; only add the two OTel packages if pydantic-ai is already installed. + +Create `tracing.py` (adapt service name / environment to the project): + +```py +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.resources import Resource +from opentelemetry.sdk.trace import TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor +from pydantic_ai import Agent, InstrumentationSettings + +provider = TracerProvider( + resource=Resource.create( + {"service.name": "support-agent", "deployment.environment.name": "production"} + ) +) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +Agent.instrument_all( + InstrumentationSettings( + tracer_provider=provider, + include_content=True, + include_binary_content=False, + ) +) +``` + +- Import it first in the entry point (app module, `main.py`, worker). It must run before the first `agent.run()`; agents constructed earlier are still covered. +- Existing provider: skip creating one; add the processor to it and call `Agent.instrument_all(InstrumentationSettings(include_content=True, include_binary_content=False))` (uses the global provider). +- Do not pass `version=`. Default 5 is the tested format; 2-4 are deprecated. +- Set a real `service.name` (never leave `unknown_service`). + +## Step 2b: Logfire path (only if the project already uses Logfire) + +Keep the Step 1 env vars. Logfire adds an OTLP exporter when `OTEL_EXPORTER_OTLP_ENDPOINT` is set. + +Do not install the Step 2a OTel packages: `logfire` already depends on the SDK and the OTLP/HTTP exporter, and logfire 5.1.x pins `opentelemetry-sdk<1.45`, so adding `opentelemetry-sdk>=1.45` makes the install unresolvable. Tested with logfire 5.1.1 (OTel SDK 1.44.0). + +```py +import logfire + +TOOL_CONTENT = {"gen_ai.tool.call.arguments", "gen_ai.tool.call.result", "final_result"} + + +def keep_tool_content(match: logfire.ScrubMatch): + if len(match.path) > 1 and match.path[1] in TOOL_CONTENT: + return match.value + + +logfire.configure( + service_name="support-agent", + environment="production", + send_to_logfire=False, # keep the project's existing value if it also sends to Logfire + scrubbing=logfire.ScrubbingOptions(callback=keep_tool_content), +) +logfire.instrument_pydantic_ai() +``` + +- Default scrubbing replaces tool args/results containing `session`, `auth`, `password`, `cookie`, `secret`, `api key`... with `[Scrubbed due to '']`. `gen_ai.conversation.id` and message attributes are exempt. If the project already has a scrubbing config, merge the callback into it instead of replacing it. +- Do not also call `Agent.instrument_all(...)` with another provider. + +## Step 3: session id (required) + +Pydantic AI resolves `gen_ai.conversation.id` per run as: explicit `conversation_id=` > id on the last message of `message_history` > fresh UUID7. Pass it explicitly on EVERY run call, from the app's chat/thread/conversation id: + +```py +result = await agent.run(text, conversation_id=chat_id, message_history=history) +``` + +```py +async with agent.run_stream(text, conversation_id=chat_id, message_history=history) as run: + async for delta in run.stream_text(delta=True): + yield delta +``` + +- Same for `run_sync(`, `iter(`, `run_stream_events(`, and the resume run that sends `DeferredToolResults`. +- The id must be stable per conversation and unique across conversations. No process-wide constants, no `uuid4()` per request, no module-level default. +- No id available in the app → ask the user where the conversation boundary is; if it's a single-shot script, generate one uuid per conversation (not per run) and reuse it. +- Streaming: keep the `async with` open until the stream is consumed (in FastAPI, inside the generator passed to `StreamingResponse`), or the `invoke_agent` span ends early. + +## Step 4: content + +- Content is on by default (`include_content=True`): `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions` on `chat` spans; tool args/results on `execute_tool` spans. Maple's transcript needs these. +- Always set `include_binary_content=False` (base64 media repeats in every later `chat` span). +- User wants content off → `include_content=False` globally, or per agent: `Agent(..., capabilities=[Instrumentation(settings=InstrumentationSettings(include_content=False))])` (`from pydantic_ai.capabilities import Instrumentation`; omit `tracer_provider` to use the global one). Tell them the transcript keeps roles but no text. +- Logfire scrubbing does not redact message content. For pattern redaction of prompts, recommend an OTel Collector `redaction`/`transform` processor. + +## Step 5: tools, errors, sub-agents + +1. Name every agent: `Agent(..., name="support")`. Unnamed agents become `agent` and share one lane. +2. Tool failures the model should see: `raise ToolFailed("message")` (`from pydantic_ai import ToolFailed`, >= 2.16). Span ERROR, message recorded as `gen_ai.tool.call.result`, run continues, no retry budget used. + - `ModelRetry` also marks the span ERROR (one failed call per retry): use it only for real retries. + - Replace `return {"error": ...}` / `return "Error: ..."` in tools with `raise ToolFailed(...)` only where the user agrees; returned errors show as successful calls. + - Uncaught exceptions mark the tool span and the run ERROR and abort the run. +3. Delegation (a tool that calls another agent): pass the caller's id and usage to EVERY nested run: + +```py +@orchestrator.tool +async def research_weather(ctx: RunContext[None], city: str) -> str: + """Delegate to the weather worker.""" + result = await weather_worker.run( + f"What is the current weather in {city}?", + usage=ctx.usage, + conversation_id=ctx.conversation_id, + ) + return result.output +``` + + Without `conversation_id=ctx.conversation_id` each delegate mints a UUID7, so one trace carries several ids and Maple may file the trace under the wrong session or split the turn. +4. Sequential pipelines of top-level runs (orchestrator run, then summary run): pass the same `conversation_id=` to each; each run is its own trace/turn in the same session. + +## Step 6: flush + +- The SDK flushes at normal interpreter exit. Add an explicit flush where that doesn't happen: + - AWS Lambda / Cloud Functions / Cloud Run jobs: `trace.get_tracer_provider().force_flush()` in a `finally` in the handler. + - Scripts, CLIs, one-shot jobs: `provider.shutdown()` at the end (`finally`). + - Celery/RQ/multiprocessing workers, notebooks: `force_flush()` after each task/cell that runs an agent. + - Logfire: `logfire.force_flush()` / `logfire.shutdown()`. + +## Step 7: verify + +Run one real conversation: 2+ messages with the same id, at least one tool call, one streamed message if the app streams, and a sub-agent call if the app delegates. Then check (Maple → Agent Sessions, filter by the service name; wait ~30 s): + +- [ ] Exactly one session per conversation; session id = the id you passed (not a UUID7 you didn't create). A second conversation is a different session. +- [ ] Framework shows as **Pydantic AI**. +- [ ] One turn per `run()`; transcript shows user prompts, assistant replies and tool calls. +- [ ] Spans: `invoke_agent `, `chat `, `execute_tool `; each `chat`/`execute_tool` is inside its run's `invoke_agent` in the same trace. +- [ ] Every `chat` span has input and output tokens, including the streamed one. +- [ ] Tool calls have their real names, arguments and results. +- [ ] A failing tool is marked failed with its message; successful tools are not. +- [ ] Sub-agents appear as separate lanes with their `name=`, all under the caller's session. +- [ ] Cost shows "unpriced" (expected: Pydantic AI writes `operation.cost`, which Maple doesn't read). +- [ ] No attribute contains an API key or `Bearer ` token. + +Quick local cue: Pydantic AI 2.51 prints an `observability: off` banner on the first run when no instrumentation is set; it disappears once `instrument_all()` (or `logfire.instrument_pydantic_ai()`) has run. + +Local check without Maple: add `ConsoleSpanExporter` via `SimpleSpanProcessor` temporarily and confirm `gen_ai.conversation.id` is identical across runs of one conversation and on delegate spans. + +## Do not + +- Do not rely on the automatic `gen_ai.conversation.id`: it's a new UUID7 per run without history. +- Do not forget `conversation_id=ctx.conversation_id` on delegated runs. +- Do not create a second `TracerProvider` when one exists, and do not add Logfire just for Maple. +- Do not set `version=2|3|4` (deprecated) or `event_mode="logs"` (content moves to logs, which Maple doesn't read). +- Do not add `session.id` or `maple_ai.session.id` attributes: Maple reads `gen_ai.conversation.id` for Pydantic AI, and `maple_ai.session.id` would re-vendor the span. +- Do not set `use_aggregated_usage_attribute_names=False`; the default keeps run totals out of the token sum. +- Do not also instrument the model client (OpenAI/Anthropic instrumentors, `logfire.instrument_openai`): duplicate model-call spans. +- Do not return error strings from tools that failed; raise `ToolFailed`. +- Do not promise cost in Maple; do not add token pricing code. +- Do not print or commit real keys beyond the repo's convention. diff --git a/skills/maple-agent-tracing-smolagents/SKILL.md b/skills/maple-agent-tracing-smolagents/SKILL.md new file mode 100644 index 0000000000..75c5f5ca88 --- /dev/null +++ b/skills/maple-agent-tracing-smolagents/SKILL.md @@ -0,0 +1,180 @@ +--- +name: maple-agent-tracing-smolagents +description: "Trace Hugging Face smolagents agents with Maple: OpenInference instrumentor with GenAI dual-write, one Maple Agent Session per conversation with transcript, model and tool calls, tokens, failed tools and managed-agent lanes. Triggers on 'trace my smolagents agent', 'add Maple to smolagents', 'agent sessions for smolagents', 'OpenTelemetry for smolagents'." +--- + +# Maple agent tracing: smolagents + +Goal: every conversation the app runs through a smolagents agent shows up in Maple **Agent Sessions** as ONE session, with the transcript, each model call (model, tokens), each tool call (name, arguments, result, failure) and one lane per managed agent. Reasoning and background for every step: https://maple.dev/docs/agent-tracing/smolagents + +smolagents has no OpenTelemetry code of its own. All spans come from `openinference-instrumentation-smolagents`. Its defaults are wrong for Maple in three ways this skill fixes: OpenInference-only attributes (Maple's session page reads `gen_ai.*` for smolagents), no session id, and no agent names / wrong tool arguments. + +## Step 0: Detect versions and existing OpenTelemetry + +- Read `pyproject.toml` / `requirements*.txt` / `uv.lock` / `poetry.lock`. Need `smolagents>=1.26` and Python >=3.10. If older, upgrade smolagents first. +- Find which model class the app uses (`OpenAIServerModel`/`OpenAIModel`, `LiteLLMModel`, `InferenceClientModel`, `AzureOpenAIModel`, `AmazonBedrockModel`, `TransformersModel`...). Note any custom `Model` subclass that overrides `generate`: it will produce NO model spans. +- Find every `agent.run(...)` call site and how conversations are identified (chat id, thread id, session row). +- Search for an existing `TracerProvider`, `trace.set_tracer_provider`, `opentelemetry-instrument`, `logfire.configure`, `phoenix.otel.register`, `langfuse`, or `SmolagentsInstrumentor` already present. If a provider exists, REUSE it: add Maple's exporter and the processor below to it, pass it to `instrument()`. Never create a second provider. If `SmolagentsInstrumentor().instrument()` already runs, change that call instead of adding another (a second call is a silent no-op). +- Search for `openinference-instrumentation-openai`, `openinference-instrumentation-litellm`, `OpenAIInstrumentor`, `LiteLLMInstrumentor`. If present only to trace smolagents' model calls, remove them (they double every model span). + +## Step 1: Ingest key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. Protocol: http/protobuf. +- Key given in the prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from **Settings → Ingestion**. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's existing secret/env convention (`.env`, settings module, secret manager) if it has one. Otherwise inlining the key is acceptable: ingest keys are write-only. + +## Step 2: Install and initialize + +```bash +pip install "smolagents[openai]>=1.26" "openinference-instrumentation-smolagents>=0.1.40" \ + "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" +``` + +Use the repo's package manager (`uv add`, `poetry add`...). Use `[litellm]` instead of `[openai]` for `LiteLLMModel`. Do NOT use the `smolagents[telemetry]` extra: it installs the Arize Phoenix server. + +Environment (in the repo's env mechanism): + +```bash +OTEL_SERVICE_NAME= +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name= +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer " +``` + +Base URL only. `OTLPSpanExporter()` with no args appends `/v1/traces`. If you pass `endpoint=` in code, it must end in `/v1/traces`. + +Create `tracing.py` (adapt the module path to the repo layout): + +```py +# tracing.py +import json + +from openinference.instrumentation import TraceConfig +from openinference.instrumentation.smolagents import SmolagentsInstrumentor +from opentelemetry import trace +from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter +from opentelemetry.sdk.trace import SpanProcessor, TracerProvider +from opentelemetry.sdk.trace.export import BatchSpanProcessor + + +class SmolagentsForMaple(SpanProcessor): + """Fixes what the smolagents instrumentor gets wrong for Maple: agent names, run token totals, tool names and arguments.""" + + def on_start(self, span, parent_context=None): + if span.instrumentation_scope.name != "openinference.instrumentation.smolagents": + return + attrs = span.attributes + if span.name.endswith(".run"): + # "weather_worker.run" -> gen_ai.agent.name "weather_worker", so sub-agents get lanes + span.set_attribute("gen_ai.agent.name", span.name.removesuffix(".run")) + # The run's token totals repeat its model calls (with reset=False, every earlier turn's too) + span.set_attribute("gen_ai.usage.input_tokens", 0) + span.set_attribute("gen_ai.usage.output_tokens", 0) + elif "tool.name" in attrs: + # Every @tool span is named "SimpleTool"; name it after the tool instead + span.update_name(f"execute_tool {attrs['tool.name']}") + # The GenAI dual-write copies the tool's input schema here; record the call's arguments + if attrs.get("input.value", "").startswith("{"): + call = json.loads(attrs["input.value"]) + span.set_attribute("gen_ai.tool.call.arguments", json.dumps(call["kwargs"] or call["args"])) + + +provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider.add_span_processor(SmolagentsForMaple()) +provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) +trace.set_tracer_provider(provider) + +SmolagentsInstrumentor().instrument( + tracer_provider=provider, + config=TraceConfig(enable_genai_semconv=True), +) +``` + +- `import tracing` at the top of every entry point (web app module, worker, CLI main) so `instrument()` runs before the first `agent.run()`. Import order relative to `smolagents` does not matter. +- Existing provider: skip the `TracerProvider()`/`set_tracer_provider` lines, add `SmolagentsForMaple()` and the exporter to the existing provider, pass it as `tracer_provider=`. +- `enable_genai_semconv=True` is required: without it Maple's list shows tokens but the session page has no transcript, model or tools. The env var `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` is equivalent only if set before `TraceConfig` is constructed; prefer the code form. +- Keep `SmolagentsForMaple` exactly: it must run in `on_start` (before the dual-write, which never overwrites existing keys). +- The zeroed `gen_ai.usage.*` on `.run` spans is deliberate. The instrumentor copies the agent monitor's totals onto the run span; they repeat the model spans (a `Step N` span sits in between, so Maple can't net them) and with `reset=False` they accumulate across turns. Without the zeros Maple counts tokens two or more times. `llm.token_count.*` keeps the totals for other tools. + +## Step 3: One session per conversation + +smolagents has no conversation id; each `agent.run()` is its own trace. Maple reads `session.id` for smolagents (NOT `gen_ai.conversation.id`), and the instrumentor sets it only inside `using_session`: + +```py +from openinference.instrumentation import using_session +from smolagents import OpenAIServerModel, ToolCallingAgent + +agents: dict[str, ToolCallingAgent] = {} + + +def handle_message(conversation_id: str, text: str) -> str: + agent = agents.get(conversation_id) + if agent is None: + agent = agents[conversation_id] = ToolCallingAgent( + tools=[get_weather, calculate], + model=OpenAIServerModel(model_id="gpt-4o-mini"), + name="assistant", + ) + with using_session(conversation_id): + return str(agent.run(text, reset=False)) +``` + +- Wrap EVERY `agent.run()` call site in `with using_session():`. Use the app's stored conversation/chat/thread id. Never a fresh UUID per request, never a constant. +- If the app already tracks a user, `using_attributes(session_id=..., user_id=...)` also works. +- With `agent.run(..., stream=True)`, iterate the generator INSIDE the `with` block. +- One agent object (or one persisted memory) per conversation. A module-level agent with `reset=False` merges every user's memory and traces into one conversation; fix it if you find it, and tell the user. +- The id is a contextvar, not baggage. Your own spans (plain OTel tracer) don't get it; give them `attributes=dict(get_attributes_from_context())` if you add any. + +## Step 4: Content + +- On by default: model spans carry the full input message list (system prompt, task, history, tool results) and the reply; tool spans carry arguments and results. Leave it on unless the user or repo says prompts are sensitive. +- To turn off: `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` (or `OPENINFERENCE_HIDE_INPUTS=true` / `OPENINFERENCE_HIDE_OUTPUTS=true`). Narrower: `hide_input_text`, `hide_output_text`, `hide_llm_invocation_parameters`. +- The hide switches do NOT mask the run span's `smolagents.task` attribute (holds the previous run's task text). If PII matters, drop it in a Collector (`attributes` processor) and tell the user. +- Payload: default system prompt is ~3.1k chars (`ToolCallingAgent`) / ~8.5k (`CodeAgent`) and with `reset=False` each model span repeats the whole history. Ingest limit is 20 MiB per request; nothing to configure. + +## Step 5: Tools, errors, managed agents + +- Give every agent (top-level and managed) a distinct `name=`. Unnamed agents become `ToolCallingAgent`/`CodeAgent` and share one lane. +- Tool exceptions are marked failed automatically (tool span status ERROR with the exception message). Do not catch exceptions inside tools just to return an error string: that hides the failure (span stays OK). +- smolagents prompts the model to retry after every tool error until `max_steps`, then asks for an answer without tools (hallucination risk). If a tool can fail permanently, suggest a `step_callbacks` hook that tells the model to stop after the first failure; don't add it unasked. +- Managed agents appear as `.run` spans under the manager's `Step N` span; the processor names their lanes. Parallel tool/agent calls (`max_tool_threads`) keep context; nothing to do. +- `CodeAgent` with `executor_type` other than local (`e2b`, `docker`, `modal`, ...) runs tools outside the process: no tool spans. Tell the user; nothing to fix in tracing. +- Known, not fixable here: no `gen_ai.tool.call.id` on tool spans; a failed tool also marks its `Step N` span ERROR. + +## Step 6: Flush + +- `TracerProvider` flushes on normal interpreter exit (atexit). That covers servers and CLIs that exit normally. +- Serverless handlers (Lambda, Cloud Run jobs, Modal functions...): call `provider.force_flush()` in a `finally` before returning. +- Scripts that may be killed or call `os._exit`, and notebooks: `provider.force_flush()` after each run; `provider.shutdown()` at the very end. + +## Step 7: Verify + +Run one real conversation (2-3 messages, same conversation id, at least one tool call), and one message in a second conversation. If the user gave no key (`MAPLE_TEST`), you can't see results in Maple; say so and list what they should check. Otherwise check in Maple **Agent Sessions** (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), filtered to the service name: + +- Exactly one session per conversation id (two here), not one per message. Framework shows **smolagents**. +- The first conversation has one turn per `agent.run()`; each turn's root span is `.run`. +- Transcript is non-empty (user text appears after `New task:`; tool results as `tool-response` messages). +- Model calls `OpenAIModel.generate` (or `.generate[_stream]`) have a model and non-zero input/output tokens, including streamed calls. +- Session token totals equal the sum of the model calls (not 2x, not growing faster each turn). +- Tool calls are `execute_tool ` with JSON arguments (the actual call's, not a schema) and results. `execute_tool final_answer` at the end of each turn is expected. +- A tool that raised is counted as failed; tools that returned are not. +- Managed agents (if any) each have their own lane named after the agent. +- Cost shows as unpriced (smolagents records no cost). Expected. +- Each model call appears once (no nested duplicate model span). + +If sessions are split per message: `using_session` missing or id changing. Transcript empty: `enable_genai_semconv` not applied. Nothing arrives: exporter endpoint/header wrong, or process exited without flushing. + +## Do not + +- Do not use the `smolagents[telemetry]` extra or `phoenix.otel.register()` to send to Maple. +- Do not add `openinference-instrumentation-openai` / `-litellm` alongside (duplicate model spans, doubled tokens). +- Do not set `gen_ai.conversation.id` manually expecting grouping: Maple reads `session.id` for smolagents. +- Do not generate a session id per request or use a constant one. +- Do not pass `OTLPSpanExporter(endpoint="https://ingest.maple.dev")` without `/v1/traces`. +- Do not create a second `TracerProvider` when one exists, and do not call `instrument()` twice. +- Do not drop the zero-token lines for `.run` spans from `SmolagentsForMaple`. +- Do not override `generate` in a custom `Model` subclass and expect model spans. +- Do not rely on `hide_inputs` alone for PII (`smolagents.task`). +- Do not wrap tools in try/except that returns error strings; let them raise. diff --git a/skills/maple-agent-tracing-spring-ai/SKILL.md b/skills/maple-agent-tracing-spring-ai/SKILL.md new file mode 100644 index 0000000000..564b6eb6f3 --- /dev/null +++ b/skills/maple-agent-tracing-spring-ai/SKILL.md @@ -0,0 +1,262 @@ +--- +name: maple-agent-tracing-spring-ai +description: "Trace Spring AI agents with Maple: wires Spring Boot's OpenTelemetry starter to Maple, samples every turn, and adds one configuration class so each ChatClient conversation is one Maple Agent Session with transcript, tool calls (failures marked), sub-agent lanes and tokens. Triggers on 'trace my spring ai agent', 'add Maple to spring ai', 'agent sessions for spring ai', 'OpenTelemetry for spring ai'." +--- + +# Maple agent tracing for Spring AI + +Goal: every conversation with the Spring AI app shows up in Maple **Agent Sessions** as exactly one session, one turn per `ChatClient` call, with transcript, model calls, tool calls (failures marked), sub-agent lanes and tokens. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/spring-ai + +Mechanism: Spring AI's Micrometer Observations → `micrometer-tracing-bridge-otel` → OpenTelemetry SDK → OTLP/HTTP to Maple, all from `spring-boot-starter-opentelemetry`. Maple detects the spans as Spring AI by their `spring.ai.*` keys. Out of the box: sampling is 10%, prompts/replies never reach spans (`log-prompt`/`log-completion` only log to SLF4J), thrown tool errors end the span OK, and advisor spans inflate call counts. Steps 2-5 fix all four. + +## Step 0: Detect versions and existing setup + +1. Read `pom.xml` / `build.gradle(.kts)`: Spring Boot version, `spring-ai-bom` version, model starter (`spring-ai-starter-model-*`). Target Spring AI 2.0.x (verified 2.0.1) on Boot 4.x (verified 4.1.1), Java 17+. Spring AI 1.1.x on Boot 3.5: see Step 2d. +2. Grep for existing tracing: `micrometer-tracing-bridge`, `spring-boot-starter-opentelemetry`, `opentelemetry-exporter-otlp`, `management.otlp`, `management.opentelemetry`, `management.tracing`, `-javaagent`, `opentelemetry-javaagent`, `OTEL_EXPORTER_OTLP`, `ObservationFilter`, `ObservationPredicate`, `ToolExecutionExceptionProcessor`. + - Existing Boot tracing to another backend: Boot has one OTLP span exporter. Ask the user whether to repoint it at Maple; to keep both, add a second exporter via an `OtlpHttpSpanExporter` bean only if they insist. + - OTel Java agent attached (`-javaagent:...opentelemetry-javaagent.jar`): see Step 2c. + - Existing `ToolExecutionExceptionProcessor` bean: patch it (Step 5) instead of adding the one in Step 3. +3. Find every `ChatClient` call site (`.prompt(`, `.call()`, `.stream()`) and where the conversation/thread id lives in the request. Find every `ChatClient.Builder` (each role/sub-agent). + +## Step 1: Key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. Protocol: OTLP/HTTP protobuf (Boot's default transport). +- Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from **Settings → Ingestion**. +- Private `maple_sk_` keys never go in browser code. Spring AI runs server-side; an ingest key is write-only. +- Follow the repo's existing secret/env convention (`${ENV_VAR}` placeholders, profile files, Vault/Config Server). If there is none, inline in `application.properties` is acceptable because ingest keys are write-only. + +## Step 2: Install and export + +### 2a. Dependencies (Boot 4, Spring AI 2.0) + +Keep the existing `spring-ai-bom` import and model starter. Add: + +```xml + + org.springframework.boot + spring-boot-starter-opentelemetry + + + org.springframework.boot + spring-boot-starter-actuator + +``` + +Gradle: `implementation("org.springframework.boot:spring-boot-starter-opentelemetry")` and `...:spring-boot-starter-actuator`. + +### 2b. Properties + +Add to `application.properties` (or the YAML equivalent): + +```properties +spring.application.name= + +management.opentelemetry.tracing.export.otlp.endpoint=https://ingest.maple.dev/v1/traces +management.opentelemetry.tracing.export.otlp.headers.Authorization=Bearer ${MAPLE_INGEST_KEY} +management.tracing.sampling.probability=1.0 +management.opentelemetry.resource-attributes.deployment.environment.name= + +management.otlp.metrics.export.url=https://ingest.maple.dev/v1/metrics +management.otlp.metrics.export.headers.Authorization=Bearer ${MAPLE_INGEST_KEY} + +maple.ai.capture-content=true +spring.ai.openai.chat.stream-options.include-usage=true +``` + +- The endpoint property takes the FULL URL including `/v1/traces` (Boot does not append it). +- `management.tracing.sampling.probability=1.0` is REQUIRED. Default is `0.1`: 90% of turns silently missing. +- The starter also exports metrics, to `localhost:4318` by default. Either point them at Maple (above) or set `management.otlp.metrics.export.enabled=false`. Never leave the default. +- `include-usage` only for the OpenAI starter (and OpenAI-compatible gateways): without it streamed calls report no tokens. Per call alternative: `OpenAiChatOptions.builder().streamUsage(true)`. +- Boot 4.1+ also maps `OTEL_EXPORTER_OTLP_ENDPOINT` (base URL, Boot appends `/v1/traces`), `OTEL_EXPORTER_OTLP_HEADERS`, `OTEL_SERVICE_NAME`, `OTEL_RESOURCE_ATTRIBUTES`. Use them if the repo configures via env; keep the sampling property. +- Do NOT set `management.opentelemetry.tracing.limits.max-attribute-value-length`: truncated JSON content no longer parses and Maple drops it. +- Leave `spring.ai.chat.observations.log-prompt/log-completion` and `spring.ai.chat.client.observations.*` as they are. They only log; they never put content on spans. +- `@SpringBootTest` disables tracing export; add `@AutoConfigureTracing` only if the user wants spans from tests. + +### 2c. OTel Java agent present + +The agent does not convert Micrometer Observations into spans, so Steps 2a-2b are still required. With both: +- add `-Dotel.instrumentation.openai-java.enabled=false` (the agent instruments the OpenAI Java SDK under Spring AI 2.0's OpenAI starter: duplicate `chat` spans); +- add `management.observations.enable.http.server.requests=false` (duplicate server spans); +- point the agent's `OTEL_EXPORTER_OTLP_*` at Maple too. +Tell the user this combination is untested by Maple; recommend removing the agent for this service if nothing else needs it. + +### 2d. Spring AI 1.1 on Boot 3.5 + +- Deps: `io.micrometer:micrometer-tracing-bridge-otel` + `io.opentelemetry:opentelemetry-exporter-otlp` + actuator (no `spring-boot-starter-opentelemetry`). +- Properties: `management.otlp.tracing.endpoint=https://ingest.maple.dev/v1/traces`, `management.otlp.tracing.headers.Authorization=Bearer ...`; sampling property unchanged. Stream usage: `spring.ai.openai.chat.options.stream-usage=true`. +- Step 3 class: replace `tools.jackson.databind.json.JsonMapper.shared().writeValueAsString(...)` with a Jackson 2 `com.fasterxml.jackson.databind.ObjectMapper` (wrap the checked `JsonProcessingException`), and delete the two `getToolCallId()` lines (not in 1.1). + +## Step 3: Add the configuration class + +Create it in a package the app's `@SpringBootApplication` scans (same package or below). Copy verbatim; change only the package: + +```java +package com.example.agent; + +import java.util.ArrayList; +import java.util.List; +import java.util.Map; + +import io.micrometer.common.KeyValue; +import io.micrometer.observation.Observation; +import io.micrometer.observation.ObservationFilter; +import io.micrometer.observation.ObservationPredicate; +import io.micrometer.observation.ObservationRegistry; +import tools.jackson.databind.json.JsonMapper; + +import org.springframework.ai.chat.client.advisor.observation.AdvisorObservationContext; +import org.springframework.ai.chat.client.observation.ChatClientObservationContext; +import org.springframework.ai.chat.messages.AssistantMessage; +import org.springframework.ai.chat.messages.Message; +import org.springframework.ai.chat.messages.ToolResponseMessage; +import org.springframework.ai.chat.observation.ChatModelObservationContext; +import org.springframework.ai.tool.execution.DefaultToolExecutionExceptionProcessor; +import org.springframework.ai.tool.execution.ToolExecutionExceptionProcessor; +import org.springframework.ai.tool.observation.ToolCallingObservationContext; +import org.springframework.beans.factory.annotation.Value; +import org.springframework.context.annotation.Bean; +import org.springframework.context.annotation.Configuration; + +/** Adds the gen_ai.* attributes Maple reads on top of Spring AI's own observations. */ +@Configuration(proxyBeanMethods = false) +public class MapleAiObservationConfig { + + /** Agent name for a ChatClient: `.defaultAdvisors(a -> a.param(AGENT_NAME, "support_agent"))`. */ + public static final String AGENT_NAME = "gen_ai.agent.name"; + + // Advisor spans carry nothing Maple reads, and their names ("tool _calling ", + // "message_chat_memory") would be counted as extra tool and LLM calls. + @Bean + ObservationPredicate skipAdvisorObservations() { + return (name, context) -> !(context instanceof AdvisorObservationContext); + } + + @Bean + ObservationFilter mapleGenAiAttributes(@Value("${maple.ai.capture-content:false}") boolean captureContent) { + return context -> { + if (context instanceof ChatClientObservationContext client) { + // One ChatClient call is one agent turn; Spring AI labels it "framework". + client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.operation.name", "invoke_agent")); + if (client.getRequest().context().get(AGENT_NAME) instanceof String agent) { + client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.agent.name", agent)); + } + } + else if (context instanceof ToolCallingObservationContext tool) { + tool.addLowCardinalityKeyValue(KeyValue.of("gen_ai.tool.name", tool.getToolDefinition().name())); + if (tool.getToolCallId() != null) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.id", tool.getToolCallId())); + } + if (captureContent) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.arguments", tool.getToolCallArguments())); + if (tool.getToolCallResult() != null) { + tool.addHighCardinalityKeyValue(KeyValue.of("gen_ai.tool.call.result", tool.getToolCallResult())); + } + } + } + else if (captureContent && context instanceof ChatModelObservationContext chat) { + chat.addHighCardinalityKeyValue(KeyValue.of("gen_ai.input.messages", + JsonMapper.shared().writeValueAsString(chat.getRequest().getInstructions().stream().map(MapleAiObservationConfig::message).toList()))); + if (chat.getResponse() != null) { + chat.addHighCardinalityKeyValue(KeyValue.of("gen_ai.output.messages", + JsonMapper.shared().writeValueAsString(chat.getResponse().getResults().stream().map(g -> message(g.getOutput())).toList()))); + } + } + return context; + }; + } + + // Spring AI hands a failing tool's message back to the model and ends the tool span OK. + // Mark the span failed first, then keep the default behaviour. + @Bean + ToolExecutionExceptionProcessor toolExecutionExceptionProcessor(ObservationRegistry registry) { + ToolExecutionExceptionProcessor fallback = DefaultToolExecutionExceptionProcessor.builder().build(); + return exception -> { + Observation toolCall = registry.getCurrentObservation(); + if (toolCall != null) { + toolCall.error(exception); + } + return fallback.process(exception); + }; + } + + /** One message in the OpenTelemetry GenAI shape: {role, parts: [...]}. */ + private static Map message(Message message) { + List> parts = new ArrayList<>(); + if (message instanceof ToolResponseMessage toolResponse) { + toolResponse.getResponses().forEach(r -> parts.add( + Map.of("type", "tool_call_response", "id", r.id(), "response", r.responseData()))); + } + else { + if (message.getText() != null && !message.getText().isEmpty()) { + parts.add(Map.of("type", "text", "content", message.getText())); + } + if (message instanceof AssistantMessage assistant) { + assistant.getToolCalls().forEach(call -> parts.add( + Map.of("type", "tool_call", "id", call.id(), "name", call.name(), "arguments", call.arguments()))); + } + } + return Map.of("role", message.getMessageType().getValue(), "parts", parts); + } + +} +``` + +Kotlin project: translate one to one (e.g. `ObservationPredicate { _, context -> context !is AdvisorObservationContext }`), same beans, same keys. + +## Step 4: Session id (one conversation = one session) + +Maple reads ONLY `spring.ai.chat.client.conversation.id` for Spring AI, on the `chat_client` span. Spring AI sets it from the `ChatMemory.CONVERSATION_ID` advisor param. + +- On EVERY top-level call: `chatClient.prompt().user(msg).advisors(a -> a.param(ChatMemory.CONVERSATION_ID, conversationId))...`. Use the id the app already stores for the chat/thread. +- If the app already passes this param for `MessageChatMemoryAdvisor`/`PromptChatMemoryAdvisor`/`VectorStoreChatMemoryAdvisor`, nothing to add. If it has no chat memory, add the param anyway; it works without a memory advisor. +- Never set the conversation id via `defaultAdvisors(...)` on a shared builder/client (constant id = all users in one session). Never mint a UUID per request. +- Sub-agent `ChatClient` calls inside tools need no id: Maple groups the whole trace by any span carrying it. Don't pass a different id to sub-agents. +- Do not add `gen_ai.conversation.id` or `session.id`; Maple ignores them on Spring AI spans. +- Give every `ChatClient` an agent name: `builder.defaultAdvisors(a -> a.param(MapleAiObservationConfig.AGENT_NAME, ""))`. + +## Step 5: Tools, errors, sub-agents + +- Tool spans are `execute_tool `; Step 3 adds `gen_ai.tool.name`, `gen_ai.tool.call.id` and (with content on) arguments/result. +- Failures: Step 3's `ToolExecutionExceptionProcessor` marks the tool span ERROR (exception message + `exception` event) and still returns the message to the model. If the app has its own processor bean, do not add a second one: insert `Observation current = registry.getCurrentObservation(); if (current != null) current.error(exception);` at the top of its `process(...)`. If the app relied on `spring.ai.tools.throw-exception-on-error=true`, build the fallback with `.alwaysThrow(true)` (the property no longer applies once the bean is replaced). +- Tools that return error strings instead of throwing count as success. Point it out; don't change behavior unasked. +- Sub-agents: one `ChatClient` per role, exposed to the orchestrator as a `@Tool(name = "", description = ...)` method that calls it. Build each from `builder.clone()` with its own `AGENT_NAME` param. Maple shows `execute_tool ` → worker `chat_client` as a delegation lane. +- Own thread pools / `CompletableFuture` fan-out: propagate the observation to worker threads (Micrometer `context-propagation`: `ContextSnapshot`, a wrapped executor, or build the worker observation with `.parentObservation(parent)`). Otherwise each worker starts a new trace and becomes its own `trace:` session. +- Spring AI 2.0 has no native tool-approval/HITL mechanism; don't invent one. + +## Step 6: Flush + +- Web app: nothing. Boot shuts down the `SdkTracerProvider` on context close, which flushes. +- `CommandLineRunner` / batch: let `run()` return or use `SpringApplication.exit(context)`. `System.exit()` is fine (shutdown hook); `Runtime.halt()` / SIGKILL lose the last batch. +- Serverless (Spring Cloud Function on Lambda etc.): inject `io.opentelemetry.sdk.trace.SdkTracerProvider` and call `tracerProvider.forceFlush().join(10, TimeUnit.SECONDS)` at the end of EVERY invocation; never shut it down. +- Batch delay is 5 s; wait before checking Maple. + +## Step 7: Verify + +Run one real conversation: 2-3 turns with the same conversation id including one tool call (one streamed turn if the app streams), plus a second conversation with a different id. Exit cleanly. Wait ~1 minute. In Maple **Agent Sessions**, filtered by the service name (or via the Maple MCP `list_agent_sessions` + `get_agent_session`), check: + +- [ ] Exactly one session per conversation id; none named `trace:` (that means a top-level call lacked the `CONVERSATION_ID` param, or a sub-agent ran on a thread without context). +- [ ] The two conversations are two different sessions. +- [ ] Framework shows **Spring AI**. +- [ ] Every turn arrived (count = number of top-level `ChatClient` calls). Missing turns = sampling property not applied. +- [ ] Spans: `spring_ai chat_client` (agent, with your agent name), `chat `, `execute_tool `. NO `tool _calling `, `call`, `message_chat_memory` spans (else the predicate bean isn't loaded). +- [ ] LLM call count = number of `chat ` spans (not doubled); tool call count = number of real tool invocations. +- [ ] Transcript shows user messages, replies, tool calls with arguments and results (else `maple.ai.capture-content` isn't `true` or the class isn't scanned). +- [ ] Input and output tokens on every model call, including the streamed one. +- [ ] A tool that threw is counted as failed, with its message; successful tools are not. +- [ ] Sub-agents appear as lanes under their agent names. +- [ ] Cost shows as unpriced (Spring AI emits no cost; expected). +- [ ] App logs have no `Failed to publish metrics` / OTLP export errors, no `401`. + +## Do not + +- Do not leave `management.tracing.sampling.probability` at the default `0.1`. +- Do not rely on `log-prompt`/`log-completion`/`include-content` for the transcript; Maple reads `gen_ai.input.messages`/`gen_ai.output.messages`/`gen_ai.tool.call.*` span attributes only. +- Do not stamp `maple_ai.session.id` on Spring AI spans; the conversation id param is the supported path. +- Do not put a constant or per-request conversation id on calls. +- Do not register a second `ToolExecutionExceptionProcessor` next to an existing one (ambiguous bean). +- Do not run the OTel Java agent's OpenAI instrumentation alongside Spring AI (duplicate `chat` spans). +- Do not set an attribute length limit (breaks content JSON). +- Do not use this skill for LangChain4j; use the generic OpenTelemetry GenAI guide: https://maple.dev/docs/agent-tracing/opentelemetry diff --git a/skills/maple-agent-tracing-strands/SKILL.md b/skills/maple-agent-tracing-strands/SKILL.md new file mode 100644 index 0000000000..1d541f7d65 --- /dev/null +++ b/skills/maple-agent-tracing-strands/SKILL.md @@ -0,0 +1,183 @@ +--- +name: maple-agent-tracing-strands +description: "Trace Strands Agents (AWS, Python or TypeScript) with Maple: export Strands' built-in OpenTelemetry spans with messages on span attributes, a session id per conversation and per-call token counts, so each conversation is one Maple Agent Session with transcript, tool calls, sub-agent lanes and tokens. Triggers on 'trace my strands agent', 'add Maple to strands', 'agent sessions for strands', 'OpenTelemetry for strands agents'." +--- + +# Maple agent tracing: Strands Agents + +Goal: every conversation = one Maple Agent Session. Each `agent(...)` / `invoke_async` / `stream_async` call = one turn (one trace) with transcript, `chat` spans with tokens, `execute_tool` spans with args/results, failed tools marked failed, sub-agents in their own lanes. + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/strands + +Mechanism: Strands' native OTel tracer (scope `strands.telemetry.tracer`, `gen_ai.provider.name=strands-agents`). No extra instrumentation package. Maple reads `session.id` as the session key for Python Strands, and reads span ATTRIBUTES only (never span events). + +## Step 0: detect + +1. Language and version. + - Python: `python -c "from importlib.metadata import version; print(version('strands-agents'))"` or read `pyproject.toml` / `uv.lock` / `requirements*.txt`. Need >= 1.54 (tested 1.57.1): span-attribute content needs 1.48, tool args/results 1.51, Maple-readable cache token names 1.54. Older → upgrade; do not work around it. + - TypeScript: `@strands-agents/sdk` in `package.json` (tested 1.19.0). Follow Step 2 TS. +2. Existing OTel setup. Search for `StrandsTelemetry(`, `TracerProvider(`, `set_tracer_provider`, `opentelemetry-instrument`, `aws-opentelemetry-distro`, `logfire.configure`, `sentry_sdk.init`, `setupTracer(`, `NodeSDK(`, `NodeTracerProvider(`. + - `StrandsTelemetry().setup_otlp_exporter()` already present → reuse it; only change env vars. + - Another global provider exists (web framework, `opentelemetry-instrument`, ADOT on AgentCore) → do NOT call `StrandsTelemetry()`. Add `BatchSpanProcessor(OTLPSpanExporter(endpoint=".../v1/traces", headers={...}))` to that provider. Strands uses the global provider automatically. + - Nothing → Step 2. +3. Find: every `Agent(` construction (and whether it is module-level/shared), every place the agent is invoked, where the chat/thread/conversation id lives in the request, every `session_manager=`, every `.as_tool(`, `Swarm(`, `GraphBuilder(` / `Graph(`. +4. Other instrumentors on the same model calls (OpenLIT, OpenLLMetry `Traceloop.init`, OpenInference, `opentelemetry-instrumentation-openai*`, botocore/Bedrock GenAI instrumentation) → they double-trace model calls. Keep Strands' spans; ask before removing the others if they serve something else. + +## Step 1: key and region + +- US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. +- Header: `Authorization=Bearer `. +- Key given in the prompt → use it. +- No key → use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Never put a private `maple_sk_` key in browser code. +- Follow the repo's secret/env convention (`.env`, settings module, secret manager, container env) if it has one. Otherwise inline is acceptable: ingest keys are write-only. + +## Step 2: install + init + +Python. Add with the repo's package manager, keeping existing extras (`openai`, `anthropic`, `litellm`, ...): + +```bash +pip install 'strands-agents[otel]>=1.57' +``` + +Env vars. Put them where the repo keeps env (shell/.env/container). `OTEL_SEMCONV_STABILITY_OPT_IN` MUST be in the process environment before the first `Agent(` is constructed (Strands reads it once, into a singleton). Setting it via `os.environ` is only acceptable at the very top of the entry point, before any strands import. + +```bash +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +OTEL_SERVICE_NAME= +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name= +OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental,gen_ai_span_attributes_only,gen_ai_use_latest_invocation_tokens +``` + +- Endpoint is the base URL; the exporter appends `/v1/traces`. +- `gen_ai_latest_experimental`: `{role, parts}` messages, `gen_ai.system_instructions`, tool args/results. +- `gen_ai_span_attributes_only`: messages as span attributes. Without it the Maple transcript is EMPTY. +- `gen_ai_use_latest_invocation_tokens`: `invoke_agent` usage = this call only, not the agent's lifetime total. +- If the repo already has an `OTEL_SEMCONV_STABILITY_OPT_IN` value, merge tokens (comma-separated), don't replace. + +Init once, imported from the entry point before any agent runs (skip if Step 0 found an existing provider): + +```py +# telemetry.py +from strands.telemetry import StrandsTelemetry + +telemetry = StrandsTelemetry().setup_otlp_exporter() +``` + +TypeScript. OTel packages are optional peers; install them: + +```bash +npm install @strands-agents/sdk @opentelemetry/api @opentelemetry/sdk-trace-base @opentelemetry/sdk-trace-node @opentelemetry/resources @opentelemetry/exporter-trace-otlp-http @opentelemetry/sdk-metrics @opentelemetry/exporter-metrics-otlp-http +``` + +Same env vars, but `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental,gen_ai_span_attributes_only` (TS has no `gen_ai_use_latest_invocation_tokens`). The TS exporter sends OTLP/HTTP JSON; that's fine. + +```ts +import { setupTracer } from "@strands-agents/sdk/telemetry" + +export const provider = setupTracer({ exporters: { otlp: true } }) // before the first Agent; reads OTEL_EXPORTER_OTLP_* +``` + +## Step 3: session id + +Python: pass the conversation id as `session.id` in `trace_attributes` on the agent that handles the turn. `gen_ai.conversation.id` is ignored for Python Strands; don't rely on it. + +```py +agent = Agent( + name="support_agent", + model=model, + tools=[...], + session_manager=FileSessionManager(session_id=conversation_id, storage_dir="./sessions"), # keep the repo's own manager + trace_attributes={"session.id": conversation_id}, + callback_handler=None, +) +``` + +Rules: +- Use the app's existing chat/thread/conversation id. Never a per-request `uuid4()`, never a constant. +- A module-level / shared `Agent` must not carry a session id: construct the agent per request or per conversation (restore history with the repo's session manager). If the repo must keep a long-lived agent per conversation, construct it with that conversation's id. +- `session_manager=...(session_id=...)` does NOT put the id on spans. `trace_attributes` is still required. +- Keep any existing `trace_attributes` keys; add `session.id`. Don't add emails/names (never redacted). +- `Agent(name=...)` on every agent. Default name is `Strands Agents` for all agents, which collapses sub-agent lanes. +- Multi-agent roots: + - Agents as tools: `session.id` on the orchestrator only is enough (one trace). + - `Swarm([...], trace_attributes={"session.id": conversation_id})`. + - Graph: `graph = builder.build()` then `graph.trace_attributes = {"session.id": conversation_id}` (`GraphBuilder` drops trace attributes). + +TypeScript: set BOTH keys. TS names `gen_ai.provider.name` and the tracer scope after `OTEL_SERVICE_NAME`, so Maple can't fingerprint it as Strands and reads `gen_ai.conversation.id` instead. + +```ts +import { Agent, FileStorage, SessionManager } from "@strands-agents/sdk" + +const agent = new Agent({ + name: "support_agent", + model, + tools: [...], + traceAttributes: { "session.id": conversationId, "gen_ai.conversation.id": conversationId }, + sessionManager: new SessionManager({ sessionId: conversationId, storage: { snapshot: new FileStorage("./sessions") } }), // keep the repo's own storage +}) +``` + +- TS: construct the agent PER REQUEST (restore history with the repo's `SessionManager` storage). The TS `invoke_agent` span always carries the agent instance's accumulated usage (no `gen_ai_use_latest_invocation_tokens`), so a reused agent re-reports every earlier turn and Maple's totals inflate. +- TS stamps `traceAttributes` on `invoke_agent` only (not chat/tool spans). That is enough. + +## Step 4: content + +- Content capture is ON by default; Step 2's tokens only move it onto attributes. Nothing else to enable. +- Redaction, only if the user asks or the repo handles regulated data: append `gen_ai_unredacted_attributes=` to `OTEL_SEMCONV_STABILITY_OPT_IN`. `;`-separated, single trailing `*` only. Covered keys: `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result`. Empty list (`gen_ai_unredacted_attributes=`) redacts all to `[REDACTED]`. Example keeping replies only: `...,gen_ai_unredacted_attributes=gen_ai.output.*;gen_ai.tool.call.result`. +- Not redactable: `gen_ai.tool.description`, `gen_ai.tool.json_schema`, `trace_attributes` values. + +## Step 5: tools, errors, sub-agents + +- Nothing to add for tool failures: a raising `@tool` (or one returning `{"status": "error"}`) gets span status ERROR with the exception message and `gen_ai.tool.status=error`. Do not catch exceptions inside tools just to return friendly text; that hides the failure unless you return `status: "error"`. +- Sub-agents: `sub.as_tool(description=...)` nests `invoke_agent ` under `execute_tool `; Maple shows a delegation lane. Give each sub-agent a distinct `name`. +- Graph node ids are not exported; the agent `name` identifies the node. +- Interrupt/resume (HITL): the resume is a new trace in the same session (same `trace_attributes`). Interrupted tool spans appear twice with the same `gen_ai.tool.call.id` (first ends OK with no result, second has the real outcome); Maple counts both as tool calls. Expected, framework-level; do not try to filter spans. +- TS: failed Graph nodes end with status OK (upstream bug harness-sdk#4166). Tool failures are fine. + +## Step 6: flush + +- Long-running server: nothing. +- Script / CLI / job / notebook / test: in `finally`: + +```py +telemetry.tracer_provider.force_flush() +telemetry.tracer_provider.shutdown() +``` + +- Lambda: `force_flush()` at the end of each invocation, no `shutdown()`. +- Existing provider (Step 0): flush that provider instead. +- TS: `await provider.forceFlush(); await provider.shutdown()` before exit. `setupTracer`'s own `beforeExit` flush does not run after `process.exit()`. + +## Step 7: verify + +Run one short conversation (2-3 messages, one tool call; plus a failing tool if easy) with the real key, flush, then check in Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), or via the Maple MCP (`list_agent_sessions`, `get_agent_session`): + +- Exactly one session per conversation id (not `trace:` sessions); a second conversation gets a different session. +- Framework shows Strands Agents (Python). TS with a custom service name shows "Unidentified"; expected. +- One turn per agent call, with the user's message as the turn label. +- Transcript shows user messages, replies, tool calls with args and results. Empty transcript → `gen_ai_span_attributes_only` missing or set after the first `Agent(`. +- Spans: `invoke_agent ` → `execute_event_loop_cycle` → `chat` / `execute_tool `; model id on `chat` spans. +- Input/output tokens on every `chat` span, including streamed turns. Session total ≈ sum of `chat` spans, not several times more. +- Failed tool counted as failed; successful tools not. +- Sub-agents in separate lanes with their own names. +- Cost shows "unpriced" (Strands emits no cost). Expected. +- No duplicate `chat` spans per model call. + +If you can't reach Maple, set `OTEL_EXPORTER_OTLP_ENDPOINT` to a local collector or use `telemetry.setup_console_exporter()` once and check a `chat` span has `gen_ai.input.messages` as an ATTRIBUTE (not in `events`) and `session.id`. + +## Do not + +- Do not omit `gen_ai_span_attributes_only`: Maple never reads span events, so the transcript would be empty. +- Do not set `OTEL_SEMCONV_STABILITY_OPT_IN` in code after an `Agent` exists. +- Do not create a second `TracerProvider` when one exists; do not call `StrandsTelemetry()` under `opentelemetry-instrument`/ADOT. +- Do not put a session id on a shared module-level agent. +- Do not rely on `session_manager` or `gen_ai.conversation.id` (Python) for Maple sessions; use `trace_attributes={"session.id": ...}`. +- Do not add another GenAI instrumentor (OpenLIT, OpenLLMetry, OpenInference, OpenAI/Bedrock instrumentation): duplicate model calls and tokens. +- Do not also enable OpenRouter Broadcast (or another gateway trace export) for the same traffic: Strands `chat` spans have no `gen_ai.response.id`, so Maple can't dedupe and tokens double. +- Do not leave agents unnamed. +- Do not use `OTEL_TRACES_SAMPLER=traceidratio` unless the user wants it: dropped traces are dropped turns. +- Do not pin `strands-agents` below 1.54. +- Do not put PII in `trace_attributes`; it is never redacted. diff --git a/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md new file mode 100644 index 0000000000..877670800a --- /dev/null +++ b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md @@ -0,0 +1,237 @@ +--- +name: maple-agent-tracing-vercel-ai-sdk +description: "Trace Vercel AI SDK agents with Maple: register the AI SDK's OpenTelemetry integration, export to Maple, and stamp a conversation id so each chat is one Agent Session with transcript, tool calls, sub-agents and tokens. Covers generateText, streamText, ToolLoopAgent, Node.js and Next.js. Triggers on 'trace my vercel ai sdk agent', 'add Maple to vercel ai sdk', 'agent sessions for vercel ai sdk', 'OpenTelemetry for vercel ai sdk', 'trace my ai sdk app'." +--- + +# Maple agent tracing: Vercel AI SDK + +Human guide with the reasoning: https://maple.dev/docs/agent-tracing/vercel-ai-sdk + +## Goal + +One conversation = one Maple Agent Session, one turn per `generate()`/`stream()` call, with the transcript, every model call (model, tokens, TTFT), every tool call (name, args, result, failures), and a lane per sub-agent. + +How it works: AI SDK 7 emits GenAI-semconv spans through `@ai-sdk/otel` (`invoke_agent ` → `step ` → `chat ` + `execute_tool `) on tracer `gen_ai`, once `registerTelemetry(new OpenTelemetry())` has run. Content is on by default. Maple detects the AI SDK by `ai.*` attributes on the `gen_ai`/`ai` scope and groups sessions by `gen_ai.conversation.id`, which the AI SDK never sets. You add it with `enrichSpan` from `runtimeContext`. + +Known gaps (tell the user, don't try to fix): cost shows as "unpriced" (AI SDK emits no cost; Maple never prices tokens); on AI SDK 5/6 the final assistant reply is missing from transcripts. + +## Step 0: Detect + +- `ai` version in `package.json` / lockfile: + - `>= 7`: this skill. Upgrade to `^7.0.106` or newer if older (earlier 7.x leave spans open when a stream errors mid-read). Node.js >= 22 required. + - `5.x` / `6.x`: ask the user whether to upgrade to 7 (recommended; `npx @ai-sdk/codemod v7`). If not, go to "AI SDK 5/6" at the end. +- Mastra (`@mastra/core`) or another framework built on `ai`: stop, use that framework's skill instead. +- Existing OpenTelemetry: search for `NodeSDK`, `NodeTracerProvider`, `registerOTel`, `@vercel/otel`, `Sentry.init`, `@langfuse/otel`, `LangfuseSpanProcessor`, `braintrust`, `registerTelemetry(`, `experimental_telemetry`, `telemetry:`. + - An SDK/provider already exists: reuse it. Add one Maple exporting span processor to it. Never start a second SDK. + - `registerTelemetry(...)` already exists: extend that call's `OpenTelemetry` options; never call it twice (it appends, so every span is emitted twice). + - `LegacyOpenTelemetry` registered: replace it with `OpenTelemetry` unless the user says another backend depends on the legacy format. Never register both. +- Next.js app (`next` dependency, `instrumentation.ts`): use Step 2b. +- Find every AI SDK call site: `generateText(`, `streamText(`, `new ToolLoopAgent(`, `createAgentUIStreamResponse(`, `agent.generate(`, `agent.stream(`. Find each one's conversation id (chat id, thread id, `useChat` request body `id`). + +## Step 1: Key and region + +- US endpoint `https://ingest.maple.dev`, EU endpoint `https://ingest.eu.maple.dev`. Header `Authorization=Bearer `. +- Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. +- Private `maple_sk_` keys never go in browser code. Ingest keys are write-only. +- Follow the repo's existing secret/env convention (e.g. `MAPLE_INGEST_KEY` or `OTEL_EXPORTER_OTLP_*` in `.env`). If there is none, inlining the ingest key is acceptable. + +## Step 2a: Install and init (Node.js) + +```bash +npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/api @opentelemetry/sdk-node \ + @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources +``` + +Use the repo's package manager. Env (or the repo's equivalent): + +```bash +OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev +OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer +OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf +``` + +Create `instrumentation.ts` (adapt service name/environment; to inline instead of env, pass `{ url: "https://ingest.maple.dev/v1/traces", headers: { authorization: "Bearer " } }` to `OTLPTraceExporter`): + +```ts +import { OpenTelemetry } from "@ai-sdk/otel" +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { resourceFromAttributes } from "@opentelemetry/resources" +import { NodeSDK } from "@opentelemetry/sdk-node" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" +import { registerTelemetry } from "ai" + +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) + +export const sdk = new NodeSDK({ + resource: resourceFromAttributes({ + "service.name": "support-agent", + "deployment.environment.name": process.env.NODE_ENV ?? "development", + }), + spanProcessors: [spanProcessor], +}) +sdk.start() + +registerTelemetry( + new OpenTelemetry({ + usage: true, + runtimeContext: true, + enrichSpan: ({ runtimeContext }) => + typeof runtimeContext?.conversationId === "string" + ? { "gen_ai.conversation.id": runtimeContext.conversationId } + : undefined, + }), +) +``` + +- `import "./instrumentation"` as the FIRST line of every entry point (server, worker, CLI). `registerTelemetry` must run before the first AI SDK call. +- `usage: true` + `runtimeContext: true` are required, not optional: they put `ai.*` keys on `invoke_agent`/`chat` spans, which is how Maple labels the vendor "Vercel AI SDK", picks the AI SDK token convention and reads TTFT. +- Do not pass `tracer:` to `OpenTelemetry` unless reusing a provider requires it; if you must, use `provider.getTracer("gen_ai")`. Other scope names break detection. +- Existing provider: add `spanProcessor` to it (`spanProcessors: [..., spanProcessor]` or the provider's add method) instead of creating `NodeSDK`. + +## Step 2b: Install and init (Next.js) + +`npm install ai@^7.0.106 @ai-sdk/otel @vercel/otel @opentelemetry/api`. In `instrumentation.ts` (root, or `src/` if the app uses `src/`), inside `register()`: + +```ts +import { OpenTelemetry } from "@ai-sdk/otel" +import { OTLPHttpProtoTraceExporter, registerOTel } from "@vercel/otel" +import { registerTelemetry } from "ai" + +export function register() { + registerOTel({ + serviceName: "support-chat", + traceExporter: new OTLPHttpProtoTraceExporter({ + url: "https://ingest.maple.dev/v1/traces", + headers: { authorization: `Bearer ${process.env.MAPLE_INGEST_KEY}` }, + }), + }) + registerTelemetry( + new OpenTelemetry({ + usage: true, + runtimeContext: true, + enrichSpan: ({ runtimeContext }) => + typeof runtimeContext?.conversationId === "string" + ? { "gen_ai.conversation.id": runtimeContext.conversationId } + : undefined, + }), + ) +} +``` + +- An existing `registerOTel` call: keep it, add the Maple exporter (or `OTEL_EXPORTER_OTLP_*` env vars if it has no `traceExporter`), add `registerTelemetry` next to it. +- Sentry in the same app: it must use `skipOpenTelemetrySetup: true`, or spans duplicate. +- `@vercel/otel` force-ends every still-open span of a trace when its root span ends. Read AI SDK streams inside the request (return them as the response); never consume them in `after()` or other post-response work, or the `chat`/`invoke_agent` spans lose output and tokens. +- Next.js 13.4/14: `experimental: { instrumentationHook: true }` in `next.config`. + +## Step 3: Conversation id (session) + +Every AI SDK call site that belongs to a conversation must pass `runtimeContext: { conversationId }` AND `telemetry.includeRuntimeContext: { conversationId: true }`. Without the include, `enrichSpan` gets `{}`. + +generateText / streamText: + +```ts +streamText({ + model, + messages, + runtimeContext: { conversationId: chatId }, + telemetry: { functionId: "support_agent", includeRuntimeContext: { conversationId: true } }, +}) +``` + +If `runtimeContext` already exists, add `conversationId` to it and to `includeRuntimeContext`; keep other keys excluded (they may hold secrets). + +ToolLoopAgent (`generate`/`stream` take no per-call `runtimeContext`): add a call option. + +```ts +new ToolLoopAgent({ + // ...existing settings + callOptionsSchema: z.object({ conversationId: z.string() }), + prepareCall: ({ options, ...rest }) => ({ + ...rest, + runtimeContext: { conversationId: options.conversationId }, + }), + telemetry: { functionId: "support_agent", includeRuntimeContext: { conversationId: true } }, +}) +await assistant.generate({ messages, options: { conversationId: chatId } }) +``` + +- Agent already has `callOptionsSchema`/`prepareCall`: extend both; keep their existing return values and merge `runtimeContext`. +- `useChat` routes: the request body has `id` (the chat id). Use it: `createAgentUIStreamResponse({ agent, uiMessages: messages, options: { conversationId: id } })`, or `runtimeContext: { conversationId: id }` on `streamText`. +- Id must be stable per conversation and unique across conversations. Never a module constant, never `Date.now()` per call, never a per-process id. If the app has no id for a conversation, create one where the conversation is created and persist it with it. +- Tool approvals (`toolApproval: { : "user-approval" }` on the call or agent; `needsApproval` on `tool()` is deprecated in 7): the resume after the `tool-approval-response` is a second `generate`/`stream` call and a second trace. Pass the same id. Maple shows it as a second turn. +- Background jobs with no conversation: one id per job run. + +## Step 4: Content + +- On by default. Do not set `recordInputs`/`recordOutputs` unless the user asks for privacy; if they do, set them per call/agent in `telemetry` and tell them the transcript will be empty for those calls. +- Never set `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` / `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT`: truncated JSON is dropped by Maple. +- Calls that send images/PDFs: warn the user they are recorded as base64 on every step; suggest `recordInputs: false` on those calls if payloads are large. + +## Step 5: Tools, errors, sub-agents + +- Set a distinct `telemetry.functionId` on EVERY agent/call (snake_case agent name). It becomes `gen_ai.agent.name`; `ToolLoopAgent.id` is not exported. +- Tool failures must throw from `execute`. If a tool returns `{ error }` / `{ ok: false }` for real failures, change it to throw (ask the user first if the model relies on the payload). An always-throwing `execute` needs an explicit return type (`Promise`) or it infers `never`. +- Sub-agents as tools: `execute: async ({ task }) => (await worker.generate({ prompt: task })).text`. The worker needs its own `functionId`; it does not need the conversation id (one span per trace carries it). Separate top-level calls in a pipeline (e.g. a summary agent after the orchestrator) are separate traces: each needs the conversation id. +- Don't pass `telemetry.integrations` on a call unless it includes the `OpenTelemetry` instance: it replaces the global integrations for that call. + +## Step 6: Flush + +- Scripts/CLIs: `await sdk.shutdown()` in a `finally` at the end. +- Serverless handlers (Lambda, Cloud Run jobs, queue consumers, cron, Vercel Workflow steps): `await spanProcessor.forceFlush()` in a `finally` per invocation. +- Next.js on Vercel: `@vercel/otel` flushes per request via `waitUntil`; nothing to add in route handlers. Work outside a request needs an explicit flush. +- Streams: read every stream to the end (`for await (const c of result.textStream)` / `await result.consumeStream()` / return it as the response) before flushing. An unread stream never ends its spans. + +## Step 7: Verify + +Run one real conversation: 2+ turns with the same id, one streamed, one tool call; plus a second conversation. Then check (via Maple MCP `list_agent_sessions` / `get_agent_session`, or the Agent Sessions page, ~30 s after the run): + +- One session per conversation (id = your `conversationId`), not `trace:` sessions; the second conversation is a separate session. +- Framework shows **Vercel AI SDK**, not Unidentified. +- Turns = number of `generate`/`stream` calls (an approval pause + resume = 2 turns); each has `invoke_agent `, `step `, `chat `, `execute_tool ` spans in one trace. +- Transcript shows user messages, assistant replies and tool calls; turn labels are the user's messages. +- Every `chat` span has input and output tokens, including the streamed turn; the streamed `chat` span has TTFT. +- Tool calls have name, arguments, result; a throwing tool is counted as failed with its message; successful tools are not failed. +- Sub-agents show as their own lanes named after their `functionId`. +- No attribute contains the provider API key or `Bearer `. +- Cost shows as unpriced (expected). +- The process exited cleanly and no turn is missing (flush ran). + +## AI SDK 5/6 (only if the user won't upgrade) + +- Tracing is per call: `experimental_telemetry: { isEnabled: true, functionId: "", metadata: { conversationId } }` on every call. Spans: `ai.generateText`/`ai.streamText`, `.doGenerate`/`.doStream`, `ai.toolCall`, tracer `ai`. Provider setup as in Step 2a without `registerTelemetry`/`@ai-sdk/otel`. +- Copy the id to `gen_ai.conversation.id` with a span processor placed before the exporting one: + +```ts +import type { Context } from "@opentelemetry/api" +import type { ReadableSpan, Span, SpanProcessor } from "@opentelemetry/sdk-trace-base" + +export class ConversationIdProcessor implements SpanProcessor { + onStart(span: Span, _parentContext: Context) { + const id = span.attributes["ai.telemetry.metadata.conversationId"] + if (typeof id === "string") span.setAttribute("gen_ai.conversation.id", id) + } + onEnd(_span: ReadableSpan) {} + forceFlush() { + return Promise.resolve() + } + shutdown() { + return Promise.resolve() + } +} +// spanProcessors: [new ConversationIdProcessor(), spanProcessor] +``` + +- Tell the user: final assistant replies won't show in the transcript (recorded only as plain `ai.response.text`); upgrading to 7 fixes it. + +## Do not + +- Do not use `experimental_telemetry: { isEnabled: true }` as the v7 setup; without `registerTelemetry` there are zero spans. +- Do not call `registerTelemetry` twice or register `LegacyOpenTelemetry` alongside `OpenTelemetry` (duplicate spans, double tokens). +- Do not start a second OpenTelemetry SDK next to an existing one; add a span processor. +- Do not rely on `runtimeContext` alone: Maple does not read `ai.settings.context.*` keys as a session id. The `enrichSpan` → `gen_ai.conversation.id` step is required. +- Do not use `telemetry.metadata` (removed in v7) or `ToolLoopAgent({ id })` for agent names. +- Do not stamp `maple_ai.session.id` on AI SDK spans: it re-vendors them and loses AI SDK decoding. +- Do not add a second AI SDK tracer that also exports to Maple (Langfuse/Braintrust/Sentry AI integrations, OpenLLMetry/OpenInference AI SDK processors): every model call gets recorded twice. +- Do not return error payloads from failing tools; throw. +- Do not set attribute length limits. +- Do not use a gRPC exporter; Maple ingest is OTLP over HTTP. diff --git a/skills/maple-agent-tracing/SKILL.md b/skills/maple-agent-tracing/SKILL.md new file mode 100644 index 0000000000..2847c70b9f --- /dev/null +++ b/skills/maple-agent-tracing/SKILL.md @@ -0,0 +1,55 @@ +--- +name: maple-agent-tracing +description: "Trace an AI agent or LLM app with Maple so each conversation shows up as one Agent Session with its transcript, model calls, tool calls, tokens, cost and failures. Detects the agent framework (Vercel AI SDK, OpenAI Agents SDK, Mastra, LangChain/LangGraph, Claude Agent SDK, Pydantic AI, LlamaIndex, CrewAI, Google ADK, Strands, smolagents, Agno, DSPy, Haystack, Microsoft Agent Framework, Spring AI, LiteLLM, OpenRouter, raw provider SDKs) and installs the matching per-framework skill. Triggers on 'trace my agent', 'add agent observability', 'LLM tracing with Maple', 'set up Maple agent sessions', 'OpenTelemetry for my AI agent'." +--- + +# Maple agent tracing (router) + +This skill only picks the right per-framework skill. Each framework has its own skill so you load the steps for your stack and nothing else. The human-readable overview is https://maple.dev/docs/agent-tracing. + +## Step 1: Find every agent in the repo + +Look for LLM and agent dependencies in every app and service: `package.json`, `pyproject.toml`, `requirements*.txt`, `uv.lock`, `pom.xml`, `build.gradle*`, `*.csproj`, `go.mod`. A repo can have more than one (a TypeScript chat backend and a Python worker, for example). Handle each one. + +## Step 2: Install the matching skill and follow it + +Pick the first row that matches the service's dependencies. Install the skill with: + +```bash +npx skills add MapleTechLabs/maple/skills --skill -y +``` + +Then read the installed `SKILL.md` and follow it. If `npx skills` is unavailable, read the file directly from `https://github.com/MapleTechLabs/maple/tree/main/skills//SKILL.md`. + +| Dependency | Skill | +| --- | --- | +| `@mastra/core` | `maple-agent-tracing-mastra` | +| `@openai/agents`, `openai-agents` | `maple-agent-tracing-openai-agents` | +| `@langchain/langgraph`, `langchain`, `@langchain/core`, `langgraph`, `langchain-core` | `maple-agent-tracing-langchain` | +| `@anthropic-ai/claude-agent-sdk`, `claude-agent-sdk`, or the `claude` CLI itself | `maple-agent-tracing-claude-agent-sdk` | +| `ai` (Vercel AI SDK) | `maple-agent-tracing-vercel-ai-sdk` | +| `pydantic-ai`, `pydantic-ai-slim` | `maple-agent-tracing-pydantic-ai` | +| `crewai` | `maple-agent-tracing-crewai` | +| `google-adk` | `maple-agent-tracing-google-adk` | +| `llama-index`, `llama-index-core` | `maple-agent-tracing-llamaindex` | +| `strands-agents`, `@strands-agents/sdk` | `maple-agent-tracing-strands` | +| `smolagents` | `maple-agent-tracing-smolagents` | +| `agno` | `maple-agent-tracing-agno` | +| `dspy` | `maple-agent-tracing-dspy` | +| `haystack-ai` | `maple-agent-tracing-haystack` | +| `agent-framework`, `Microsoft.Agents.AI`, `semantic-kernel`, `Microsoft.SemanticKernel` | `maple-agent-tracing-microsoft-agent-framework` | +| `spring-ai-*` (Maven/Gradle) | `maple-agent-tracing-spring-ai` | +| `litellm` (SDK or proxy) | `maple-agent-tracing-litellm` | +| Requests go through OpenRouter (`openrouter.ai` base URL) and the user wants gateway-side traces | `maple-agent-tracing-openrouter` | +| Only a provider SDK: `openai`, `@anthropic-ai/sdk`, `anthropic`, `google-genai`, `@google/genai` | `maple-agent-tracing-provider-sdks` | +| Anything else, a hand-rolled agent loop, or another language | `maple-agent-tracing-opentelemetry` | + +Order matters: a framework row wins over the provider-SDK row, because frameworks depend on provider SDKs and instrumenting both records every model call twice. Mastra and LangChain JS depend on `ai` or `openai` too; match the framework first. + +## Step 3: Hand-off + +After the per-framework skill's own checks pass, tell the user: + +- which services you instrumented and with which skill; +- that each conversation appears in Maple under **Agent Sessions** (`https://app.maple.dev/agent-sessions`, or `app.eu.maple.dev` for EU organizations) once one conversation has run; +- anything the per-framework skill said is not captured for that framework (for example, cost or message content), so the gap is expected rather than a bug. From db727b586d876753be48fc92c156938f27e6416f Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Mon, 28 Sep 2026 19:31:19 +0200 Subject: [PATCH 02/39] docs(agent-tracing): apply verifier fixes from end-to-end runs of every guide --- .../content/docs/agent-sessions/overview.md | 4 +- .../src/content/docs/agent-tracing/agno.md | 4 +- .../src/content/docs/agent-tracing/crewai.md | 52 ++++-- .../src/content/docs/agent-tracing/dspy.md | 4 +- .../content/docs/agent-tracing/google-adk.md | 4 +- .../content/docs/agent-tracing/langchain.md | 78 +++------ .../src/content/docs/agent-tracing/litellm.md | 155 +++++++++++++----- .../content/docs/agent-tracing/llamaindex.md | 44 +++-- .../src/content/docs/agent-tracing/mastra.md | 116 ++++++++----- .../microsoft-agent-framework.md | 10 +- .../docs/agent-tracing/openai-agents.md | 8 +- .../content/docs/agent-tracing/openrouter.md | 8 +- .../docs/agent-tracing/opentelemetry.md | 12 +- .../docs/agent-tracing/provider-sdks.md | 18 +- .../content/docs/agent-tracing/pydantic-ai.md | 2 +- .../content/docs/agent-tracing/smolagents.md | 4 +- .../content/docs/agent-tracing/spring-ai.md | 79 +++++---- .../src/content/docs/agent-tracing/strands.md | 4 +- .../docs/agent-tracing/vercel-ai-sdk.md | 2 +- .../content/docs/getting-started/ai-agents.md | 3 +- .../src/content/docs/instrumentation.mdx | 4 + skills/maple-agent-tracing-crewai/SKILL.md | 8 +- skills/maple-agent-tracing-litellm/SKILL.md | 64 ++++++-- .../maple-agent-tracing-llamaindex/SKILL.md | 19 ++- skills/maple-agent-tracing-mastra/SKILL.md | 78 ++++++--- .../SKILL.md | 7 +- .../SKILL.md | 3 +- .../references/python.md | 2 +- .../references/typescript.md | 4 +- .../SKILL.md | 11 +- .../references/python.md | 4 +- .../references/typescript.md | 3 + skills/maple-agent-tracing-spring-ai/SKILL.md | 50 +++--- 33 files changed, 558 insertions(+), 310 deletions(-) diff --git a/apps/landing/src/content/docs/agent-sessions/overview.md b/apps/landing/src/content/docs/agent-sessions/overview.md index 91f956bd58..9856f3f868 100644 --- a/apps/landing/src/content/docs/agent-sessions/overview.md +++ b/apps/landing/src/content/docs/agent-sessions/overview.md @@ -94,14 +94,14 @@ Failures are grouped by what went wrong: Maple fingerprints the failed result (o You don't need this to use Agent Sessions, but it explains what the framework guides ask you to configure. 1. **Maple recognizes AI spans at ingest.** A span counts if it carries `gen_ai.operation.name` (the OpenTelemetry GenAI conventions) or matches the fingerprint of a framework Maple knows: Vercel AI SDK, OpenAI Agents SDK, LangChain and LangGraph, Mastra, Pydantic AI, CrewAI, Google ADK, Strands, Claude Code, Spring AI and others. Traces with no AI span don't appear. -2. **It reads the session id that framework uses.** For most frameworks that's `gen_ai.conversation.id`; a few use `session.id`, LangGraph uses its thread id. One span per trace is enough. A trace without one becomes a session of its own, which is why "every message is its own session" is the most common setup problem. +2. **It reads the session id that framework uses.** For most frameworks that's `gen_ai.conversation.id`; several, including CrewAI, DSPy and Strands, use `session.id`. One span per trace is enough. A trace without one becomes a session of its own, which is why "every message is its own session" is the most common setup problem. 3. **It splits the session into turns**, normally one per trace, and decodes each model call's model, tokens and content, and each tool call's name, arguments and result. 4. **It counts tokens once.** Providers disagree on whether cached and reasoning tokens are included in the input and output counts. Maple resolves that per provider, and when a framework records usage on both an agent span and the model calls inside it, Maple keeps the model calls' numbers. Two limits are worth knowing up front: - **Maple reads span attributes.** Prompts and replies that a framework writes only to span events or OpenTelemetry logs are still stored (on the span, or under [Logs](/docs/explore/logs)), but they don't show up in the transcript. The framework guides say where each framework puts its content and how to move it onto spans when that's possible. -- **Maple shows cost your instrumentation reports; it doesn't price tokens itself.** Cost appears when spans carry `gen_ai.usage.cost` (or OpenInference's `llm.cost.total`). OpenRouter and some instrumentations send it. Otherwise the session shows tokens and reads as unpriced. +- **Maple shows cost your instrumentation reports; it doesn't price tokens itself.** Cost appears when spans carry `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or OpenInference's `llm.cost.total`. OpenRouter and some instrumentations send it. Otherwise the session shows tokens and reads as unpriced. A session view loads up to 2,000 spans. Longer sessions show the first 2,000 and say they were cut. diff --git a/apps/landing/src/content/docs/agent-tracing/agno.md b/apps/landing/src/content/docs/agent-tracing/agno.md index 7dafa92aec..de3de5483b 100644 --- a/apps/landing/src/content/docs/agent-tracing/agno.md +++ b/apps/landing/src/content/docs/agent-tracing/agno.md @@ -9,7 +9,9 @@ icon: "agno" Agno's tracing is built on OpenInference. The `openinference-instrumentation-agno` package wraps every agent and team run, every model call and every tool call, and Agno's own `setup_tracing()` uses the same instrumentor. The catch is where `setup_tracing()` sends the spans: into your AgentOS database, not to an OpenTelemetry endpoint. To get them into Maple you install the instrumentor yourself with an OTLP exporter. -The thing that goes wrong by default is the session id. Agno always stamps `session.id` on the run span, but when you don't pass `session_id=`, it generates one and keeps it on the `Agent` instance. A chat server with one module-level agent then puts every user's conversation into the same Maple session. This guide covers Agno 3.0 (tested with 3.0.11) and `openinference-instrumentation-agno` 1.0.10 on Python 3.10 to 3.14. +The thing that goes wrong by default is the session id. Agno always stamps `session.id` on the run span, but when you don't pass `session_id=`, it generates one and keeps it on the `Agent` instance. A chat server with one module-level agent then puts every user's conversation into the same Maple session. + +This guide covers Agno 3.0 (tested with 3.0.11) and `openinference-instrumentation-agno` 1.0.10 on Python 3.10 to 3.14. ## Quick setup with a coding agent diff --git a/apps/landing/src/content/docs/agent-tracing/crewai.md b/apps/landing/src/content/docs/agent-tracing/crewai.md index 7fd6ba2e48..eebf6f6481 100644 --- a/apps/landing/src/content/docs/agent-tracing/crewai.md +++ b/apps/landing/src/content/docs/agent-tracing/crewai.md @@ -9,7 +9,9 @@ icon: "crewai" CrewAI sends nothing to your OpenTelemetry backend on its own. Its built-in telemetry is anonymous usage analytics that goes to CrewAI on a private tracer provider, and its OpenTelemetry export is a CrewAI AMP feature. The traces come from OpenInference: `openinference-instrumentation-crewai` records crews, flows, tasks and tools, and a second instrumentor for the SDK CrewAI calls records the model calls, prompts and tokens. -Most broken CrewAI traces are missing that second instrumentor. With only the CrewAI one, you get agent and tool spans with no model, no tokens and no transcript. The other gap is the conversation: CrewAI has no chat thread, so every `kickoff()` is its own trace with nothing linking it to the previous message. This guide covers CrewAI 1.15 with `openinference-instrumentation-crewai` 1.1.18 and `openinference-instrumentation-openai` 0.1.61, on Python 3.10 to 3.13. +Most broken CrewAI traces are missing that second instrumentor. With only the CrewAI one, you get agent and tool spans with no model, no tokens and no transcript. The other gap is the conversation: CrewAI has no chat thread, so every `kickoff()` is its own trace with nothing linking it to the previous message. + +This guide covers CrewAI 1.15 with `openinference-instrumentation-crewai` 1.1.18 and `openinference-instrumentation-openai` 0.1.61, on Python 3.10 to 3.13. ## Quick setup with a coding agent @@ -147,14 +149,26 @@ Give the crew a `name`. An unnamed crew's root span is `Crew_.kickoff`, a If you skip `using_session`, every message shows up in **Agent Sessions** as its own one-turn session named after its trace id. Setting `gen_ai.conversation.id` on your own spans doesn't help, because Maple reads `session.id` for CrewAI. -`using_session` also works for conversational flows, where CrewAI's own session id is the flow's `state.id`. Use the same value for both: +`using_session` also works for conversational flows, where CrewAI's own session id is the flow's `state.id`. Use the same value for both, and give the flow a `name`: ```py -with using_session(conversation_id): - reply = support_flow.handle_turn(text, session_id=conversation_id) +from crewai.flow import ConversationConfig, ConversationState, Flow + + +@ConversationConfig(llm=llm, system_prompt="You are a concise assistant.") +class SupportFlow(Flow[ConversationState]): + name = "support_flow" + + +support_flow = SupportFlow() + + +def handle_turn(conversation_id: str, text: str) -> str: + with using_session(conversation_id): + return support_flow.handle_turn(text, session_id=conversation_id) ``` -Each `handle_turn` runs one `kickoff()`, so each message is a trace under the flow's `.kickoff` span. +Each `handle_turn` runs one `kickoff()`, so each message is a trace under a `support_flow.kickoff` span, with `support_flow.route_conversation` and `support_flow.converse_turn` below it. Without `name`, the root is `Flow_.kickoff`, and the name changes to `Flow_` after the first turn. ### Async kickoffs @@ -162,7 +176,7 @@ Use `kickoff()` or `await crew.kickoff_async()`. Don't use `await crew.akickoff( ### Streaming -`Crew(stream=True)` returns a `CrewStreamingOutput` right away and runs the crew again in a thread once you iterate it. The instrumentor records both calls, so each streamed message produces two traces: an empty `support.kickoff` and the real one. Wrap the turn in one span of your own so both land in one trace and one turn: +`Crew(stream=True)` returns a `CrewStreamingOutput` right away and calls `kickoff()` again in a thread once you iterate it. The instrumentor records both calls, so each streamed message produces two traces: an empty `support.kickoff` and the real one. Wrap the turn in one span of your own so both land in one trace and one turn: ```py from opentelemetry import trace @@ -196,6 +210,8 @@ config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=T `hide_inputs` drops the input messages and replaces `input.value` with `__REDACTED__`, and `hide_outputs` does the same for outputs. The session keeps its turns, model and tool calls, tokens and failures, with an empty transcript. `hide_input_text` and `hide_output_text` keep the message structure but redact the text. Each switch also has an `OPENINFERENCE_HIDE_*` environment variable. +The switches don't cover everything. The crew's `.kickoff` span always records `crew_tasks` (every task description, which is where the user's message goes in the example above), `crew_inputs` (the `kickoff(inputs=...)` values) and `crew_agents` (roles, goals and backstories). A flow's kickoff span records `flow_inputs` the same way. If prompts must never leave your infrastructure, delete those attributes in an OpenTelemetry Collector with an `attributes` processor before the data reaches Maple. + Agent roles, task names and tool names are always recorded, because they're span names. CrewAI's analytics, if you leave them on, also collect roles and tool names, so keep personal data out of both. ## Tools, errors and agents @@ -208,11 +224,25 @@ Each task is an agent span named `.._execute_core`, with `gen_a In a hierarchical crew (`process=Process.hierarchical`), CrewAI adds a manager agent called `Crew Manager` that delegates with the `Delegate work to coworker` and `Ask question to coworker` tools. The delegated work runs through `Agent.execute_task`, which the instrumentor doesn't patch, so the coworker's model calls appear inside the delegation's tool span with no agent span of their own and no lane. The manager's own task is an agent span like any other. -In a flow, `.kickoff` is the root and each `@start`, `@listen` and `@router` method is a `.` span, with any crew or `Agent.kickoff()` it runs nested inside. A flow paused by `@human_feedback` and continued with `flow.resume(...)` continues in a new request, and `resume()` isn't patched, so wrap it in `using_session` with the same id and a span of your own, like the streaming wrapper above. +Tool approvals with CrewAI's `@before_tool_call` hook need no tracing changes. The hook runs inside the `kickoff()`, before the tool span starts, so the approval stays in the turn's trace and an approved call is one tool span: + +```py +from crewai.hooks import before_tool_call + + +@before_tool_call(tools=["delete_file"]) +def approve_delete(context): + answer = context.request_human_input(prompt=f"Allow delete_file({context.tool_input})?") + return None if answer.strip().lower() == "approve" else False +``` + +A blocked call has no tool span at all: the model gets `Tool execution blocked by hook` as the result, and the refusal shows only in the next model call's input. + +In a flow, `.kickoff` is the root and each `@start`, `@listen` and `@router` method is a `.` span, with any crew or `Agent.kickoff()` it runs nested inside. A flow paused by `@human_feedback` and continued with `flow.resume(...)` continues in a new request, and `resume()` isn't patched, so wrap it in `using_session` with the same id and a span of your own, like the streaming wrapper above. ## Tokens and cost -Every model call carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals, along with cached and reasoning tokens when the provider reports them. Crew and agent spans carry no usage of their own, so nothing is counted twice. The model is the one the provider returned, such as `openai/gpt-4o-mini` behind OpenRouter. +Every model call carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals, along with cached and reasoning tokens when the provider reports them. Crew and agent spans carry no usage of their own, so nothing is counted twice. The model is the one the provider returned, such as `openai/gpt-4o-mini` or `anthropic/claude-haiku-4.5` behind OpenRouter. The provider is the SDK the call went through, so every model behind OpenRouter shows as `openai`. Streamed calls keep their token counts: CrewAI's OpenAI provider requests `stream_options={"include_usage": True}` whenever it streams. @@ -239,7 +269,7 @@ Call `provider.shutdown()` instead when the process is about to exit and won't t Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: -- **One session** for the conversation, with one turn per `kickoff()`. Each turn's trace starts at `support.kickoff` (your crew's name), or `.kickoff` for a flow. +- **One session** for the conversation, with one turn per `kickoff()`. Each turn's trace starts at `support.kickoff` (your crew's name), or `support_flow.kickoff` for a flow named `support_flow`. - **The transcript**: the system message built from the agent's role, goal and backstory, `Current Task: …` with your message, and the model's replies. - **Model calls** named `ChatCompletion` (from the OpenAI instrumentor), each with a model and input and output tokens. - **Tool calls** named `get_weather.run` and `calculate.run`, with results. @@ -255,10 +285,12 @@ A second conversation with a different id is a second session. If a turn is miss - **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. - **One session per message.** The kickoff isn't inside `using_session(...)`, or the id changes per request. Wrap every `kickoff()` and use the stored conversation id. - **Every model call and tool call is its own trace.** The crew runs with `akickoff()`. Use `kickoff()` or `kickoff_async()`. -- **An empty extra turn for each streamed message.** `Crew(stream=True)` runs the crew twice. Wrap the turn in one span, as in [Streaming](#streaming). +- **An empty extra turn for each streamed message.** `Crew(stream=True)` calls `kickoff()` twice. Wrap the turn in one span, as in [Streaming](#streaming). - **Agent and tool spans have no details on the session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)` to both instrumentors. - **All agents in one lane.** `CrewAIAgentNames` isn't on the provider, or it was added to a different provider than the one passed to `instrument()`. - **Tool arguments show the tool's input schema.** The dual-write copies `tool.parameters`, which is the schema, into `gen_ai.tool.call.arguments`. The call's actual arguments are in the tool span's `input.value`. This is an OpenInference mapping bug with no workaround in the span processor, because the arguments are set after the span starts. +- **Flow turns are named `Flow_.kickoff`.** The flow class has no `name`. Set `name = "support_flow"` on the class. +- **Prompts still visible with `hide_inputs=True`.** They're in the crew span's `crew_tasks` and `crew_inputs` attributes, which the switches don't cover. Delete them in a Collector. - **A delegated coworker has no lane.** Hierarchical delegation runs the coworker through `Agent.execute_task`, which isn't instrumented. Its model calls are inside the `Delegate work to coworker` tool span. - **Every model call appears twice.** Two model-layer instrumentors cover the same call, for example LiteLLM's and OpenAI's with a LiteLLM model that calls the `openai` SDK, or `litellm.callbacks=["otel"]` next to an OpenInference instrumentor. Keep one. - **The process hangs at exit asking about traces.** CrewAI's first-run trace prompt. Set `CREWAI_TRACING_ENABLED=false`. diff --git a/apps/landing/src/content/docs/agent-tracing/dspy.md b/apps/landing/src/content/docs/agent-tracing/dspy.md index bd2e2ff4ed..1528db99cb 100644 --- a/apps/landing/src/content/docs/agent-tracing/dspy.md +++ b/apps/landing/src/content/docs/agent-tracing/dspy.md @@ -9,7 +9,9 @@ icon: "python" DSPy has no OpenTelemetry code of its own. Spans come from OpenInference's `openinference-instrumentation-dspy`, which patches `Module.__call__`, `Predict.forward`, the adapters, `LM.__call__` and `Tool.__call__`. That gives you the shape of a program (which module called which predictor, which tool ran, what the model was sent), but no token counts, no tool names and no conversation id. -The usual fix for the missing tokens, adding `openinference-instrumentation-litellm`, stopped working in DSPy 3.4. Most models now run on DSPy's own `lm15` engine instead of LiteLLM, so the LiteLLM instrumentor records nothing: in our test, a call to `openrouter/openai/gpt-4o-mini` produced zero LiteLLM spans. This guide fills the gaps with a DSPy callback instead, which works on both engines. It covers DSPy 3.4 with `openinference-instrumentation-dspy` 0.1.45 on Python 3.10 or later, for `dspy.ReAct` agents and your own `dspy.Module` programs. +The usual fix for the missing tokens, adding `openinference-instrumentation-litellm`, stopped working in DSPy 3.4. Most models now run on DSPy's own `lm15` engine instead of LiteLLM, so the LiteLLM instrumentor records nothing: in our test, a call to `openrouter/openai/gpt-4o-mini` produced zero LiteLLM spans. This guide fills the gaps with a DSPy callback instead, which works on both engines. + +This guide covers DSPy 3.4 with `openinference-instrumentation-dspy` 0.1.45 on Python 3.10 or later, for `dspy.ReAct` agents and your own `dspy.Module` programs. ## Quick setup with a coding agent diff --git a/apps/landing/src/content/docs/agent-tracing/google-adk.md b/apps/landing/src/content/docs/agent-tracing/google-adk.md index 43d6d2910b..a68c3338d3 100644 --- a/apps/landing/src/content/docs/agent-tracing/google-adk.md +++ b/apps/landing/src/content/docs/agent-tracing/google-adk.md @@ -9,7 +9,9 @@ icon: "googleadk" Google's Agent Development Kit (ADK) creates OpenTelemetry spans itself, with no instrumentation package: one span per run, per agent, per model call and per tool call, under the instrumentation scope `gcp.vertex.agent`. Every span carries the ADK session id as `gen_ai.conversation.id`, so a multi-turn chat groups into one Maple session without any extra code. -What goes wrong by default is the transcript. ADK writes prompts, replies and tool payloads into its own `gcp.vertex.agent.llm_request`, `llm_response`, `tool_call_args` and `tool_response` attributes, which Maple doesn't read, so the session shows models and tokens next to an empty conversation. Two environment variables switch ADK to the OpenTelemetry GenAI message format that Maple renders. The other trap: with a plain `Runner`, nothing exports at all until you register a tracer provider yourself. This guide covers ADK for Python 2.10 and later. ADK for Go and Kotlin emit the same span names, but their setup isn't covered here. +What goes wrong by default is the transcript. ADK writes prompts, replies and tool payloads into its own `gcp.vertex.agent.llm_request`, `llm_response`, `tool_call_args` and `tool_response` attributes, which Maple doesn't read, so the session shows models and tokens next to an empty conversation. Two environment variables switch ADK to the OpenTelemetry GenAI message format that Maple renders. The other trap: with a plain `Runner`, nothing exports at all until you register a tracer provider yourself. + +This guide covers ADK for Python 2.10 and later. ADK for Go and Kotlin emit the same span names, but their setup isn't covered here. ## Quick setup with a coding agent diff --git a/apps/landing/src/content/docs/agent-tracing/langchain.md b/apps/landing/src/content/docs/agent-tracing/langchain.md index 02b5626f1e..b2639f009b 100644 --- a/apps/landing/src/content/docs/agent-tracing/langchain.md +++ b/apps/landing/src/content/docs/agent-tracing/langchain.md @@ -7,9 +7,11 @@ navLabel: "LangChain & LangGraph" icon: "langchain" --- -LangChain and LangGraph report every run through their callback system: each graph node, chat model call and tool call starts and ends a run. Two libraries turn those runs into OpenTelemetry spans. OpenInference's `openinference-instrumentation-langchain` adds its own callback handler, and LangSmith's SDK can export its runs over OTLP instead of to smith.langchain.com. Both work with Maple, and this guide uses OpenInference with its GenAI dual-write, because that's the one that gives Maple a transcript it can render. +LangChain and LangGraph report every run through their callback system: each graph node, chat model call and tool call starts and ends a run. Two libraries turn those runs into OpenTelemetry spans. OpenInference's `openinference-instrumentation-langchain` adds its own callback handler, and LangSmith's SDK can export its runs over OTLP instead of to smith.langchain.com. Both can export to Maple. This guide uses OpenInference with its GenAI dual-write, because that's the one that gives Maple a transcript it can render. -The default that goes wrong is the conversation. Every `invoke()` of an agent or graph is a new trace, so a chat of ten messages arrives as ten traces, and nothing links them until you pass a `thread_id`. The same `thread_id` a LangGraph checkpointer already needs is the one Maple groups sessions by. This guide covers Python: LangChain 1.4 (`create_agent`) and LangGraph 1.2 (`StateGraph`) with `openinference-instrumentation-langchain` 0.1.76, on Python 3.10 or later. +The default that goes wrong is the conversation. Every `invoke()` of an agent or graph is a new trace, so a chat of ten messages arrives as ten traces, and nothing links them until you pass a `thread_id`. The same `thread_id` a LangGraph checkpointer already needs is the one Maple groups sessions by. + +This guide covers Python: LangChain 1.4 (`create_agent`) and LangGraph 1.2 (`StateGraph`) with `openinference-instrumentation-langchain` 0.1.76, on Python 3.10 or later. ## Quick setup with a coding agent @@ -59,10 +61,12 @@ from opentelemetry.sdk.trace.export import BatchSpanProcessor # The name= you gave create_agent(), and graph nodes that act as agents AGENT_NAMES = {"assistant"} +# LangGraph's tool node and prompt templates: steps, not tool or model calls +STEP_NAMES = {"tools", "ChatPromptTemplate"} class AgentSpans(SpanProcessor): - """Marks your agents' spans as agent invocations, so Maple can name them and give each a lane.""" + """Names your agents' spans for Maple (one lane per agent) and keeps graph steps out of the tool and model counts.""" def on_start(self, span, parent_context=None): if span.instrumentation_scope.name != "openinference.instrumentation.langchain": @@ -70,6 +74,8 @@ class AgentSpans(SpanProcessor): if span.name in AGENT_NAMES: span.set_attribute("gen_ai.operation.name", "invoke_agent") span.set_attribute("gen_ai.agent.name", span.name) + elif span.name in STEP_NAMES: + span.set_attribute("gen_ai.operation.name", "invoke_workflow") provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES @@ -86,13 +92,13 @@ LangChainInstrumentor().instrument( What each part does: - **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.conversation.id` next to its OpenInference attributes when each span ends. Without it, Maple reads the spans as generic OpenInference: the transcript is a raw JSON blob and the `thread_id` is ignored, so every turn is its own session. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, as long as it's set before `TraceConfig` is built. -- **`AgentSpans`** fills the one thing the instrumentor leaves out. It names agent spans only by the word "agent": a `create_agent(name="support_agent")` span becomes an agent, `name="assistant"` stays a plain chain, and neither gets `gen_ai.agent.name`, which Maple needs to name agents and open a lane per sub-agent. The processor runs at span start, and the dual-write never overwrites a key that's already set. +- **`AgentSpans`** fills two gaps in how the spans are labeled. The instrumentor names agent spans only by the word "agent": a `create_agent(name="support_agent")` span becomes an agent, `name="assistant"` stays a plain chain, and neither gets `gen_ai.agent.name`, which Maple needs to name agents and open a lane per sub-agent. And a span with no operation is classified by its name, so LangGraph's `tools` node would count as one more tool call and a `ChatPromptTemplate` as one more model call. Marking them `invoke_workflow` keeps both out of the counts. The processor runs at span start, and the dual-write never overwrites a key that's already set. The instrumentor hooks LangChain's callback manager, so import order doesn't matter, as long as `instrument()` runs before the first `invoke()`. It traces every LangChain runnable in the process: agents, graphs, chains, chat models, tools and retrievers. If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second one. Add `AgentSpans()` and the OTLP exporter to the existing provider and pass that provider to `instrument()`. -LangSmith keeps working next to this. With `LANGSMITH_TRACING=true` and a LangSmith key, runs still go to smith.langchain.com over LangSmith's own API, and nothing is sent twice to Maple. Don't also set `LANGSMITH_OTEL_ENABLED`, which would add a second copy of every span to your provider (see [the LangSmith exporter](#langsmiths-opentelemetry-exporter-instead) below). +LangSmith keeps working next to this. With `LANGSMITH_TRACING=true` and a LangSmith key, runs still go to smith.langchain.com over LangSmith's own API, and nothing is sent twice to Maple. Don't also set `LANGSMITH_OTEL_ENABLED`, which would add a second copy of every span to your provider (see [why not LangSmith's exporter](#why-not-langsmiths-opentelemetry-exporter) below). ## Group a conversation into one session with thread_id @@ -191,18 +197,18 @@ The worker's run nests under the tool span and inherits the orchestrator's `thre Don't put "agent" in a tool's name. The instrumentor names a span of any kind an agent when its name contains the word, so a tool called `ask_weather_agent` loses its tool kind and Maple doesn't count it as a tool call. -Two gaps remain in what Maple shows for LangGraph tools. LangGraph runs tools inside a node, named `tools` in `create_agent` and wherever you add a `ToolNode` under that name, and Maple currently counts that node's span as an extra, unnamed tool call, because its name contains "tool". So a turn with one tool call shows two. The real tool calls are the ones with a name. The missing `gen_ai.tool.call.id` means Maple matches tool spans to the model's calls by name, which works unless one reply calls the same tool twice. +LangGraph runs tools inside a node, named `tools` in `create_agent` and usually in a `StateGraph` with a `ToolNode` too. Without an operation, Maple would count that node's span as a tool call because its name contains "tool", which is why `STEP_NAMES` lists it. If your tool node has another name with "tool" in it, like `run_tools`, add that name. + +The missing `gen_ai.tool.call.id` means Maple matches tool spans to the model's calls by name, which works unless one reply calls the same tool twice. ## Tokens and cost Every chat model span carries input and output tokens from LangChain's `usage_metadata`, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` next to the OpenInference `llm.token_count.*` originals. The model is the one you configured, in `gen_ai.request.model`, and the provider comes from the LangChain integration: `ChatOpenAI` is `openai`, even for an Anthropic model behind OpenRouter or another OpenAI-compatible gateway. -Streaming is where tokens go missing. `ChatOpenAI` only asks for usage on streamed responses (`stream_options.include_usage`) when it talks to api.openai.com. With a `base_url`, or `OPENAI_BASE_URL` set, it doesn't, and every streamed call arrives with no token counts. Set `stream_usage=True` on the model, as in the example above. +Streaming is where tokens go missing. `ChatOpenAI` only asks for usage on streamed responses (`stream_options.include_usage`) when it talks to api.openai.com. With a `base_url`, or `OPENAI_BASE_URL` set, it doesn't, and a server that only reports streamed usage when asked, such as vLLM, sends none. OpenRouter includes usage either way. Set `stream_usage=True` on the model, as in the example above, so it doesn't depend on the server. Maple shows cost only when a span carries one, and neither LangChain nor the instrumentor records cost. Sessions show as **unpriced**, with token counts. -A `ChatPromptTemplate` in a chain produces a span named `ChatPromptTemplate`, and Maple currently counts it as a model call because the name contains "chat". It has no model or tokens, so totals are right, but the call count is one too high per template. `create_agent` and LangGraph nodes that call a model directly don't use templates. - Don't add `openinference-instrumentation-openai`, `-anthropic` or OpenLLMetry's LangChain instrumentor next to this one. Each wraps the same model request again, and every call gets a second model span with its own tokens. ## Flush spans before the process exits @@ -224,51 +230,21 @@ Context survives LangGraph's parallel nodes, `ainvoke`, and agents called from i ## LangGraph Server deployments -On LangGraph's Agent Server (`langgraph dev` or a self-hosted server), the server imports the module that defines your graph, the one `langgraph.json` points to. Import `tracing` at the top of that module and set the `OTEL_*` variables in the server's environment (the `env` file in `langgraph.json`, or the container). The server is long-running, so `BatchSpanProcessor` exports on its own schedule and no flush is needed. +On LangGraph's Agent Server (`langgraph dev` or a self-hosted server), the server imports the module that defines your graph, the one `langgraph.json` points to. Import `tracing` at the top of that module and set the `OTEL_*` variables in the server's environment (the `env` file in `langgraph.json`, or the container). The server is long-running, so `BatchSpanProcessor` exports on its own schedule and no flush is needed. We checked this with `langgraph dev` (langgraph-api 0.10.3). -Every run on a server thread already carries that thread's id as `configurable.thread_id`, so each LangGraph thread becomes one Maple session with no extra code. +Every run on a server thread already carries that thread's id as `configurable.thread_id`, so each LangGraph thread becomes one Maple session with no extra code. If `tracing.py` lives outside the graph's directory, list its directory in `dependencies` in `langgraph.json` so the server can import it. -## LangSmith's OpenTelemetry exporter instead +## Why not LangSmith's OpenTelemetry exporter -LangSmith's SDK can write its runs as OTLP spans (`LANGSMITH_OTEL_ENABLED`). Maple recognizes those spans and labels them **LangChain**, and reads the session from `langsmith.metadata.thread_id`, the same `configurable.thread_id`. It's the only path that shows the framework name, and it records `gen_ai.tool.call.id`. It gives Maple less to work with everywhere else: +LangSmith's SDK can write its runs as OTLP spans (`LANGSMITH_OTEL_ENABLED` with `LANGSMITH_OTEL_ONLY`, no LangSmith account needed). Maple recognizes those spans, labels them **LangChain**, and reads the session from `langsmith.metadata.thread_id`, the same `configurable.thread_id`. Sessions, token counts and `gen_ai.tool.call.id` all arrive. We ran the same chat through it with `langsmith` 0.14.1, and the session page is worse on every other count: -- The prompt and completion are LangChain's serialized objects (`{"lc":1,"type":"constructor",...}`) in `gen_ai.prompt` and `gen_ai.completion`, so the transcript is one raw JSON blob per model call, with no turn labels. -- An interrupt marks the interrupted node's span `ERROR`, with `GraphInterrupt(...)` as an `exception` event, so human-in-the-loop pauses read as failures. -- Middleware wrappers such as `HumanInTheLoopMiddleware.wrap_tool_call` get their own spans, and Maple counts them as extra tool calls. A failing tool's error passes through the ones inside your error handler, so one failure counts more than once. -- Prompt templates get `gen_ai.operation.name` `chat`, and `gen_ai.system` is guessed from the model name: `anthropic/claude-haiku-4.5` through OpenRouter reads as `anthropic`. -- There are no agent names. `create_agent`'s name is only in `langsmith.metadata.lc_agent_name`, which Maple doesn't read, and the `AgentSpans` trick doesn't work because LangSmith overwrites `gen_ai.operation.name` after the span starts. - -If you want it anyway, set the variables and the provider before anything imports LangChain. LangSmith looks for a global provider when its client is created, and if there isn't one it builds its own, pointed at smith.langchain.com: - -```py -import os - -os.environ["LANGSMITH_TRACING"] = "true" -os.environ["LANGSMITH_OTEL_ENABLED"] = "true" -os.environ["LANGSMITH_OTEL_ONLY"] = "true" # OTLP only, no LangSmith account or key needed - -from opentelemetry import trace -from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExporter -from opentelemetry.sdk.trace import TracerProvider -from opentelemetry.sdk.trace.export import BatchSpanProcessor - -provider = TracerProvider() -provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) -trace.set_tracer_provider(provider) - -# Only now import langchain, langgraph and your agents -``` - -Install `langsmith[otel]>=0.14`. Flushing takes two steps, because LangSmith converts runs to spans on a background thread and the provider has nothing to export until it's done. `provider.force_flush()` on its own exports nothing: - -```py -from langchain_core.tracers.langchain import wait_for_all_tracers - -wait_for_all_tracers() -provider.force_flush() -``` +- The prompt and completion are sent as byte attributes holding LangChain's serialized objects. Maple stores bytes as hex, so the transcript is unreadable and turns have no labels. +- An interrupt marks the interrupted node's span `ERROR`, with `GraphInterrupt(...)` as an `exception` event, so every human-in-the-loop pause reads as a failure. +- Middleware wrappers such as `HumanInTheLoopMiddleware.wrap_tool_call` get their own spans, and Maple counts them, and the `tools` node, as extra tool calls. A tool the model called once shows up to three times. +- There are no agent names. `create_agent`'s name is only in `langsmith.metadata.lc_agent_name`, which Maple doesn't read, and a start-time processor can't fix it, because LangSmith sets `gen_ai.operation.name` after the span starts. +- Flushing takes two steps: `wait_for_all_tracers()` from `langchain_core.tracers.langchain`, then `provider.force_flush()`. The provider alone exports nothing, because LangSmith converts runs to spans on a background thread. -Don't combine it with the OpenInference instrumentor; you'd get every run twice. +If `LANGSMITH_OTEL_ENABLED` or `LANGSMITH_TRACING_MODE=otel` is already set in your app, remove it when you add the OpenInference setup, or every run arrives twice. ## LangChain.js and LangGraph.js @@ -293,11 +269,11 @@ If a turn is missing, check that the process flushed. - **No spans at all.** `instrument()` never ran, or ran with a different provider than the one exporting. Import `tracing` first, pass `tracer_provider=provider`, and look for `OTLPSpanExporter` errors in the logs. - **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. - **One session per message.** No `thread_id` reached the run, or it changes per request. Pass `{"configurable": {"thread_id": conversation_id}}` to every `invoke()`, `stream()` and resume, or `{"metadata": {"thread_id": ...}}` for plain chains. -- **Sessions work but the session page is a JSON blob, or tokens show only in the list.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)`. +- **One session per message and a raw JSON transcript, even with a `thread_id`.** The GenAI dual-write is off, so Maple only sees OpenInference's `session.id`, which it doesn't read for these spans. Pass `TraceConfig(enable_genai_semconv=True)`. - **Streamed replies have no tokens.** `ChatOpenAI` with a custom `base_url` doesn't request streamed usage. Set `stream_usage=True`. - **One failing tool ends the whole run.** `create_agent` re-raises tool exceptions. Add the `wrap_tool_call` middleware, or `handle_tool_errors=True` on your `ToolNode`. - **A tool shows up as an agent, not a tool call.** Its name contains "agent". Rename it. -- **Twice as many tool calls as the agent made.** The `tools` node span is counted as a tool call, as described above. It's a known gap in how Maple classifies LangGraph node spans. +- **More tool calls than the agent made.** A graph node whose name contains "tool" is counted as a tool call. Add its name to `STEP_NAMES`. - **Every model call appears twice.** A provider instrumentor or `LANGSMITH_OTEL_ENABLED` is also active. Keep one. - **No lanes for sub-agents.** Their names aren't in `AGENT_NAMES`, or two agents share a name. - **Turns split into several traces on Python 3.10.** Pass `config` to nested async calls, or upgrade to Python 3.11. diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md index 725d36efaa..e60342ebac 100644 --- a/apps/landing/src/content/docs/agent-tracing/litellm.md +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -7,9 +7,9 @@ navLabel: "LiteLLM" icon: "litellm" --- -LiteLLM traces model calls, and only model calls. Its OpenTelemetry logger writes one span per `acompletion()`, with the model, the provider, token counts and the prompt and reply. LiteLLM has no agent loop, no tool executor and no notion of a conversation, so the loop around those calls is your code, and the agent and tool spans have to come from your code too. This guide shows the 60 lines of Python that add them. +LiteLLM traces model calls, and only model calls. Its OpenTelemetry logger writes one span per `acompletion()`, with the model, the provider, token counts and the prompt and reply. LiteLLM has no agent loop, no tool executor and no notion of a conversation, so the loop around those calls is your code, and the agent and tool spans have to come from your code too. This guide shows the Python that adds them. -Two defaults go wrong for Maple. The default (v1) logger has no conversation id at all, so every model call lands in its own one-call session, and when your code already has a span open, v1 writes its attributes onto that span after it has ended and they are dropped. LiteLLM's newer v2 logger fixes both and turns `litellm_session_id` into `gen_ai.conversation.id`, but it is off by default, only traces the async API, and in LiteLLM 1.103 it silently disables itself on OpenTelemetry 1.44 or newer. +Two defaults go wrong for Maple. The default (v1) logger has no conversation id at all, so every model call lands in its own one-call session, and when your code already has a span open, v1 writes its attributes onto that span after it has ended and they are dropped. LiteLLM's newer v2 logger fixes both and turns `litellm_session_id` into `gen_ai.conversation.id`, but it is off by default, only traces the async API, and in LiteLLM 1.103 it can't load on OpenTelemetry 1.44 or newer. This guide covers the LiteLLM Python SDK and the self-hosted LiteLLM Proxy. It was tested with `litellm` 1.103.0 and the OpenTelemetry Python SDK 1.43.0 on Python 3.12. @@ -44,7 +44,7 @@ Install LiteLLM with the OpenTelemetry SDK and the OTLP/HTTP exporter: pip install "litellm==1.103.0" "opentelemetry-sdk==1.43.0" "opentelemetry-exporter-otlp-proto-http==1.43.0" ``` -Keep OpenTelemetry below 1.44 on LiteLLM 1.103. OpenTelemetry 1.44 removed the Events API that LiteLLM's v2 logger imports, and LiteLLM catches the import error, logs `Error initializing custom logger: No module named 'opentelemetry._events'` once, and exports nothing ([BerriAI/litellm#41990](https://github.com/BerriAI/litellm/issues/41990)). The fix is merged and ships in LiteLLM 1.104; from that release on you can drop the pin. +Keep OpenTelemetry below 1.44 on LiteLLM 1.103, the latest stable release as of September 2026. OpenTelemetry 1.44 removed the Events API that LiteLLM's v2 logger imports ([BerriAI/litellm#41990](https://github.com/BerriAI/litellm/issues/41990)). Importing `OpenTelemetryV2` as below then fails with `ModuleNotFoundError: No module named 'opentelemetry._events'`. The `callbacks: ["otel"]` form the proxy uses is worse: LiteLLM catches the error, logs `Error initializing custom logger` and keeps serving requests without exporting anything. The fix is merged, and the 1.104.0rc1 release candidate traces correctly with OpenTelemetry 1.45. Drop the pin once 1.104 is out. Point the exporter at Maple with the standard OpenTelemetry variables: @@ -100,6 +100,8 @@ Each `acompletion()` becomes a `chat ` span. To get turns, sub-agents and ```py # agent.py +import asyncio +import inspect import json from contextlib import contextmanager from dataclasses import dataclass, field @@ -115,7 +117,7 @@ class Agent: name: str model: str instructions: str - tools: dict = field(default_factory=dict) # tool name -> function + tools: dict = field(default_factory=dict) # tool name -> function (sync or async) schemas: list = field(default_factory=list) # OpenAI-style tool definitions @@ -127,7 +129,7 @@ def agent_span(name: str): yield span -def run_tool(agent: Agent, call) -> str: +async def run_tool(agent: Agent, call) -> str: name = call.function.name with tracer.start_as_current_span(f"execute_tool {name}") as span: span.set_attribute("gen_ai.operation.name", "execute_tool") @@ -136,6 +138,8 @@ def run_tool(agent: Agent, call) -> str: span.set_attribute("gen_ai.tool.call.arguments", call.function.arguments) try: result = agent.tools[name](**json.loads(call.function.arguments or "{}")) + if inspect.isawaitable(result): + result = await result except Exception as exc: # The model gets the error as the tool result; the span is marked failed. span.set_status(StatusCode.ERROR, str(exc)) @@ -159,8 +163,10 @@ async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: messages.append(message.model_dump(exclude_none=True)) if not message.tool_calls: return message.content or "" - for call in message.tool_calls: - messages.append({"role": "tool", "tool_call_id": call.id, "content": run_tool(agent, call)}) + # Parallel tool calls run concurrently; each keeps the agent span as its parent. + results = await asyncio.gather(*(run_tool(agent, call) for call in message.tool_calls)) + for call, output in zip(message.tool_calls, results): + messages.append({"role": "tool", "tool_call_id": call.id, "content": output}) ``` The v2 logger only traces the async API. `litellm.completion()` and the other sync calls produce no span at all, because the logger closes spans in its async success callback. Use `acompletion()`. If your code is sync throughout, see the troubleshooting entry on sync code. @@ -192,6 +198,8 @@ Leave `gen_ai.conversation.id` off your own `invoke_agent` span. Maple labels a The v2 logger records no content by default. `capture_message_content="span_only"` in `tracing.py` turns it on, and each `chat` span then carries `gen_ai.input.messages` and `gen_ai.output.messages` as JSON in the OpenAI chat format: the system prompt, the history, tool calls and tool results. Maple builds the transcript from those attributes. The environment variable `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only` does the same when LiteLLM builds the config itself, as the proxy does. +One gap in the transcript today: Maple doesn't read the OpenAI `tool_calls` field inside these messages. A model call whose reply is only a tool request shows no reply text, and the tool call itself appears once, from your `execute_tool` span, with its arguments and result. Without the `execute_tool` spans below, tool calls would be missing from the transcript entirely. + Don't use `event_only`. It moves content to OpenTelemetry log events, which Maple doesn't read, and the transcript is empty. To keep prompts out of Maple, set `capture_message_content="no_content"` and remove the `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result` lines from `run_tool`. Sessions keep their turns, models, tool names, tokens and errors, and the transcript is empty. @@ -206,27 +214,48 @@ Each tool call is an `execute_tool ` span with the tool name, the mod A failed model call needs nothing from you. LiteLLM marks its `chat` span ERROR and sets `error.type` to the exception class, for example `RateLimitError`. -For several agents, give each its own `name`. It becomes `gen_ai.agent.name` on its `invoke_agent` span, and Maple draws one lane per agent name. Run the workers inside an outer agent span so the whole run is one trace, and pass the same conversation id to every call: +For several agents, give each its own `name`. It becomes `gen_ai.agent.name` on its `invoke_agent` span, and Maple draws one lane per agent name. The simplest orchestrator is an agent whose tools are the other agents: each tool function calls `run_agent` with the same conversation id, so the whole run is one trace in one session: ```py -import asyncio +from agent import Agent, run_agent -from agent import Agent, agent_span, run_agent +# weather_worker, budget_worker, transport_worker and summary are Agent(...) instances +WORKERS = (weather_worker, budget_worker, transport_worker, summary) -# weather_worker, transport_worker and summary are Agent(...) instances with their own tools and models +def delegate(worker: Agent, conversation_id: str): + async def call(task: str) -> str: + return await run_agent(worker, conversation_id, [{"role": "user", "content": task}]) -async def briefing(conversation_id: str, city: str) -> str: - with agent_span("orchestrator"): - weather, transport = await asyncio.gather( - run_agent(weather_worker, conversation_id, [{"role": "user", "content": f"Weather in {city}?"}]), - run_agent(transport_worker, conversation_id, [{"role": "user", "content": f"Transport in {city}?"}]), - ) - notes = f"Weather: {weather}\nTransport: {transport}" - return await run_agent(summary, conversation_id, [{"role": "user", "content": notes}]) + return call + + +def task_tool(name: str) -> dict: + return { + "type": "function", + "function": { + "name": name, + "description": f"Delegate a task to the {name} agent.", + "parameters": {"type": "object", "properties": {"task": {"type": "string"}}, "required": ["task"]}, + }, + } + + +async def briefing(conversation_id: str, request: str) -> str: + orchestrator = Agent( + "orchestrator", + "openrouter/openai/gpt-4o-mini", + "Call weather_worker, budget_worker and transport_worker in parallel, " + "then call summary once with their results, then reply with the summary.", + {w.name: delegate(w, conversation_id) for w in WORKERS}, + [task_tool(w.name) for w in WORKERS], + ) + return await run_agent(orchestrator, conversation_id, [{"role": "user", "content": request}]) ``` -`asyncio.gather` keeps the OpenTelemetry context, so the parallel workers are siblings under `invoke_agent orchestrator` and overlap in time. If a worker delegates through a tool call instead, call `run_agent` inside that tool's function. An `execute_tool` span whose only child is an `invoke_agent` span shows as a delegation, with the tool's arguments and result as the lane's input and output. +Each delegation is an `execute_tool weather_worker` span whose only child is `invoke_agent weather_worker`, which Maple shows as a sub-agent lane with the tool's arguments and result as the lane's input and output. When the model requests several workers in one reply, `asyncio.gather` in `run_agent` runs them in parallel, and they overlap in time as siblings under `invoke_agent orchestrator`. + +If your code calls the workers directly instead of through the model, wrap those calls in `with agent_span("orchestrator"):` so they still share one trace. `asyncio.gather` keeps the OpenTelemetry context, so workers started with it nest correctly. Thread pools don't: wrap work sent to `run_in_executor` or a `ThreadPoolExecutor` with `contextvars.copy_context().run`. ## Tokens and cost @@ -253,30 +282,55 @@ async def stream_agent(agent: Agent, conversation_id: str, messages: list): messages.append({"role": "assistant", "content": text}) ``` -Cost shows as unpriced by default. LiteLLM prices every call, but the v2 logger writes the price to `litellm.cost.total` (and v1 buries it in the `hidden_params` JSON), and Maple reads cost only from `gen_ai.usage.cost` and never prices tokens itself. To get cost into Maple, add up LiteLLM's price per agent run and put it on the `invoke_agent` span: +Cost shows as unpriced by default. LiteLLM prices every call, but the v2 logger writes the price to `litellm.cost.total` (and v1 buries it in the `hidden_params` JSON), and Maple reads cost only from `gen_ai.usage.cost` and never prices tokens itself. + +To get cost into Maple, add up LiteLLM's price for every call of a turn, sub-agents included, and put the total on the outermost `invoke_agent` span. Only the outermost one: Maple subtracts a sub-agent's reported cost from the agent above it, on the assumption that the parent's figure already includes it, so a per-agent total on every level undercounts the orchestrator. Replace `agent_span` in `agent.py`: ```py -async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: - with agent_span(agent.name) as span: - cost = 0.0 - while True: - response = await litellm.acompletion( - model=agent.model, - messages=[{"role": "system", "content": agent.instructions}, *messages], - tools=agent.schemas or None, - litellm_session_id=conversation_id, - ) - cost += response._hidden_params.get("response_cost") or 0.0 - message = response.choices[0].message - messages.append(message.model_dump(exclude_none=True)) - if not message.tool_calls: - span.set_attribute("gen_ai.usage.cost", cost) - return message.content or "" - for call in message.tool_calls: - messages.append({"role": "tool", "tool_call_id": call.id, "content": run_tool(agent, call)}) +from contextvars import ContextVar + +# LiteLLM's price for every model call of the current turn, sub-agents included. +turn_costs: ContextVar[list | None] = ContextVar("turn_costs", default=None) + + +@contextmanager +def agent_span(name: str): + with tracer.start_as_current_span(f"invoke_agent {name}") as span: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", name) + if turn_costs.get() is not None: # a sub-agent: the outermost agent reports the cost + yield span + return + costs: list[float] = [] + token = turn_costs.set(costs) + try: + yield span + finally: + turn_costs.reset(token) + span.set_attribute("gen_ai.usage.cost", sum(costs)) +``` + +Then record each call's price. In `run_agent`, right after `acompletion` returns: + +```py + turn_costs.get().append(response._hidden_params.get("response_cost") or 0.0) ``` -The session total is then correct, but the cost sits on the agent, not on each model call, so the per-model breakdown stays unpriced. For a stream, the price is on the last chunk's `usage.cost`. +A stream carries its price on the last chunk's `usage.cost`. In `stream_agent`, keep it while you consume the stream and record it once at the end: + +```py + text, cost = "", 0.0 + async for chunk in stream: + if getattr(chunk, "usage", None) is not None: # the last chunk carries usage and cost + cost = getattr(chunk.usage, "cost", None) or cost + delta = chunk.choices[0].delta.content if chunk.choices else None + if delta: + text += delta + yield delta + turn_costs.get().append(cost) +``` + +The session and turn totals are then exact. The cost sits on the agent rather than on each model call, so the per-model breakdown stays unpriced. ## Trace at the LiteLLM Proxy @@ -305,7 +359,13 @@ OTEL_ENVIRONMENT_NAME=production OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only ``` -The official `ghcr.io/berriai/litellm` image ships OpenTelemetry 1.28 and the FastAPI instrumentation, so the 1.44 problem above doesn't apply to it. A pip-installed proxy needs `opentelemetry-sdk`, `opentelemetry-exporter-otlp-proto-http` and `opentelemetry-instrumentation-fastapi` (1.43.0 and 0.64b0 with LiteLLM 1.103). +The official `ghcr.io/berriai/litellm` image ships OpenTelemetry 1.28 and the FastAPI instrumentation, so the 1.44 problem above doesn't apply to it. A pip-installed proxy needs the same pin plus the FastAPI instrumentation, which is what continues your app's trace: + +```bash +pip install "litellm[proxy]==1.103.0" "opentelemetry-sdk==1.43.0" \ + "opentelemetry-exporter-otlp-proto-http==1.43.0" "opentelemetry-instrumentation-fastapi==0.64b0" +litellm --config config.yaml +``` In your app, keep `agent_span` and `run_tool` from above, drop the LiteLLM logger from `tracing.py`, and send two headers with every request: `traceparent`, so the proxy's spans join the app's trace under the agent span, and `x-litellm-session-id`, which the proxy turns into `gen_ai.conversation.id`: @@ -330,7 +390,9 @@ async def call_model(conversation_id: str, messages: list, tools: list | None): Don't add an OpenAI client instrumentor (OpenInference, OpenLLMetry, `opentelemetry-instrumentation-openai-v2`) to the app on top of this. The app's span and the proxy's span describe the same call, and Maple can't always tell them apart, so LLM call counts and tokens double. If you can't configure the proxy, do the opposite: leave its OpenTelemetry off and instrument the client in the app. -The proxy also emits its own housekeeping spans (database, Redis, guardrails) in the same trace. They carry no model or tokens, and appear in the trace view next to the `chat` span. +Each request then shows up in your app's trace as `invoke_agent` → `POST /chat/completions` (the proxy's server span) → `chat gpt-4o-mini`, next to an `auth /chat/completions` span for the proxy's key check. Depending on its setup, the proxy adds more housekeeping spans (database, Redis, guardrails). + +Maple currently counts each `auth /chat/completions` span as an extra LLM call, because it comes from LiteLLM and its name contains "chat". On the proxy path the LLM call count in Agent Sessions is double the real number. Tokens, cost, the transcript and the session grouping are not affected. ## Short-lived processes @@ -368,16 +430,16 @@ Call `flush_tracing()` at the end of every Lambda or Cloud Run job invocation, a Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: - one session per conversation, with the id you passed as `litellm_session_id`, and framework **LiteLLM**; -- one turn per `invoke_agent` span, labeled with the user's message, and a transcript with the prompts, replies and tool calls; +- one turn per top-level `invoke_agent` span, labeled with the user's message, and a transcript with the prompts, replies and tool calls; - `invoke_agent ` spans from your code, `chat ` spans from LiteLLM inside them, and `execute_tool ` spans for tool calls; - a lane per agent name for multi-agent runs, all in the caller's session; - input and output tokens on every `chat` span, including streamed ones; - failed tool calls marked as failed, with the error as the result; -- cost shown as unpriced, or the per-run total if you added `gen_ai.usage.cost`. +- cost shown as unpriced, or the turn totals if you added `gen_ai.usage.cost`. ## Troubleshooting -- **No LiteLLM spans at all, and the log shows `No module named 'opentelemetry._events'`.** LiteLLM 1.103 with OpenTelemetry 1.44 or newer. Pin `opentelemetry-sdk` and the exporter to 1.43.0, or upgrade to LiteLLM 1.104. +- **`ModuleNotFoundError: No module named 'opentelemetry._events'` at startup, or no LiteLLM spans and a logged `Error initializing custom logger`.** LiteLLM 1.103 with OpenTelemetry 1.44 or newer. Pin `opentelemetry-sdk` and the exporter to 1.43.0, or upgrade to LiteLLM 1.104 once it's released. - **Your spans arrive but no `chat` spans.** The code calls the sync `litellm.completion()`, which the v2 logger doesn't trace. Switch to `acompletion()`. - **Spans named `litellm_request` and `raw_gen_ai_request`, and each call is its own session.** That is the v1 logger: `litellm.callbacks = ["otel"]` without the v2 instance. Use the `OpenTelemetryV2` setup above. - **Model, tokens and prompts missing, and the SDK logs `Setting attribute on ended span`.** Also v1: with a span already open it writes onto that span instead of creating its own. If you must stay on v1 for sync code, set `USE_OTEL_LITELLM_REQUEST_SPAN=true` and `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental`, and put `gen_ai.conversation.id` on your `invoke_agent` span, since v1 can't carry it. The framework then shows as **Unidentified**. @@ -387,6 +449,9 @@ Run one conversation with at least two messages and a tool call, then open **Age - **The last call of a script or Lambda is missing.** The process ended before LiteLLM's logging queue ran. Await `flush_tracing()` before the loop ends. - **Every model call appears twice.** Two loggers (the v2 instance plus `"otel"` in `litellm.callbacks`), or the proxy and an in-app OpenAI instrumentor both tracing the same call. Keep one. - **Proxy spans land in their own traces, apart from the app's agent span.** The request carried no `traceparent`. Inject it with `propagate.inject(headers)` inside the agent span, and don't set `OTEL_IGNORE_CONTEXT_PROPAGATION` on the proxy. +- **The LLM call count is twice what the app made, on the proxy path only.** Maple counts the proxy's `auth /chat/completions` spans as calls. Token and cost totals are correct. +- **A model call shows an empty reply in the transcript.** That call only requested tools. The tool calls appear as their own rows from your `execute_tool` spans. +- **Session cost is lower than LiteLLM's spend.** Sub-agents report their own `gen_ai.usage.cost` and Maple subtracts it from the parent agent's. Report cost only on the outermost agent span, as in the cost section. - **Nothing arrives from the proxy.** It is still on the default `console` exporter because no endpoint reached it. Check that `OTEL_EXPORTER_OTLP_ENDPOINT` is set in the proxy's environment, not only in your app's. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/llamaindex.md b/apps/landing/src/content/docs/agent-tracing/llamaindex.md index 5bc3fc98ad..432bce860b 100644 --- a/apps/landing/src/content/docs/agent-tracing/llamaindex.md +++ b/apps/landing/src/content/docs/agent-tracing/llamaindex.md @@ -9,7 +9,9 @@ icon: "llamaindex" LlamaIndex reports what it does through its own instrumentation dispatcher: every agent run, workflow step, model call and tool call opens a dispatcher span and fires events. Two packages turn those into OpenTelemetry spans. LlamaIndex's own `llama-index-observability-otel` copies the span tree but puts the payload in span events, and it loses the model's reply and token usage on every streamed call. OpenInference's `openinference-instrumentation-llama-index` writes model, tokens, messages and tool results as span attributes, which is what Maple reads. Use the OpenInference one. -It needs four adjustments before a chat looks right in Maple. The instrumentor writes OpenInference attribute names unless you turn on its GenAI output, it never sets a conversation id, it doesn't record agent names, and it opens two or three "LLM" spans for every model call, so Maple would count each call two or three times. This guide covers all four for llama-index-core 0.14.25 with `openinference-instrumentation-llama-index` 4.5.2 on Python 3.10 or later, using `FunctionAgent`, `AgentWorkflow` and custom `Workflow` classes. +It needs four adjustments before a chat looks right in Maple. The instrumentor writes OpenInference attribute names unless you turn on its GenAI output, it never sets a conversation id, it doesn't record agent names, and it opens extra spans around every model and tool call, so Maple would count each model call two or three times and each tool call three times. + +This guide covers all four for llama-index-core 0.14.25 with `openinference-instrumentation-llama-index` 4.5.2 on Python 3.10 or later, using `FunctionAgent`, `AgentWorkflow` and custom `Workflow` classes. ## Quick setup with a coding agent @@ -72,17 +74,22 @@ LLM_METHODS = (".chat", ".achat", ".stream_chat", ".astream_chat", class LlamaIndexForMaple(SpanProcessor): - """Sits in front of the exporter: one span per model call, agent names, no false HITL failures.""" + """Sits in front of the exporter: one span per model and tool call, agent names, no false HITL failures.""" def __init__(self, exporter_processor: SpanProcessor): self._next = exporter_processor self._open_llm_spans = {} def on_start(self, span, parent_context=None): + # Maple labels a span "LlamaIndex" by a llamaindex.* key; OpenInference writes none + span.set_attribute("llamaindex.instrumentor", "openinference") # instrument_tags({"gen_ai.agent.name": ...}) becomes an attribute, so sub-agents get lanes agent_name = active_instrument_tags.get().get("gen_ai.agent.name") if agent_name: span.set_attribute("gen_ai.agent.name", agent_name) + if span.name.endswith((".call_tool", ".aggregate_tool_results")): + # agent workflow steps around the tool span; by name alone Maple would count them as tool calls + span.set_attribute("gen_ai.operation.name", "invoke_workflow") if span.name.endswith(LLM_METHODS): self._open_llm_spans[span.context.span_id] = span self._next.on_start(span, parent_context) @@ -119,10 +126,10 @@ LlamaIndexInstrumentor().instrument( What each part does: - **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.conversation.id` next to its OpenInference attributes when each span ends. Without it, Maple can count tokens but the session page has no transcript. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, but only if it's set before `TraceConfig` is built. -- **`LlamaIndexForMaple`** wraps the exporting processor and fixes what the instrumentor gets wrong for LlamaIndex. The next sections explain each fix. +- **`LlamaIndexForMaple`** wraps the exporting processor and fixes what the instrumentor gets wrong for LlamaIndex. The next sections explain each fix. It also stamps `llamaindex.instrumentor` on every span, because Maple recognises LlamaIndex by a `llamaindex.*` attribute and the OpenInference spans carry none. Without it the sessions still work but show the framework as "Unidentified". - **One `TracerProvider`.** If the app already has one (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second. Add `LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))` to the existing provider and pass that provider to `instrument()`. -### Why each model call opens three spans +### Why each model and tool call opens three spans LlamaIndex's dispatcher wraps every method that implements an abstract base method, and the instrumentor marks every span on an LLM object as kind `LLM`. A single `FunctionAgent` step with an OpenAI-compatible model (`OpenRouter`, `OpenAILike` and the others built on it) produces: @@ -133,10 +140,20 @@ BaseWorkflowAgent.run_agent_step └── OpenRouter.astream_chat kind LLM, OpenAI's method: messages, usage ``` -Maple counts every inference span without usage from a reporting ancestor as a model call, so this one call would count three times. A class that implements the call itself, such as `OpenAI`, produces two spans, the helper and the call. Tokens are not doubled, since only the innermost span carries usage. +Maple counts every inference span without usage from a reporting ancestor as a model call, so this one call would count three times; in a test run, three model calls counted as nine. A class that implements the call itself, such as `OpenAI`, produces two spans, the helper and the call. Tokens are not doubled, since only the innermost span carries usage. `LlamaIndexForMaple` drops the `_prepare_chat_with_tools` helper, copies the inner span's attributes onto its same-named parent and drops the inner span, leaving one `OpenRouter.astream_chat` span per call with the messages and usage. It only merges spans with identical names that nest directly, so a chat engine's `CondensePlusContextChatEngine.chat` calling `OpenAI.chat` is left alone. +Tool calls have the same problem one level up: + +```text +BaseWorkflowAgent.call_tool kind CHAIN, workflow step +└── FunctionTool.acall kind TOOL, the tool call +BaseWorkflowAgent.aggregate_tool_results kind CHAIN, workflow step +``` + +The two step spans have no GenAI operation, and Maple classifies such spans by name, so "tool" in the name makes each of them a tool call. `LlamaIndexForMaple` marks them `gen_ai.operation.name=invoke_workflow`, which Maple reads as agent work, leaving `FunctionTool.acall` as the only tool call. + ## Group a conversation into one session LlamaIndex keeps a conversation in a workflow `Context` (`Context(agent)`, or the chat memory you pass), and every `agent.run()` starts a new root span and a new trace. Nothing in those spans says which conversation they belong to, so without a conversation id every message becomes its own one-turn session in Maple, named after its trace id. @@ -228,11 +245,11 @@ class Briefing(Workflow): ctx.send_event(WeatherTask(city=ev.city)) ctx.send_event(TransportTask(city=ev.city)) - @step(num_workers=2) + @step async def weather(self, ev: WeatherTask) -> WorkerDone: return WorkerDone(text=await run_agent(weather_worker, f"Weather in {ev.city}?")) - @step(num_workers=2) + @step async def transport(self, ev: TransportTask) -> WorkerDone: return WorkerDone(text=await run_agent(transport_worker, f"Transport in {ev.city}?")) @@ -244,11 +261,11 @@ class Briefing(Workflow): return StopEvent(result=await run_agent(summary_agent, "\n\n".join(d.text for d in done))) ``` -Run the whole workflow inside `using_session(conversation_id)`. The workflow is one trace: `Briefing.run` at the root, one span per step, and each worker's `FunctionAgent.run` under its step with its own `gen_ai.agent.name`. Maple opens a lane for every agent span whose name differs from its caller's. Parallel steps (`num_workers`) stay in the same trace and keep the session and agent tags. +Run the whole workflow inside `using_session(conversation_id)`. The workflow is one trace: `Briefing.run` at the root, one span per step, and each worker's `FunctionAgent.run` under its step with its own `gen_ai.agent.name`. Maple opens a lane for every agent span whose name differs from its caller's. `weather` and `transport` run concurrently because `plan` sends both events at once; parallel steps, including `@step(num_workers=N)`, stay in the same trace and keep the session and agent tags. The same works for agents called as tools: put `instrument_tags` inside the tool function around the sub-agent's `run()`. The tool span with one `FunctionAgent.run` child then shows as a delegation, with the tool's arguments and result as the lane's input and output. -`AgentWorkflow` handoffs are different. The whole multi-agent run is one `AgentWorkflow.run` span, and LlamaIndex switches the active agent inside it without a span per agent, so there is nothing to tag. Handoffs show as one agent in Maple. If you need lanes, run each agent as its own `FunctionAgent.run()` from a workflow step or a tool, as above. +`AgentWorkflow` handoffs are different. The whole multi-agent run is one `AgentWorkflow.run` span, and LlamaIndex switches the active agent inside it without a span per agent, so there is nothing to tag. Every span carries the tag you put around `run()`, the handoff itself is a `handoff` tool call, and the run shows as one agent in Maple. If you need lanes, run each agent as its own `FunctionAgent.run()` from a workflow step or a tool, as above. ## Tokens and cost @@ -266,7 +283,7 @@ LlamaIndex strips the option from non-streaming requests, so it's safe to set on The provider comes from the model class: `OpenRouter` and every `OpenAILike` model report `openai`, even for an Anthropic model behind OpenRouter. -Maple shows cost only when a span carries one, and neither LlamaIndex nor the instrumentor records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call, and Maple joins them to the same session. +Maple shows cost only when a span carries one, and neither LlamaIndex nor the instrumentor records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call, and Maple joins them to the same session. These model spans have no `gen_ai.response.id`, so nest the Broadcast spans under them as described in [Join Broadcast to your own traces](/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces), or each call is counted twice. Don't add `openinference-instrumentation-openai` (or another provider instrumentor) next to the LlamaIndex one. It wraps the same HTTP call and gives every model call a second span with its own usage. @@ -290,10 +307,10 @@ Call `provider.shutdown()` instead when the process is about to exit and won't t Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: - **One session** for the conversation, with one turn per `agent.run()`. Each turn's trace starts at `FunctionAgent.run` (or `AgentWorkflow.run`, or your workflow class's `.run`). -- **Framework: Unidentified.** Maple doesn't recognise the OpenInference LlamaIndex scope as LlamaIndex yet, so the session is filed under the generic GenAI bucket. Everything else on this list still works. +- **Framework: LlamaIndex**, from the `llamaindex.instrumentor` attribute `LlamaIndexForMaple` adds. - **The transcript**: your messages, the model's replies and the tool calls it made. - **Model calls** named after your model class and method, for example `OpenRouter.astream_chat` or `OpenAI.achat`, one per call, each with a model and input and output tokens. -- **Tool calls** named `FunctionTool.acall`, with the tool's name and result. +- **Tool calls** named `FunctionTool.acall`, one per call, with the tool's name and result. `BaseWorkflowAgent.call_tool` and `aggregate_tool_results` are agent steps, not tool calls. - **Agents**: `assistant`, plus one lane per tagged sub-agent. - **Cost**: unpriced. @@ -306,7 +323,8 @@ A second conversation with a different id is a second session. If a turn is miss - **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. - **Tokens in the list, empty session page.** The GenAI output is off. Pass `TraceConfig(enable_genai_semconv=True)`. - **One session per message.** `agent.run()` isn't inside `using_session(...)`, or the id changes per request. `Context` and `llamaindex.run_id` are not session ids. -- **Each model call counted two or three times.** `LlamaIndexForMaple` isn't in front of the exporter. Add the exporter through it, not directly with `add_span_processor(BatchSpanProcessor(...))`. +- **Each model or tool call counted two or three times.** `LlamaIndexForMaple` isn't in front of the exporter. Add the exporter through it, not directly with `add_span_processor(BatchSpanProcessor(...))`. +- **Framework shows "Unidentified".** The spans lack a `llamaindex.*` attribute: `LlamaIndexForMaple` is missing, or it's an older copy without the `llamaindex.instrumentor` line. - **Model calls take 1 ms.** Streamed model spans end when LlamaIndex hands back the stream, not when the last token arrives, so their duration isn't the model's latency. The enclosing `BaseWorkflowAgent.run_agent_step` span has the real time. If you don't stream tokens to users, `FunctionAgent(..., streaming=False)` records model spans with their full duration. - **No tokens on streamed calls.** The provider didn't send usage on the stream. Add `stream_options={"include_usage": True}` through `additional_kwargs`. - **Tool arguments show the tool's schema.** A known gap in the instrumentor's GenAI output; see [Tools, errors and sub-agents](#tools-errors-and-sub-agents). diff --git a/apps/landing/src/content/docs/agent-tracing/mastra.md b/apps/landing/src/content/docs/agent-tracing/mastra.md index 31e46d1d02..6c6f1eea16 100644 --- a/apps/landing/src/content/docs/agent-tracing/mastra.md +++ b/apps/landing/src/content/docs/agent-tracing/mastra.md @@ -9,7 +9,7 @@ icon: "mastra" Mastra traces itself. Every `agent.generate()` or `agent.stream()` produces an `invoke_agent` span, a `chat` span per model call with the prompt, the reply and the token counts, and an `execute_tool` span per tool call with its arguments and result. The `@mastra/otel-exporter` package turns those spans into OpenTelemetry GenAI spans (semantic conventions v1.38) and sends them to any OTLP endpoint, including Maple. You don't need an OpenTelemetry SDK or an instrumentation package. -Two things go wrong by default. The exporter ignores the standard `OTEL_EXPORTER_OTLP_*` variables and falls back to OTLP/JSON, so a setup copied from another guide exports nothing and logs a single warning. And the conversation id Maple groups by is Mastra's memory thread id: an agent called without `memory: { thread, resource }` sends no id at all, so every message becomes its own session. +Three things go wrong by default. The exporter ignores the standard `OTEL_EXPORTER_OTLP_*` variables and falls back to OTLP/JSON, so a setup copied from another guide exports nothing and logs a single warning. The conversation id Maple groups by is Mastra's memory thread id: an agent called without `memory: { thread, resource }` sends no id at all, so every message becomes its own session. And Mastra 1.71 exports each model call without its prompt, so the transcript shows the replies but none of the user's messages. A short span processor, shown below, fixes the prompt and two smaller gaps. This guide covers Mastra 1.x on Node.js 22.13 or newer. It was written against `@mastra/core` 1.71, `@mastra/observability` 1.18 and `@mastra/otel-exporter` 1.4. @@ -29,13 +29,53 @@ Use your key from **Settings → Ingestion**. Without one, the agent uses a plac ## Export Mastra spans to Maple -Install the observability packages next to `@mastra/core`, plus the OTLP/protobuf trace exporter: +Install the observability packages next to `@mastra/core`: ```bash -npm install @mastra/observability@latest @mastra/otel-exporter@latest @opentelemetry/exporter-trace-otlp-proto +npm install @mastra/observability@latest @mastra/otel-exporter@latest ``` -Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week. The exporter decides which span is the model call from features both packages report, and a stale `@mastra/observability` can leave you with spans but no model calls. +The OTLP exporters, including `@opentelemetry/exporter-trace-otlp-proto`, are optional dependencies of `@mastra/otel-exporter` and install with it. Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week. The exporter decides which span is the model call from features both packages report, so mismatched releases change what Maple counts as a model call. + +### Add the Maple span processor + +Mastra 1.71 has three gaps in what it exports, and one span processor closes all of them: + +- The `chat` span of each model call has the reply but not the prompt. Mastra records the messages on the enclosing step span instead. The processor copies them onto the `chat` span, where the exporter turns them into `gen_ai.input.messages`. +- Sub-agents called by a supervisor get their own thread id, so one trace carries several conversation ids. The processor gives every span the thread id of the trace's root span. +- Step spans carry the provider's raw HTTP response as metadata: response headers, including cookies, and the full response body with the reply text. The processor drops both. + +```ts +// src/mastra/maple-span-processor.ts +import { SpanType, type SpanOutputProcessor } from "@mastra/core/observability" + +export const mapleSpanProcessor: SpanOutputProcessor = { + name: "maple-span-processor", + process(span) { + if (!span) return span + // One conversation id per trace: sub-agents get their own thread ids otherwise. + let root = span + while (root.parent) root = root.parent + const threadId = root.metadata?.threadId + if (threadId) span.metadata = { ...span.metadata, threadId } + // The model call span is created without its prompt: take the step's messages. + if (span.type === SpanType.MODEL_INFERENCE && span.input === undefined && span.parent?.input !== undefined) { + span.input = { messages: span.parent.input } + } + // Step spans carry the raw provider response (headers, cookies, full body) as metadata. + if (span.type === SpanType.MODEL_STEP && span.metadata) { + const { headers: _headers, body: _body, ...metadata } = span.metadata + span.metadata = metadata + } + return span + }, + async shutdown() {}, +} +``` + +A processor has to change the span it receives and return that same object; Mastra drops the span if you return a copy. Mastra's own `SensitiveDataFilter` runs after your processors, so the copied prompt is still redacted. + +### Configure the exporter Then configure observability on your `Mastra` instance: @@ -46,6 +86,7 @@ import { SpanType } from "@mastra/core/observability" import { Observability } from "@mastra/observability" import { OtelExporter } from "@mastra/otel-exporter" import { supportAgent } from "./agents/support" +import { mapleSpanProcessor } from "./maple-span-processor" export const mapleExporter = new OtelExporter({ provider: { @@ -67,6 +108,7 @@ export const mastra = new Mastra({ exporters: [mapleExporter], // One span per streamed chunk adds nothing Maple uses excludeSpanTypes: [SpanType.MODEL_CHUNK], + spanOutputProcessors: [mapleSpanProcessor], }, }, }), @@ -142,7 +184,9 @@ The same `tracingOptions` works on `agent.generate()` for an agent that has no m ## Record prompts, responses and tool calls -Content capture is on by default. Each `chat` span carries `gen_ai.input.messages` and `gen_ai.output.messages` as JSON in the GenAI message format, the `invoke_agent` span carries the agent's instructions as `gen_ai.system_instructions`, and each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Maple builds the transcript from those attributes. +Content capture is on by default. With the span processor in place, each `chat` span carries `gen_ai.input.messages` and `gen_ai.output.messages` as JSON in the GenAI message format, and each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Maple builds the transcript from those attributes. The agent's instructions are the first, `system`, message of every prompt. + +Two details of Mastra's format show up in the transcript. Tool calls from earlier in the conversation appear in later prompts as short `[tool: get_weather]` placeholders; the full arguments and results are on the `execute_tool` spans. And the `invoke_agent` span's `gen_ai.system_instructions` is plain text rather than the JSON the convention asks for, so Maple skips it and shows the instructions from the system message instead. Mastra bounds what it serializes into a span. The defaults are 128 KiB per string, 50 items per array, 50 keys per object and 8 levels deep. A cut string ends in `…[truncated]`, which Maple shows as truncated. A message list longer than 50 entries loses the newest messages, so a turn with a long history can be missing its latest user message. Raise the array limit if your agents keep long histories: @@ -152,6 +196,8 @@ new Observability({ maple: { serviceName: "support-agent", exporters: [mapleExporter], + excludeSpanTypes: [SpanType.MODEL_CHUNK], + spanOutputProcessors: [mapleSpanProcessor], serializationOptions: { maxArrayLength: 200 }, }, }, @@ -167,15 +213,15 @@ await agent.generate(text, { }) ``` -Sessions keep their turns, models, tool names, tokens and errors, but the transcript is empty and tool calls have no arguments or results. +Sessions keep their turns, models, tool names, tokens and errors, but the transcript is empty and tool calls have no arguments or results. These options hide span input and output only, not metadata. Without the span processor, the full provider response, reply included, still leaves in the step span's `mastra.metadata.body`. Mastra also applies a `SensitiveDataFilter` to every span by default. It redacts values under keys like `password`, `token`, `apiKey`, `authorization` and `secret`, including inside JSON strings, and replaces them with `[REDACTED]`. It matches key names, not free text, so a user who types a password into the chat still sends it to Maple. If you need pattern-based redaction of message text, run it in an OpenTelemetry Collector between your app and Maple. ## Tools, errors and sub-agents -Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the model's `gen_ai.tool.call.id`, the tool description, and the arguments and result. Maple matches each call to the model reply that requested it by that id. +Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the model's `gen_ai.tool.call.id`, the tool description, and the arguments and result. The transcript shows each call from its `execute_tool` span. The exporter leaves tool calls out of the `chat` span's output messages, so a model reply that only requests tools shows as an empty assistant message. -A tool that throws is marked failed: status ERROR, `error.type` set to Mastra's error id (`TOOL_EXECUTION_FAILED`), and the exception message as the status message. The agent keeps running and the model sees the error. A tool that returns an error value instead of throwing stays green, and Maple counts the call as a success, so throw for failures you want to see: +A tool that throws is marked failed: status ERROR with the exception message as the status message, `error.type` set to `unknown`, and an `exception` event. The agent keeps running and the model sees the error. A tool that returns an error value instead of throwing stays green, and Maple counts the call as a success, so throw for failures you want to see: ```ts import { createTool } from "@mastra/core/tools" @@ -193,41 +239,25 @@ export const fetchTransportData = createTool({ Tools with `requireApproval: true` suspend the run before the tool executes. The resumed run (`approveToolCallGenerate()` or `declineToolCallGenerate()`) continues the same trace; pass the same `memory` to it. A declined call produces no `execute_tool` span. +Approvals leave one artifact. The model call that asks for the tool is exported twice: once without tokens or reply when the run suspends, and again with its tokens when the resumed run replays it. Maple shows one extra model call with 0 tokens per approval. Token totals are right. + ### Sub-agents: keep one conversation id Mastra's multi-agent idiom is a supervisor: an agent with an `agents` property calls each sub-agent through a tool named `agent-`. Each delegation runs the sub-agent inside the supervisor's trace, under that tool span. Give every agent a `name`; it becomes `gen_ai.agent.name`, and Maple draws one lane per agent name. An `execute_tool agent-weather_worker` span whose only child is `invoke_agent weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. -The catch is memory. When the supervisor runs with a thread, Mastra gives each delegation its own new thread id, so the sub-agent's spans carry a different `gen_ai.conversation.id` from the conversation they belong to. One trace then has several ids, Maple picks one of them for the whole trace, not necessarily yours, and it can split the turn into one turn per id. Add a span processor that copies the root span's thread id to every span of the trace: - ```ts -// src/mastra/conversation-id.ts -import type { SpanOutputProcessor } from "@mastra/core/observability" - -export const conversationIdFromRoot: SpanOutputProcessor = { - name: "conversation-id-from-root", - process(span) { - let root = span - while (root?.parent) root = root.parent - const threadId = root?.metadata?.threadId - if (span && threadId) span.metadata = { ...span.metadata, threadId } - return span - }, - async shutdown() {}, -} -``` - -```ts -new Observability({ - configs: { - maple: { - serviceName: "support-agent", - exporters: [mapleExporter], - spanOutputProcessors: [conversationIdFromRoot], - }, - }, +export const orchestrator = new Agent({ + id: "orchestrator", + name: "orchestrator", + instructions: "Call weather_worker, budget_worker and transport_worker, then summary.", + model: "openrouter/openai/gpt-4o-mini", + agents: { weather_worker: weatherWorker, budget_worker: budgetWorker, transport_worker: transportWorker, summary }, + memory, }) ``` +The catch is memory. When the supervisor runs with a thread, Mastra gives each delegation its own thread id, `-`. The sub-agent's spans carry that id as `gen_ai.conversation.id`, so one trace has several ids. Maple takes the largest one for the whole trace, which is always a sub-agent's, so the run lands in a session of its own and its turn splits into one turn per id. The `mapleSpanProcessor` above rewrites every span to the root span's thread id, so the whole run stays in your conversation's session. + The processor also covers workflows whose steps call agents with their own `memory`. Pass `tracingContext` from the step's `execute` arguments to `agent.generate()`, so the agent's spans join the workflow's trace instead of starting a new one: ```ts @@ -252,7 +282,9 @@ Reasoning tokens are exported as `gen_ai.usage.reasoning_tokens`, a key Maple do Streamed calls report usage too. Time to first token is exported only as `mastra.completion_start_time`, a timestamp Maple doesn't read, so sessions show no time to first token. -Cost shows as unpriced. Mastra doesn't put a cost attribute on its spans, and Maple never prices tokens itself. Tokens, models and call counts are complete. +Cost shows as unpriced. Mastra doesn't put a cost attribute on its spans, and Maple never prices tokens itself. Tokens and models are complete, and call counts are too, apart from the extra call per tool approval described above. + +Mastra doesn't export `gen_ai.response.id`. Maple uses it only to merge two reports of the same call, and Mastra reports each call's usage once, so nothing is counted twice. ## Short-lived processes @@ -293,7 +325,7 @@ Run one conversation with at least two messages and a tool call, then open **Age - `invoke_agent ` spans for runs, `chat ` spans for model calls, and `execute_tool ` spans for tool calls, with `agent_step` and `model_generation` spans in between; - `invoke_workflow ` as the turn for workflow runs; - a lane per sub-agent, named after its `name`; -- token counts on every model call, including streamed ones; +- token counts on every model call, including streamed ones, except one 0-token call per tool approval; - failed tool calls marked as failed, with the thrown message; - cost shown as unpriced. @@ -302,14 +334,16 @@ To see what the exporter does locally, set `logLevel: "debug"` on `OtelExporter` ## Troubleshooting - **Nothing arrives and there is no error.** `observability` is a plain object instead of `new Observability(...)`, or the agent isn't registered on the `Mastra` instance. Check the startup log for a no-op observability warning. -- **`Traces http/json exporter is not installed` or `http/protobuf exporter is not installed` at startup.** The protocol's exporter package is missing. Set `protocol: "http/protobuf"` and install `@opentelemetry/exporter-trace-otlp-proto`. +- **`Traces http/json exporter is not installed` or `http/protobuf exporter is not installed` at startup.** The protocol's exporter package is missing. It ships as an optional dependency of `@mastra/otel-exporter`, so an install with `--omit=optional` or `--no-optional` skips it. Set `protocol: "http/protobuf"` and install `@opentelemetry/exporter-trace-otlp-proto`. - **`Custom configuration requires endpoint. Tracing will be disabled.`** The exporter got no `endpoint`. It doesn't read `OTEL_EXPORTER_OTLP_ENDPOINT`; pass the endpoint in code. - **`Export FAILED` with 401 or 403 in debug output.** The `Authorization` header is missing or the key is wrong. The header value is `Bearer ` followed by the ingest key. - **Every message is its own session.** The call has no `memory: { thread, resource }`, or the thread id changes per request. Pass the conversation's id on every call; for workflows, use `tracingOptions.metadata.threadId`. -- **A supervisor run shows several turns, or lands in another session.** Delegations got their own thread ids. Add the `conversationIdFromRoot` processor. +- **The transcript has the replies but none of the user's messages.** `mapleSpanProcessor` isn't in `spanOutputProcessors`, so the `chat` spans have no `gen_ai.input.messages`. +- **A supervisor run shows several turns, or lands in a session named `-`.** Delegations got their own thread ids. Add `mapleSpanProcessor`. - **Nothing arrives from a script or serverless function.** The process ended before the batch was exported. Call `mastra.shutdown()` in a script, or `mastra.observability.flush()` at the end of each request. -- **Spans arrive, but no model calls, models or tokens.** `@mastra/observability` and `@mastra/otel-exporter` are from different releases. Update `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` together. -- **Thousands of tiny `model_chunk` spans per streamed reply.** Add `excludeSpanTypes: [SpanType.MODEL_CHUNK]`. +- **One model call per turn with the turn's summed tokens, instead of one per request.** `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` are from different releases, and the exporter fell back to the older `model_generation` span as the model call. Update the three together. +- **One model call per tool approval has no tokens and no reply.** Expected: Mastra exports the call that requested the tool again when the run resumes, and that copy carries the tokens. +- **Dozens of tiny `model_chunk` spans per reply.** Mastra exports one per streamed chunk by default, for `generate()` too. Add `excludeSpanTypes: [SpanType.MODEL_CHUNK]`. - **Hundreds of `workflow_step` spans per turn.** `includeInternalSpans: true` is set. Mastra runs its agent loop as internal workflows; the flag exports all of them, around 5 times the spans and 25 times the bytes per turn. Leave it off. - **A workflow's agents show up as separate traces.** The step called `agent.generate()` without `tracingContext`. Pass it from the step's `execute` arguments. - **The latest user message is missing from a long conversation's transcript.** The message list hit `maxArrayLength` (50). Raise it in `serializationOptions`. diff --git a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md index 43eea0cf07..d8f8ef001c 100644 --- a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md +++ b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md @@ -1,6 +1,6 @@ --- title: "Trace Microsoft Agent Framework and Semantic Kernel agents with OpenTelemetry" -description: "Send Microsoft Agent Framework and Semantic Kernel traces to Maple so each conversation becomes one agent session with its transcript, tool calls, tokens and failures, in Python and .NET." +description: "Send Microsoft Agent Framework and Semantic Kernel traces to Maple so each conversation becomes one Agent Session with its transcript, tool calls, tokens and failures, in Python and .NET." group: "AI Agents" order: 30 navLabel: "Microsoft Agent Framework" @@ -9,7 +9,9 @@ icon: "dotnet" Microsoft Agent Framework (MAF) ships its own OpenTelemetry instrumentation in Python and .NET. Every `agent.run()` produces an `invoke_agent` span, a `chat` span per model call and an `execute_tool` span per tool call, using the current GenAI semantic conventions, with tokens (including cache and reasoning buckets) and, once you switch content capture on, the full prompts and replies as span attributes. Workflows add `workflow.run`, `executor.process` and `edge_group.process` spans. -What it doesn't emit is a conversation id. MAF only sets `gen_ai.conversation.id` when the model provider stores the conversation server side, so with Chat Completions, OpenRouter or any local chat history, every turn arrives as its own one-turn session. This guide adds the id with a 15-line span processor. It covers `agent-framework` 1.19 (Python), `Microsoft.Agents.AI` 1.22 (.NET), and Semantic Kernel 1.44 (Python), the framework MAF replaces, which has its own section below. +What it doesn't emit is a conversation id. MAF only sets `gen_ai.conversation.id` when the model provider stores the conversation server side, so with Chat Completions, OpenRouter or any local chat history, every turn arrives as its own one-turn session. This guide adds the id with a 15-line span processor. + +This guide covers `agent-framework` 1.19 (Python), `Microsoft.Agents.AI` 1.22 (.NET), and Semantic Kernel 1.44 (Python), the framework MAF replaces, which has its own section below. ## Quick setup with a coding agent @@ -364,7 +366,7 @@ In .NET, SK reads the same variables, or the `AppContext` switches `Microsoft.Se Run one conversation of two or three turns, one of which calls a tool. Sessions appear in **Agent Sessions** within about a minute. You should see: - **One session per conversation**, labeled **Microsoft Agent Framework** (or **Semantic Kernel**) for Python, whose id is your conversation id, not `trace:…`. A second conversation is a second session. -- **One turn per `agent.run()`**, each rooted at `invoke_agent support_agent` with `chat gpt-4o-mini` and `execute_tool get_weather` spans under it. +- **One turn per `agent.run()`**, each rooted at `invoke_agent support_agent` with `chat gpt-4o-mini` and `execute_tool get_weather` spans under it. An approval resume is its own turn in the same session. In .NET the root is `invoke_agent support_agent()`. - **A transcript** with the system instructions, your messages and the replies, labeled by the first line of each user message. - **Tool calls** with arguments and results, and failed tools counted under **Tool errors**. - **Tokens** on every model call, including streamed ones. Cost shows as unpriced. @@ -381,7 +383,9 @@ Run one conversation of two or three turns, one of which calls a tool. Sessions - **The agent forgets earlier turns after adding tracing.** You passed `conversation_id` as a chat option, which disables the in-memory history. Remove it and use the processor. - **Every message part is one character.** `Message("user", text)` iterates a bare string; pass a list: `Message("user", [text])`. - **OpenRouter rejects the second turn with a `previous_response_id` error.** `OpenAIChatClient` is the Responses API client. Use `OpenAIChatCompletionClient` for OpenRouter and other Chat Completions endpoints. +- **A WARN log "Ignored an approval response ... did not match the active approval occurrence identity", but the tool ran.** Seen on every approval resume with `to_function_approval_response()` in 1.19. The `execute_tool` span shows the approved call ran once; the log is noise. - **.NET: no spans at all.** `AddSource` doesn't match the default `Experimental.` prefix. Use `AddSource("*Microsoft.Agents.AI*")`. +- **.NET: tools named like `_Main_g_GetWeather_0_3`.** The tool is a local function in a top-level `Program.cs`. Pass `name:` to `AIFunctionFactory.Create`. - **Semantic Kernel: `AutoFunctionInvocationLoop` spans but no `chat` or `invoke_agent`.** The `SEMANTICKERNEL_EXPERIMENTAL_GENAI_*` variables were set after `semantic_kernel` was imported. - **Semantic Kernel: an orchestration shows as many small sessions or traces.** Wrap the run in your own span and `conversation()` and start the runtime inside it. diff --git a/apps/landing/src/content/docs/agent-tracing/openai-agents.md b/apps/landing/src/content/docs/agent-tracing/openai-agents.md index 0fa2fbe359..5321a32644 100644 --- a/apps/landing/src/content/docs/agent-tracing/openai-agents.md +++ b/apps/landing/src/content/docs/agent-tracing/openai-agents.md @@ -9,14 +9,16 @@ icon: "openai" The OpenAI Agents SDK traces every run out of the box, but not with OpenTelemetry. Its tracing pipeline builds its own traces and spans (agent, generation, function, handoff, guardrail) and uploads them to the OpenAI dashboard. To get them into Maple you swap that uploader for OpenInference's `openinference-instrumentation-openai-agents`, which turns each SDK span into an OpenTelemetry span as it ends. -The SDK's own way to tie the traces of one chat together, `group_id`, never reaches OpenTelemetry. The bridge drops it, so a ten-message conversation shows up in Maple as ten one-turn sessions until you wrap each run in OpenInference's `using_session`. A few more defaults need changing: the bridge writes OpenInference attributes that Maple's session page doesn't decode for this framework, and streamed calls to any provider other than OpenAI lose their tokens and their reply. This guide covers `openai-agents` 0.22 with `openinference-instrumentation-openai-agents` 2.5 on Python 3.10 to 3.14. TypeScript (`@openai/agents`) works with less detail; see [TypeScript](#typescript-openaiagents). +The SDK's own way to tie the traces of one chat together, `group_id`, never reaches OpenTelemetry. The bridge drops it, so a ten-message conversation shows up in Maple as ten one-turn sessions until you wrap each run in OpenInference's `using_session`. A few more defaults need changing: the bridge writes OpenInference attributes that Maple's session page doesn't decode for this framework, and streamed calls to any provider other than OpenAI lose their tokens and their reply. + +This guide covers `openai-agents` 0.22 with `openinference-instrumentation-openai-agents` 2.5 on Python 3.10 to 3.14. TypeScript (`@openai/agents`) works with less detail; see [TypeScript](#typescript-openaiagents). ## Quick setup with a coding agent Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-openai-agents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) skill, which contains every step of this guide. ```text -Set up Maple agent tracing for OpenAI Agents SDK in this project. +Set up Maple agent tracing for the OpenAI Agents SDK in this project. Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-openai-agents -y`, then follow it. @@ -231,7 +233,7 @@ Model spans are named `generation` on the Chat Completions path and `response` o The SDK also totals usage per run and per turn, but the bridge doesn't export those totals, so nothing is counted twice. -Maple shows cost only when a span carries one, and neither the SDK nor the bridge records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call. +Maple shows cost only when a span carries one, and neither the SDK nor the bridge records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call. Chat Completions model spans have no `gen_ai.response.id`, so nest the Broadcast spans under them as described in [Join Broadcast to your own traces](/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces), or each call is counted twice. Don't add `openinference-instrumentation-openai` next to the Agents bridge. It patches the `openai` client the SDK calls, so every model call gets a second model span under the `generation` span. The same goes for Logfire's `instrument_openai_agents()`, Langfuse's or Traceloop's Agents instrumentation, and the OpenTelemetry project's `opentelemetry-instrumentation-genai-openai-agents`: pick one. diff --git a/apps/landing/src/content/docs/agent-tracing/openrouter.md b/apps/landing/src/content/docs/agent-tracing/openrouter.md index 69cf5935c8..a0846e5de7 100644 --- a/apps/landing/src/content/docs/agent-tracing/openrouter.md +++ b/apps/landing/src/content/docs/agent-tracing/openrouter.md @@ -9,7 +9,9 @@ icon: "openrouter" OpenRouter Broadcast exports a trace for every request that goes through your OpenRouter account. You configure it once in the OpenRouter dashboard, with no SDK and no code in your app. Each trace carries the model, the provider that served it, input, output, cached and reasoning tokens, the cost OpenRouter charged you, and the prompt and completion. Maple reads all of it and shows these traces under **Agent Sessions** as vendor **OpenRouter**. -Out of the box, though, every model call is its own trace and its own session. Broadcast only knows what is in the request body. Unless your code sends a `session_id`, a ten-turn conversation shows up as thirty one-call "sessions". This guide covers the dashboard setup, the two request fields that fix the grouping (`session_id` and `trace`), and what Broadcast can't see: your tools and your agent structure. Code samples use the `openai` SDK (npm 7.23, PyPI 3.20), `@openrouter/ai-sdk-provider` 3.1 for the Vercel AI SDK, and `@openrouter/sdk` 1.3. +Out of the box, though, every model call is its own trace and its own session. Broadcast only knows what is in the request body. Unless your code sends a `session_id`, a ten-turn conversation shows up as thirty one-call "sessions". + +This guide covers the dashboard setup, the two request fields that fix the grouping (`session_id` and `trace`), and what Broadcast can't see: your tools and your agent structure. Code samples use the `openai` SDK (npm 7.23, PyPI 3.20), `@openrouter/ai-sdk-provider` 3.1 for the Vercel AI SDK, and `@openrouter/sdk` 1.3. ## Quick setup with a coding agent @@ -42,7 +44,7 @@ The agent can change your code, but not your OpenRouter dashboard. It finishes b 4. Set **Headers** to a JSON object with your Maple ingest key: ```json - { "Authorization": "Bearer maple_pk_your_key" } + { "Authorization": "Bearer YOUR_INGEST_KEY" } ``` 5. Leave the sampling rate at 1.0 and Privacy Mode off for now (see [privacy](#prompts-completions-and-privacy-mode) below). @@ -258,7 +260,7 @@ The service name on Broadcast spans is always `openrouter`, and there is no envi ## Troubleshooting -- **Test Connection fails.** The endpoint must be the full `https://ingest.maple.dev/v1/traces` URL and the headers valid JSON with `"Authorization": "Bearer "`. EU organizations use `ingest.eu.maple.dev`. +- **Test Connection fails.** The endpoint must be the full `https://ingest.maple.dev/v1/traces` URL and the headers valid JSON with `"Authorization": "Bearer YOUR_INGEST_KEY"`. EU organizations use `ingest.eu.maple.dev`. - **Test Connection passes but nothing arrives.** Check the destination's API key filter and data regions against the key and endpoint your app actually uses, and that **Enable Broadcast** is on for the account or organization your app's key belongs to. A placeholder key such as `MAPLE_TEST` passes the test, but Maple discards everything it sends. - **Every call is its own session, named `trace:`.** The request has no `session_id`. Check the outgoing body, not just your code: a wrapper or framework may drop unknown fields. - **A session named `trace:00000000000000000000000000000001` with one span.** That's the `openrouter-connection-test` span from **Test Connection**. Ignore it. diff --git a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md index fcbf5f0310..771f7f3bf0 100644 --- a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md +++ b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md @@ -11,7 +11,7 @@ If your agent is a loop you wrote yourself, runs on a framework without OpenTele What goes wrong is the detail. The GenAI conventions are still in Development status and have renamed attributes several times, and a span that looks right can still arrive with an empty transcript: messages sent as plain text, content put in span events, a JSON string cut off by an attribute length limit, or a fresh conversation id on every request. -This guide lists the exact keys and value formats Maple reads, with full examples in TypeScript and Python and a shorter one in Go. It follows the conventions in [`semantic-conventions-genai`](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. The code was written against the OpenTelemetry JS SDK 2.11, Python SDK 1.45 and Go SDK 1.46, with the OpenAI SDK 7.23 for JavaScript and 3.20 for Python. +This guide lists the exact keys and value formats Maple reads, with full examples in TypeScript and Python and a shorter one in Go. It follows the conventions in [`semantic-conventions-genai`](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. The TypeScript and Python examples were run against OpenRouter with the OpenTelemetry JS SDK 2.11 (OTLP exporter 0.222) on Node.js 26 and the Python SDK 1.45 on Python 3.14, using the OpenAI SDK 7.23 and 3.20. The Go example targets the Go SDK 1.46 and was not compiled for this guide. If you use a framework, check the [framework guides](/docs/agent-tracing) first. Most of them emit these spans for you. @@ -409,6 +409,8 @@ try { } ``` +Run it with `npx tsx main.ts` or your bundler. The files are ES modules (`"type": "module"` in `package.json`) for the top-level `await`, and use extensionless imports, which Node's built-in type stripping doesn't resolve. + ## Instrument the agent loop in Python The same loop with the OpenAI Python SDK: @@ -724,7 +726,7 @@ func Chat(ctx context.Context, model string, input []Message, call func(context. } ``` -An `execute_tool` span follows the same pattern with the attributes from the table. Rust (`opentelemetry` crate), Ruby (`opentelemetry-sdk`), Elixir (`opentelemetry_api`), Java and .NET follow the same pattern: set the attributes from the tables above as strings, ints, doubles and string arrays, and serialize every message and tool payload to a JSON string first. +This snippet was not compiled for this guide; run `go vet ./...` after adding it. An `execute_tool` span follows the same pattern with the attributes from the table. Rust (`opentelemetry` crate), Ruby (`opentelemetry-sdk`), Elixir (`opentelemetry_api`), Java and .NET follow the same pattern: set the attributes from the tables above as strings, ints, doubles and string arrays, and serialize every message and tool payload to a JSON string first. ## Group turns into one session @@ -740,7 +742,7 @@ A few things break grouping: ### When a framework's session key isn't read: `maple_ai.session.id` -Some frameworks write their session id under a key Maple doesn't read for that framework, for example OpenInference's `session.id` from LangChain or LlamaIndex instrumentation, or the Vercel AI SDK's `runtimeContext`. The fix is a wrapper span of your own around each turn, carrying Maple's own session key: +Some frameworks write their session id under a key Maple doesn't read for that framework, for example OpenInference's `session.id` from LangChain or LlamaIndex instrumentation, or the Vercel AI SDK's `runtimeContext`. The framework guides fix those three without a wrapper. For anything they don't cover, add a wrapper span of your own around each turn, carrying Maple's own session key: ```ts // chatId comes from your request; frameworkAgent is the framework's agent @@ -829,9 +831,9 @@ Providers disagree on whether input includes cached tokens, so Maple interprets | `gcp.gemini`, `gcp.vertex_ai` | `promptTokenCount`, including cache | `candidatesTokenCount`, excluding thoughts | | `openai`, `openrouter`, anything else | `prompt_tokens`, including cache | includes reasoning | -The Anthropic row differs from the spec, which asks for the inclusive total on every provider. If you send Anthropic's input as `input_tokens + cache_read + cache_write`, Maple counts cached tokens twice. A call to Claude through OpenRouter's OpenAI-compatible API uses `openrouter`, and OpenAI-shaped numbers. +The Anthropic row differs from the spec, which asks for the inclusive total on every provider. If you send Anthropic's input as `input_tokens + cache_read + cache_write`, Maple counts cached tokens twice. A call to Claude through OpenRouter's OpenAI-compatible API uses `openrouter`, and OpenAI-shaped numbers: in our test, a cached Claude Haiku 4.5 call reported 11,076 `prompt_tokens` with 11,058 of them in `cached_tokens`, so nothing is counted twice. OpenRouter also reports `prompt_tokens_details.cache_write_tokens`; copy it to `gen_ai.usage.cache_write.input_tokens` if you want cache writes shown separately. -Streaming needs `stream_options: { include_usage: true }` on OpenAI's API, or the stream carries no usage. The usage arrives in the last chunk, which has no choices. OpenRouter always sends usage. +Streaming needs `stream_options: { include_usage: true }` on OpenAI's API, or the stream carries no usage. OpenAI sends it in an extra last chunk with no choices. OpenRouter always sends usage and cost, on the chunk that carries the finish reason. Read `chunk.usage` before skipping chunks without choices, as the examples do, and both work. Maple never prices tokens. A session shows cost only when `chat` spans carry `gen_ai.usage.cost` in USD (`gen_ai.usage.total_cost` also works); otherwise it's shown as unpriced. OpenRouter returns the cost in `usage.cost`, which the examples copy. OpenAI and Anthropic don't return a cost, so compute it from your own price table or leave it out. diff --git a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md index aa25f42076..a90a2b0403 100644 --- a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md +++ b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md @@ -11,7 +11,7 @@ If your agent is your own loop around `client.chat.completions.create`, `client. Without them, every model call is its own trace, and Maple files every trace without a conversation id as its own session. A four-message chat with two tool calls shows up as six one-call sessions, with no tool calls and nothing tying them together. The second surprise is that the instrumentations record no prompts or replies until you turn content capture on. -This guide covers Python 3.10+ with `openai` 3.x, `anthropic` 1.x and `google-genai` 2.x, and TypeScript on Node.js with `openai` 7.x (the same pattern works for `@anthropic-ai/sdk` and `@google/genai`). If you use an agent framework on top of these SDKs, such as the OpenAI Agents SDK, LangChain or Pydantic AI, use [that framework's guide](/docs/agent-tracing) instead. +This guide covers Python 3.10+ with `openai` 3.x, `anthropic` 1.x and `google-genai` 2.x, and TypeScript on Node.js with `openai` 7.x (the same pattern works for `@anthropic-ai/sdk` and `@google/genai`). We ran the OpenAI and Anthropic paths end to end (Python `openai` 3.20.0 and `anthropic` 1.8.0 with the 1.2b0 instrumentations, TypeScript `openai` 7.23.0). The Gemini path follows the instrumentation's documentation and hasn't been run against a live Gemini model yet. If you use an agent framework on top of these SDKs, such as the OpenAI Agents SDK, LangChain or Pydantic AI, use [that framework's guide](/docs/agent-tracing) instead. ## Quick setup with a coding agent @@ -281,6 +281,9 @@ export function tracedChat(client: OpenAI, params: ChatParams, onText?: (delta: "gen_ai.usage.cache_read.input_tokens": completion.usage.prompt_tokens_details?.cached_tokens ?? 0, "gen_ai.usage.reasoning.output_tokens": completion.usage.completion_tokens_details?.reasoning_tokens ?? 0, }) + // OpenRouter adds the call's price in USD to usage. Other providers don't send one. + const cost = (completion.usage as { cost?: number }).cost + if (cost !== undefined) span.setAttribute("gen_ai.usage.cost", cost) } if (captureContent) { const output = completion.choices.map((c) => ({ ...toGenAiMessage(c.message), finish_reason: c.finish_reason })) @@ -429,9 +432,11 @@ stream = client.chat.completions.create( reply = "".join(chunk.choices[0].delta.content or "" for chunk in stream if chunk.choices) ``` -Without `stream_options`, the streamed call shows 0 tokens in Maple. Anthropic and Gemini always send usage on a stream. The Python instrumentations also record time to first chunk on streamed calls, which Maple shows per model call. +Without `stream_options`, the streamed call shows 0 tokens in Maple. Anthropic and Gemini always send usage on a stream. The Python OpenAI instrumentation and the TypeScript helper also record time to first chunk on streamed calls, which Maple shows per model call. The Anthropic instrumentation (1.2b0) doesn't record it for `messages.stream()`. -None of these instrumentations record cost, and Maple doesn't price tokens itself, so sessions show as **unpriced**. In the TypeScript helper you own the span: if your gateway returns the cost (OpenRouter puts it in `usage.cost`), set it as `gen_ai.usage.cost` in USD. +If you call Claude or Gemini models through OpenRouter's OpenAI-compatible endpoint with the `openai` SDK, the spans say `gen_ai.provider.name=openai`. That is correct: the usage arrives in OpenAI's shape, and Maple does the arithmetic for that shape. + +None of the Python instrumentations record cost, and Maple doesn't price tokens itself, so those sessions show as **unpriced**. In TypeScript you own the span: the helper copies OpenRouter's `usage.cost` (USD, sent on streamed calls too) to `gen_ai.usage.cost`, which Maple reads as the call's cost. Called directly, OpenAI sends no cost and the session stays unpriced. ## Short-lived processes @@ -449,7 +454,7 @@ Run one conversation of two or three messages with at least one tool call, flush - **Model calls** named `chat ` for OpenAI and Anthropic (`chat gpt-4o-mini`) or `generate_content ` for Gemini, with input and output tokens. - **Tool calls** named `execute_tool get_weather` with their arguments and results, and a failing tool counted as a tool error. - **A transcript** with the user messages, the replies and the tool calls between them. -- **Cost** shown as unpriced. +- **Cost** shown as unpriced, except for TypeScript calls through OpenRouter, where the helper records OpenRouter's price. ## Troubleshooting @@ -458,12 +463,13 @@ Run one conversation of two or three messages with at least one tool call, flush - **Turns show models and tokens but the transcript is empty.** Content capture is off, or set to `EVENT_ONLY` or `true`. Set `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` in the process that makes the calls. - **No model spans in Python, only yours.** The instrumentor wasn't installed or `instrument()` ran after the first request. Check that `tracing` is imported first, and that you installed `opentelemetry-instrumentation-genai-openai`, not `opentelemetry-instrumentation-openai`. - **Every model call appears twice.** Two instrumentations wrap the same SDK: the GenAI package plus OpenLLMetry, OpenInference, the deprecated `openai-v2` package, `logfire.instrument_openai()`, Sentry's OpenAI integration or a framework's own tracing. `opentelemetry-instrument` loads every instrumentation package that's installed, so uninstall the extras rather than just not calling them. -- **Twin traces for every call when you use OpenRouter.** OpenRouter Broadcast is also exporting the same calls. Keep one source; see the [OpenRouter guide](/docs/agent-tracing/openrouter). +- **Twin traces for every call when you use OpenRouter.** OpenRouter Broadcast is also exporting the same calls. Keep one source, or nest Broadcast under your spans as the [OpenRouter guide](/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces) shows. - **The streamed turn has 0 tokens.** Add `stream_options={"include_usage": True}` (OpenAI Chat Completions). - **A failing tool shows as successful.** The tool's exception was caught outside `run_tool`. Let `run_tool` catch it, or set the span status and `error.type` where you catch it. - **Sub-agent calls land in the orchestrator's lane.** The sub-agent's `agent_span` has the same name as the orchestrator's, or it was never wrapped. Give each agent its own name. - **Two conversation ids in one trace.** With the OpenAI Responses API and a `conversation` parameter, the instrumentation also sets `gen_ai.conversation.id` to that `conv_...` id. Pass the same id to `agent_span`, or Maple picks one of the two and can split the turn. -- **Cached Anthropic calls show more input tokens than you were billed for (Python).** The Anthropic GenAI instrumentation reports input tokens including cached ones, while Maple reads Anthropic's figure as excluding them, so cache reads count twice in the totals. Calls without prompt caching are unaffected. +- **Cached Anthropic calls show more input tokens than you were billed for (Python).** The Anthropic GenAI instrumentation reports `gen_ai.usage.input_tokens` as the raw input plus cache reads plus cache writes (a call that sent 330 new tokens and wrote 7,581 to the cache reports 7,911), while Maple reads Anthropic's figure as excluding the cache and adds both buckets again. Calls without prompt caching are unaffected. +- **Using the Anthropic SDK through OpenRouter.** Point it at `https://openrouter.ai/api` (no `/v1`): `Anthropic(base_url="https://openrouter.ai/api", api_key=OPENROUTER_API_KEY)`. OpenRouter accepts the key as `api_key` or `auth_token`. Model ids are OpenRouter's, such as `anthropic/claude-haiku-4.5`, and the spans say `gen_ai.provider.name=anthropic`. - **Nothing arrives, and the exporter logs 401.** The key or region is wrong. EU keys only work with `ingest.eu.maple.dev`. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md index 945fb5e01d..d2a8dd0546 100644 --- a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md @@ -244,7 +244,7 @@ The `invoke_agent` span reports the run's own total under `gen_ai.aggregated_usa Streaming needs nothing extra. Pydantic AI requests usage on OpenAI-compatible streams (`stream_options.include_usage`), so streamed calls have token counts too. The streamed `chat` span also records time to first chunk, under a key Maple doesn't read yet. -Cost shows as unpriced. Pydantic AI prices each call and writes the result to `operation.cost` on the `chat` span, but Maple reads cost only from `gen_ai.usage.cost` and never prices tokens itself. Tokens, models and call counts are complete. +Cost shows as unpriced. Pydantic AI prices each call and writes the result to `operation.cost` on the `chat` span, but Maple reads cost only from `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or `llm.cost.total`, and never prices tokens itself. Tokens, models and call counts are complete. ## Short-lived processes diff --git a/apps/landing/src/content/docs/agent-tracing/smolagents.md b/apps/landing/src/content/docs/agent-tracing/smolagents.md index c054c38461..58ce1a22b2 100644 --- a/apps/landing/src/content/docs/agent-tracing/smolagents.md +++ b/apps/landing/src/content/docs/agent-tracing/smolagents.md @@ -9,7 +9,9 @@ icon: "huggingface" smolagents has no OpenTelemetry code of its own. Every span comes from OpenInference's `openinference-instrumentation-smolagents`, which patches `MultiStepAgent.run`, each agent step, every model's `generate` and `Tool.__call__`. By default those spans use OpenInference attribute names (`llm.input_messages.0.message.role`, `llm.token_count.prompt`), and every `agent.run()` starts a new trace with no conversation id. -Two defaults have to change for Maple. Maple's session page reads the OpenTelemetry GenAI attributes (`gen_ai.*`) for smolagents, not OpenInference's own, so without the instrumentor's GenAI dual-write the session list shows token counts and the session page shows no transcript. And without `using_session(...)` around each run, a ten-message chat becomes ten one-turn sessions. This guide covers smolagents 1.26 with `openinference-instrumentation-smolagents` 0.1.40 on Python 3.10 or later, for both `ToolCallingAgent` and `CodeAgent`. +Two defaults have to change for Maple. Maple's session page reads the OpenTelemetry GenAI attributes (`gen_ai.*`) for smolagents, not OpenInference's own, so without the instrumentor's GenAI dual-write the session list shows token counts and the session page shows no transcript. And without `using_session(...)` around each run, a ten-message chat becomes ten one-turn sessions. + +This guide covers smolagents 1.26 with `openinference-instrumentation-smolagents` 0.1.40 on Python 3.10 or later, for both `ToolCallingAgent` and `CodeAgent`. ## Quick setup with a coding agent diff --git a/apps/landing/src/content/docs/agent-tracing/spring-ai.md b/apps/landing/src/content/docs/agent-tracing/spring-ai.md index bf99ca2512..1240f134a6 100644 --- a/apps/landing/src/content/docs/agent-tracing/spring-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/spring-ai.md @@ -27,7 +27,7 @@ Use your key from **Settings → Ingestion**. Without one, the agent uses a plac ## Install the OpenTelemetry starter and export to Maple -Spring AI 2.0 requires Spring Boot 4. Import the Spring AI BOM and add your model starter, Boot's OpenTelemetry starter and Actuator: +Spring AI 2.0 requires Spring Boot 4. Import the Spring AI BOM and add your model starter and Boot's OpenTelemetry starter: ```xml @@ -51,14 +51,10 @@ Spring AI 2.0 requires Spring Boot 4. Import the Spring AI BOM and add your mode org.springframework.boot spring-boot-starter-opentelemetry - - org.springframework.boot - spring-boot-starter-actuator - ``` -`spring-boot-starter-opentelemetry` brings the Micrometer-to-OpenTelemetry tracing bridge, the OpenTelemetry SDK and the OTLP exporter. With Gradle, use the same three artifacts and `platform("org.springframework.ai:spring-ai-bom:2.0.1")`. +`spring-boot-starter-opentelemetry` brings the Micrometer-to-OpenTelemetry tracing bridge, the OpenTelemetry SDK and the OTLP exporter. On Boot 4 you don't need Actuator for tracing. With Gradle, use the same two artifacts and `platform("org.springframework.ai:spring-ai-bom:2.0.1")`. Then point the exporter at Maple in `application.properties`: @@ -76,8 +72,6 @@ management.otlp.metrics.export.headers.Authorization=Bearer ${MAPLE_INGEST_KEY} # Read by the configuration class below maple.ai.capture-content=true -# Token usage on streamed OpenAI calls -spring.ai.openai.chat.stream-options.include-usage=true ``` EU organizations use `https://ingest.eu.maple.dev`. Unlike the `OTEL_EXPORTER_OTLP_ENDPOINT` variable, this property takes the full URL, so keep `/v1/traces` on the end. The transport defaults to OTLP over HTTP with protobuf, which is what Maple ingest expects. `spring.application.name` becomes `service.name`. @@ -100,7 +94,6 @@ import java.util.Map; import io.micrometer.common.KeyValue; import io.micrometer.observation.Observation; import io.micrometer.observation.ObservationFilter; -import io.micrometer.observation.ObservationPredicate; import io.micrometer.observation.ObservationRegistry; import tools.jackson.databind.json.JsonMapper; @@ -124,13 +117,6 @@ public class MapleAiObservationConfig { /** Agent name for a ChatClient: `.defaultAdvisors(a -> a.param(AGENT_NAME, "support_agent"))`. */ public static final String AGENT_NAME = "gen_ai.agent.name"; - // Advisor spans carry nothing Maple reads, and their names ("tool _calling ", - // "message_chat_memory") would be counted as extra tool and LLM calls. - @Bean - ObservationPredicate skipAdvisorObservations() { - return (name, context) -> !(context instanceof AdvisorObservationContext); - } - @Bean ObservationFilter mapleGenAiAttributes(@Value("${maple.ai.capture-content:false}") boolean captureContent) { return context -> { @@ -141,6 +127,11 @@ public class MapleAiObservationConfig { client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.agent.name", agent)); } } + else if (context instanceof AdvisorObservationContext advisor) { + // Advisor span names ("tool _calling ", "message_chat_memory") would be + // counted as extra tool and LLM calls. A neutral name keeps them as plumbing. + advisor.setContextualName("spring_ai advisor"); + } else if (context instanceof ToolCallingObservationContext tool) { tool.addLowCardinalityKeyValue(KeyValue.of("gen_ai.tool.name", tool.getToolDefinition().name())); if (tool.getToolCallId() != null) { @@ -203,11 +194,12 @@ public class MapleAiObservationConfig { What each bean does: -- **`ObservationPredicate`** drops the advisor observations. Their spans hold only an advisor name and order, and Maple classifies spans without a known operation by name: `tool _calling ` would count as a tool call and `message_chat_memory` as a model call on every turn. Child spans still attach to the `chat_client` span, because Micrometer keeps the scope of a skipped observation. -- **`ObservationFilter`** runs when each observation stops, after Spring AI's own conventions. It relabels the `chat_client` span as `invoke_agent` (Spring AI calls it `framework`, which Maple would otherwise count as a model call because the span name contains "chat"), copies the tool name and call id to the `gen_ai.tool.*` keys, and writes the conversation as `gen_ai.input.messages` and `gen_ai.output.messages`. +- **`ObservationFilter`** runs when each observation stops, after Spring AI's own conventions. It relabels the `chat_client` span as `invoke_agent` (Spring AI calls it `framework`, which Maple would otherwise count as a model call because the span name contains "chat"), copies the tool name and call id to the `gen_ai.tool.*` keys, and writes the conversation as `gen_ai.input.messages` and `gen_ai.output.messages`. It also renames the advisor spans to `spring_ai advisor`. Maple classifies spans without a known operation by name, so `tool _calling ` would count as a tool call and `message_chat_memory` as a model call on every turn. - **`ToolExecutionExceptionProcessor`** marks the running tool span as failed before Spring AI turns the exception into a message for the model. See [Tools, errors and sub-agents](#tools-errors-and-sub-agents). -Boot applies `ObservationPredicate` and `ObservationFilter` beans to the observation registry by itself; there is nothing else to register. The class works the same in Kotlin; for example, the predicate is `ObservationPredicate { _, context -> context !is AdvisorObservationContext }`. +Boot applies `ObservationFilter` beans to the observation registry by itself; there is nothing else to register. The class works the same in Kotlin. + +Renaming the advisor spans is deliberate. Dropping them with an `ObservationPredicate` looks cleaner and works for `.call()`, but on `.stream()` Spring AI reads the model span's parent from the Reactor context, where it finds the skipped advisor. The streamed `chat` span then starts a trace of its own, outside the session. ### Spring AI 1.1 on Spring Boot 3.5 @@ -215,7 +207,7 @@ The same approach works, with these differences: | | Spring AI 2.0, Boot 4 | Spring AI 1.1, Boot 3.5 | |---|---|---| -| Tracing dependencies | `spring-boot-starter-opentelemetry` | `io.micrometer:micrometer-tracing-bridge-otel` and `io.opentelemetry:opentelemetry-exporter-otlp` | +| Tracing dependencies | `spring-boot-starter-opentelemetry` | `io.micrometer:micrometer-tracing-bridge-otel`, `io.opentelemetry:opentelemetry-exporter-otlp` and `spring-boot-starter-actuator` | | Endpoint property | `management.opentelemetry.tracing.export.otlp.endpoint` | `management.otlp.tracing.endpoint` | | Header property | `management.opentelemetry.tracing.export.otlp.headers.*` | `management.otlp.tracing.headers.*` | | JSON in the filter | Jackson 3 `JsonMapper.shared()` | Jackson 2 `ObjectMapper` (checked exception) | @@ -346,7 +338,7 @@ Spring AI runs the tool calls of one model response one after another, on the ca Every `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports them. Maple reads all four. The `chat_client` spans carry no usage, so nothing is counted twice. -Streaming has one catch. OpenAI (and OpenAI-compatible gateways such as OpenRouter) only return usage on a stream when the request asks for it. Without `spring.ai.openai.chat.stream-options.include-usage=true`, or `OpenAiChatOptions.builder().streamUsage(true)` per call, streamed turns show zero tokens. +Streaming has one catch. OpenAI (and OpenAI-compatible gateways such as OpenRouter) only return usage on a stream when the request asks for it. Spring AI 2.0's OpenAI model asks by default, so streamed turns report tokens out of the box. That default disappears as soon as you set any `spring.ai.openai.chat.stream-options.*` property: then add `spring.ai.openai.chat.stream-options.include-usage=true` too, or streamed turns show zero tokens. On Spring AI 1.1, set `spring.ai.openai.chat.options.stream-usage=true`. Spring AI records the provider as `gen_ai.system`, derived from the client class rather than the model. Behind OpenRouter, a Claude model called through the OpenAI starter is labeled `openai`. Maple uses the provider only to decide whether cached tokens are included in the input count, so this matters only for cache figures. @@ -356,7 +348,15 @@ Spring AI emits no cost attribute, and Maple never prices tokens itself, so sess Spring Boot owns the `SdkTracerProvider` and shuts it down when the application context closes, which flushes the batch of pending spans. A web app needs nothing extra. For other shapes: -- **`CommandLineRunner` apps and batch jobs.** Let `run()` return, or exit through `SpringApplication.exit(context)`. `System.exit()` still runs Boot's shutdown hook; `kill -9` and `Runtime.halt()` lose the last batch. +- **`CommandLineRunner` apps and batch jobs.** Exit explicitly. The OpenAI starter's HTTP client keeps non-daemon threads alive for about 60 seconds after the last call, so a `main` that simply returns leaves the JVM, and the unflushed spans, waiting that long. `SpringApplication.exit` closes the context, which flushes: + +```java +public static void main(String[] args) { + System.exit(SpringApplication.exit(SpringApplication.run(Application.class, args))); +} +``` + +`kill -9` and `Runtime.halt()` lose the last batch. - **Serverless (Spring Cloud Function on AWS Lambda and similar).** The runtime freezes the process between invocations, so flush before returning from each one. Don't shut the provider down. ```java @@ -372,23 +372,35 @@ The exporter sends a batch every 5 seconds by default, so allow a few seconds af ## Running the OpenTelemetry Java agent too -Use one tracing setup per service. The OpenTelemetry Java agent does not turn Micrometer Observations into spans, so with the agent alone you get the HTTP spans but no `chat_client`, `chat` or `execute_tool` span. You still need the starter from this guide. +Use one tracing setup per service. The OpenTelemetry Java agent does not turn Micrometer Observations into spans, so with the agent alone you get HTTP spans but no `chat_client`, `chat` or `execute_tool` span. You still need the starter and the configuration class from this guide. + +Adding the starter next to the agent is not enough, though. Boot then runs its own OpenTelemetry SDK, and the agent doesn't share context with it. In our test with agent 2.31.1, every Spring AI span arrived as its own trace: the model and tool calls lost their `chat_client` parent, so they fell out of their sessions. + +If the agent has to stay (it also instruments JDBC, Kafka and other libraries), give Micrometer the agent's `OpenTelemetry` instead of Boot's: -If the agent has to stay (it also instruments JDBC, Kafka and other libraries), three things change: +```java +import io.opentelemetry.api.GlobalOpenTelemetry; +import io.opentelemetry.api.OpenTelemetry; + +@Bean +OpenTelemetry openTelemetry() { + return GlobalOpenTelemetry.get(); +} +``` -- **Model calls appear twice.** The agent instruments the OpenAI Java SDK, which Spring AI 2.0's OpenAI starter uses underneath, and adds its own `chat ` span to every call. Turn it off with `-Dotel.instrumentation.openai-java.enabled=false`. -- **HTTP requests appear twice**, once from the agent and once from Boot's `http.server.requests` observation. Set `management.observations.enable.http.server.requests=false`. -- **Two exporters run.** Point the agent (`OTEL_EXPORTER_OTLP_*`) and Boot (the properties above) at Maple, or traces arrive with gaps. +Add this bean only where the agent is attached; without the agent, `GlobalOpenTelemetry.get()` returns a no-op and nothing is traced. With the bean, the agent exports every span, so configure the agent rather than Boot: -We tested the starter setup end to end; the agent combination above is not covered by our tests. +- Set `OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev`, `OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer YOUR_INGEST_KEY`, `OTEL_SERVICE_NAME` and `OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production`. Boot's `management.opentelemetry.*` export properties and the sampling property no longer apply; the agent samples everything by default. +- Add `-Dotel.instrumentation.openai-java.enabled=false`. The agent ships instrumentation for the OpenAI Java SDK that Spring AI's OpenAI starter uses. It added no spans in our test with Spring AI 2.0.1, but it would duplicate every model call if it matched. +- HTTP client calls show up twice, once from Boot and once from the agent. They aren't AI spans, so sessions, counts and tokens are unaffected. ## Check that it works Run one conversation of two or three messages with the same conversation id, including one that calls a tool, plus a message with a second id. Wait about a minute, then open **Agent Sessions** in Maple. You should see: - **One session per conversation id**, with the framework shown as Spring AI and one turn per `ChatClient` call. A list of one-turn sessions named after trace ids means the `ChatMemory.CONVERSATION_ID` parameter is missing. -- **Spans named** `spring_ai chat_client` (as an agent span named after your `gen_ai.agent.name`), `chat openai/gpt-4o-mini` (your model id) and `execute_tool get_weather`. HTTP `POST` spans to the model provider appear muted next to them. -- **No spans named** `tool _calling `, `call` or `message_chat_memory`. If they show up, the `ObservationPredicate` bean isn't loaded. +- **Spans named** `spring_ai chat_client` (as an agent span named after your `gen_ai.agent.name`), `chat openai/gpt-4o-mini` (your model id) and `execute_tool get_weather`, with `spring_ai advisor` spans in between. HTTP `POST` spans to the model provider appear muted next to them. +- **No spans named** `tool _calling `, `call`, `stream` or `message_chat_memory`. If they show up, the `ObservationFilter` bean isn't loaded. - **The transcript**: each user message, the assistant's replies, and the tool calls with their arguments and results. - **Tokens** on every model call, including streamed ones. - **Sub-agents** as lanes named after each client's agent name, and a tool that threw counted as a failed tool call under its own name. @@ -400,9 +412,12 @@ Run one conversation of two or three messages with the same conversation id, inc - **Every message is its own session.** The `chat_client` span has no `spring.ai.chat.client.conversation.id`. Pass `.advisors(a -> a.param(ChatMemory.CONVERSATION_ID, id))` on every `prompt()` call. - **All users land in one session.** The conversation id is a constant, usually set once with `defaultAdvisors` on the builder. Pass it per request. - **Transcript is empty, but tokens and tools show up.** `maple.ai.capture-content` isn't `true`, or `MapleAiObservationConfig` isn't in a package Spring scans. `log-prompt` and `log-completion` don't help; they write to the log. -- **Twice as many model calls as expected, and a tool called `tool _calling ` in the tool list.** The advisor and `chat_client` spans are being counted. Make sure the `ObservationPredicate` and `ObservationFilter` beans are loaded. +- **Twice as many model calls as expected, and a tool called `tool _calling ` in the tool list.** The advisor and `chat_client` spans are being counted. Make sure the `ObservationFilter` bean is loaded. +- **The streamed turn's model call is missing from its session.** Something drops Spring AI's advisor observations, usually an `ObservationPredicate`. Rename advisor spans as the filter does instead of skipping them. +- **Every span is its own trace.** The OpenTelemetry Java agent is attached next to the starter. See [Running the OpenTelemetry Java agent too](#running-the-opentelemetry-java-agent-too). +- **A command-line app hangs for a minute after its last reply.** The OpenAI client's threads keep the JVM alive. Exit with `System.exit(SpringApplication.exit(...))`. - **A tool that threw shows as successful.** The custom `ToolExecutionExceptionProcessor` isn't active, or the app defines its own. Add the `registry.getCurrentObservation().error(exception)` call to the one that runs. -- **Streamed turns show zero tokens.** Add `spring.ai.openai.chat.stream-options.include-usage=true`. +- **Streamed turns show zero tokens.** You set a `spring.ai.openai.chat.stream-options.*` property without `include-usage=true`. Add it. - **A sub-agent shows up as its own session.** It ran on a thread without the caller's trace context. Propagate the observation to the executor, or run the sub-agent on the calling thread. - **Tool names are missing on the tool pages.** The `ObservationFilter` isn't running; Spring AI alone emits only `spring.ai.tool.definition.name`. - **Framework shows as Unidentified on some model spans.** With starters other than OpenAI, a `chat` span for a call without tools has no Spring AI marker, so Maple files it as a generic GenAI span. The session, transcript and tokens are unaffected. diff --git a/apps/landing/src/content/docs/agent-tracing/strands.md b/apps/landing/src/content/docs/agent-tracing/strands.md index 6aa4dba250..45690ed40d 100644 --- a/apps/landing/src/content/docs/agent-tracing/strands.md +++ b/apps/landing/src/content/docs/agent-tracing/strands.md @@ -9,7 +9,9 @@ icon: "strands" Strands Agents ships its own OpenTelemetry tracer. Every `agent(...)` call becomes one trace with an `invoke_agent` span, an `execute_event_loop_cycle` span per reasoning step, a `chat` span per model call and an `execute_tool` span per tool call. Maple recognizes these spans as Strands without any extra instrumentation library. -The catch is where Strands puts the conversation. By default, prompts, replies and tool results are written as span events, and Maple reads span attributes only, so the transcript comes out empty even though tokens and tool calls show up. One environment variable fixes it. This guide covers the Python SDK (`strands-agents` 1.54 or newer, tested on 1.57.1) and notes where the TypeScript SDK (`@strands-agents/sdk` 1.19) differs. +The catch is where Strands puts the conversation. By default, prompts, replies and tool results are written as span events, and Maple reads span attributes only, so the transcript comes out empty even though tokens and tool calls show up. One environment variable fixes it. + +This guide covers the Python SDK (`strands-agents` 1.54 or newer, tested on 1.57.1) and notes where the TypeScript SDK (`@strands-agents/sdk` 1.19) differs. ## Quick setup with a coding agent diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md index 3baa006824..a73204dda1 100644 --- a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -18,7 +18,7 @@ It covers AI SDK 7 (`ai` 7.0.106 or newer) on Node.js 22 or newer, in a plain No Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-vercel-ai-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-vercel-ai-sdk) skill, which contains every step of this guide. ```text -Set up Maple agent tracing for Vercel AI SDK in this project. +Set up Maple agent tracing for the Vercel AI SDK in this project. Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing-vercel-ai-sdk -y`, then follow it. diff --git a/apps/landing/src/content/docs/getting-started/ai-agents.md b/apps/landing/src/content/docs/getting-started/ai-agents.md index 142bbfb52d..c23f060820 100644 --- a/apps/landing/src/content/docs/getting-started/ai-agents.md +++ b/apps/landing/src/content/docs/getting-started/ai-agents.md @@ -113,10 +113,11 @@ When no tool fits, `describe_warehouse_tables` lists the tables and columns, and ## Instrument with a coding agent -Two open-source skills teach a coding agent how to set up OpenTelemetry for Maple: +Open-source skills teach a coding agent how to set up OpenTelemetry for Maple: - [maple-onboard](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-onboard) instruments every app and service in a repository: traces, logs and metrics, using the native OpenTelemetry SDK for each language. - [maple-audit](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-audit) reviews an existing setup, reports gaps per service (missing service map edges, missing `service.version`, errors without exceptions) and fixes them. +- [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) traces an AI agent so each conversation shows up as one [Agent Session](/docs/agent-sessions/overview). It detects the framework and installs a skill for that framework only. See [Trace your AI agent](/docs/agent-tracing). Install them together with the per-language guides they read: diff --git a/apps/landing/src/content/docs/instrumentation.mdx b/apps/landing/src/content/docs/instrumentation.mdx index 1524a4688b..7a89120399 100644 --- a/apps/landing/src/content/docs/instrumentation.mdx +++ b/apps/landing/src/content/docs/instrumentation.mdx @@ -53,6 +53,10 @@ Maple ingests OpenTelemetry over OTLP/HTTP, so any OTLP exporter works without a +## AI agents and LLM calls + +Agent frameworks and LLM gateways have their own guides, so each conversation shows up as one [Agent Session](/docs/agent-sessions/overview) with its transcript, tool calls, tokens and cost: Vercel AI SDK, OpenAI Agents SDK, LangChain and LangGraph, Mastra, Pydantic AI, CrewAI, Google ADK, OpenRouter, LiteLLM and more. See [Trace your AI agent](/docs/agent-tracing). +
## Infrastructure and data sources diff --git a/skills/maple-agent-tracing-crewai/SKILL.md b/skills/maple-agent-tracing-crewai/SKILL.md index d5f2c4ab2d..9bc2e4d83b 100644 --- a/skills/maple-agent-tracing-crewai/SKILL.md +++ b/skills/maple-agent-tracing-crewai/SKILL.md @@ -136,12 +136,12 @@ def handle_message(conversation_id: str, text: str, history: str) -> str: ``` - Wrap EVERY kickoff call site in `with using_session():`. Use the app's stored conversation/chat/thread id. Never a fresh UUID per request, never a constant. -- Conversational flows: `with using_session(sid): flow.handle_turn(text, session_id=sid)`. Use the same id for both. +- Conversational flows: `with using_session(sid): flow.handle_turn(text, session_id=sid)`. Use the same id for both. Give the flow class a `name = ""` attribute: unnamed flows produce `Flow_.kickoff` roots (then `Flow_` from the second turn). - Batch/one-shot crews (no chat): one id per run/job is correct; still wrap the kickoff so the run is a named session instead of `trace:`. - Give every `Crew` a `name=` (otherwise the root span is `Crew_.kickoff`) and every `Task` a `name=`. Put the user's message FIRST in the task description: Maple labels turns with the first line of `Current Task: …`. - Do not change the app's own history handling beyond that; CrewAI has no chat memory, so the app already passes history somehow. - `akickoff()` is NOT instrumented (no crew/agent spans, each model/tool call becomes its own trace). Replace `await crew.akickoff(...)` with `await crew.kickoff_async(...)` (same result, runs instrumented `kickoff` in a thread). Same for `Agent.akickoff` -> `Agent.kickoff` in a thread. Tell the user why. -- `Crew(stream=True)` runs the crew twice (a stub `kickoff` span + the real one in a second trace). Wrap each streamed turn: +- `Crew(stream=True)` calls `kickoff` twice (an empty stub `kickoff` span + the real one in a second trace). Wrap each streamed turn: ```py from opentelemetry import trace @@ -162,6 +162,7 @@ def stream_message(conversation_id: str, text: str, history: str, send) -> None: ``` Iterate INSIDE the `with`. `LLM(stream=True)` alone (no `Crew(stream=True)`) needs no wrapper. +- Tool approvals via `@before_tool_call` + `context.request_human_input(...)` need nothing: the hook runs inside the kickoff before the tool span starts (approved call = one tool span; blocked call = no tool span, model gets `Tool execution blocked by hook`). - `flow.resume(...)` after `@human_feedback` is not instrumented: wrap it the same way (`using_session` with the same id + the wrapper span). - Your own spans (plain OTel tracer) don't get `session.id` automatically; give them `attributes=dict(get_attributes_from_context())` (from `openinference.instrumentation`) if you add any beyond the wrapper above. @@ -169,6 +170,7 @@ def stream_message(conversation_id: str, text: str, history: str, send) -> None: - On by default: model spans carry the messages CrewAI sent (system = role/goal/backstory, user = `Current Task: …` + context) and the reply; agent spans the task and output; tool spans arguments and results. Leave it on unless the user or repo says prompts are sensitive. - To turn off: `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` passed to every instrumentor (or `OPENINFERENCE_HIDE_INPUTS=true` / `OPENINFERENCE_HIDE_OUTPUTS=true`). Narrower: `hide_input_text`, `hide_output_text`. +- The hide switches do NOT cover `crew_tasks` (task descriptions, i.e. the user's message), `crew_inputs`, `crew_agents` on the `.kickoff` span, or `flow_inputs` on a flow's kickoff span. If prompts must never leave the infrastructure, tell the user to delete those attributes in an OpenTelemetry Collector (`attributes` processor, `action: delete`). - Agent roles, task names and tool names are span names and always recorded. Don't put PII in them. - Do NOT set `share_crew=True` to get content; it only adds data to CrewAI's analytics. @@ -192,7 +194,7 @@ def stream_message(conversation_id: str, text: str, history: str, send) -> None: Run one real conversation (2-3 messages, same conversation id, at least one tool call), and one message in a second conversation. If the user gave no key (`MAPLE_TEST`), you can't see results in Maple; say so and list what they should check. Otherwise check in Maple **Agent Sessions** (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), filtered to the service name: - Exactly one session per conversation id (two here), not one per message and no `trace:` sessions. -- One turn per kickoff; each turn's root span is `.kickoff` (or `.kickoff`, or your `invoke_agent` wrapper when streaming). No empty extra turns. +- One turn per kickoff; each turn's root span is `.kickoff` (or `.kickoff`, or your `invoke_agent` wrapper when streaming). No empty extra turns. - Transcript is non-empty (system message from role/goal/backstory, `Current Task: …`, replies). - Model calls (`ChatCompletion` for the OpenAI instrumentor) have a model and non-zero input/output tokens, including streamed calls. Each call appears once. - Tool calls `.run` with results; a tool that raised is counted as failed and nothing else is. diff --git a/skills/maple-agent-tracing-litellm/SKILL.md b/skills/maple-agent-tracing-litellm/SKILL.md index ef7d07673a..3bb5c0bfac 100644 --- a/skills/maple-agent-tracing-litellm/SKILL.md +++ b/skills/maple-agent-tracing-litellm/SKILL.md @@ -17,10 +17,10 @@ Mechanism: ## Step 0: detect 1. Versions: `python -c "import importlib.metadata as m; print(m.version('litellm'), m.version('opentelemetry-api'))"` (or read `pyproject.toml` / lock files). - - LiteLLM 1.103.x (tested 1.103.0): OpenTelemetry must be **<= 1.43.0**. On >= 1.44 the v2 logger fails to import (`No module named 'opentelemetry._events'`), LiteLLM logs one line and exports nothing. Fixed in LiteLLM >= 1.104 (then any OTel version). + - LiteLLM 1.103.x (tested 1.103.0, latest stable on 2026-09-28): OpenTelemetry must be **<= 1.43.0**. On >= 1.44, `from litellm.integrations.otel.logger import OpenTelemetryV2` raises `ModuleNotFoundError: No module named 'opentelemetry._events'`; the `callbacks: ["otel"]` form (proxy) only logs `Error initializing custom logger` and exports nothing. Fixed in LiteLLM 1.104 (1.104.0rc1 verified with OTel 1.45); once 1.104 stable exists, use it and drop the pin. - `python -c "from litellm.integrations.otel.logger import OpenTelemetryV2"` must succeed; if the module is missing, upgrade LiteLLM to >= 1.103. 2. Which path: - - App calls `litellm.acompletion(` / `litellm.completion(` / `Router(` → **SDK path** (Steps 2a-6). + - App calls `litellm.acompletion(` / `litellm.completion(` / `Router(` → **SDK path** (Steps 2a-6). `Router.acompletion` is traced like `acompletion` (span name and request model = the router alias, response model = the real model). - App calls a LiteLLM Proxy (OpenAI client with `base_url` pointing at the proxy, a `config.yaml` with `model_list`, docker `ghcr.io/berriai/litellm`) → **proxy path** (Step 2b). Only if the user controls the proxy; otherwise tell them and trace in-app with an OpenAI client instrumentor (out of scope here). 3. Sync vs async: grep for `litellm.completion(`, `litellm.text_completion(`, `router.completion(`. The v2 logger traces **async calls only** (`acompletion`, `Router.acompletion`); sync calls produce no span. Convert the agent loop to async where feasible. If the project is sync-only and can't change, use the v1 fallback in Step 2c. 4. Existing OTel: search `TracerProvider(`, `set_tracer_provider`, `opentelemetry-instrument`, `logfire.configure`, `sentry_sdk.init`, `litellm.callbacks`, `success_callback`, `"otel"`, `LITELLM_OTEL_V2`, `OpenAIInstrumentor`, `Traceloop.init`. @@ -104,7 +104,7 @@ OTEL_ENVIRONMENT_NAME=production OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only ``` -- Docker image `ghcr.io/berriai/litellm` ships OTel 1.28 + FastAPI instrumentation: fine. pip-installed proxy on 1.103: add `opentelemetry-sdk==1.43.0 opentelemetry-exporter-otlp-proto-http==1.43.0 opentelemetry-instrumentation-fastapi==0.64b0`. +- Docker image `ghcr.io/berriai/litellm` ships OTel 1.28 + FastAPI instrumentation: fine. pip-installed proxy on 1.103: `pip install "litellm[proxy]==1.103.0" "opentelemetry-sdk==1.43.0" "opentelemetry-exporter-otlp-proto-http==1.43.0" "opentelemetry-instrumentation-fastapi==0.64b0"`, run `litellm --config config.yaml`. The FastAPI instrumentation is what continues the app's `traceparent`. - App side: keep Step 3-5 spans (`agent_span`, `run_tool`), do NOT register the LiteLLM logger in the app, and on every request to the proxy send `traceparent` (so proxy spans join the app trace under `invoke_agent`) and `x-litellm-session-id` (becomes `gen_ai.conversation.id`): ```py @@ -124,7 +124,8 @@ async def call_model(conversation_id: str, messages: list, tools: list | None): ) ``` -- Alternative session carrier: body `metadata: {"session_id": ...}` (`extra_body={"metadata": {...}}`). +- Alternative session carrier: body `metadata: {"session_id": ...}` (`extra_body={"metadata": {...}}`), verified. +- Resulting tree per request: `invoke_agent` → `POST /chat/completions` (proxy FastAPI server span) → `chat ` + `auth /chat/completions`. Known Maple limitation: the `auth /chat/completions` span is counted as an extra LLM call (litellm scope + "chat" in the name), so LLM call counts double on the proxy path; tokens, cost, transcript and sessions are correct. Tell the user; nothing to fix app-side. - Never also instrument the app's OpenAI client when the proxy traces: double LLM calls and tokens. Pick gateway OR in-app. - Do not set `OTEL_IGNORE_CONTEXT_PROPAGATION=true` on the proxy. @@ -134,7 +135,7 @@ async def call_model(conversation_id: str, messages: list, tools: list | None): ## Step 3: agent loop spans + session id (required) -Wrap each agent run in `invoke_agent` and pass `litellm_session_id=` on EVERY `acompletion`: +Wrap each agent run in `invoke_agent` and pass `litellm_session_id=` on EVERY `acompletion`: (imports: `asyncio`, `inspect`, `json`, `contextlib.contextmanager`, `opentelemetry.trace.StatusCode`, `tracer` from `tracing.py`): ```py @contextmanager @@ -158,8 +159,9 @@ async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: messages.append(message.model_dump(exclude_none=True)) if not message.tool_calls: return message.content or "" - for call in message.tool_calls: - messages.append({"role": "tool", "tool_call_id": call.id, "content": run_tool(agent, call)}) + results = await asyncio.gather(*(run_tool(agent, call) for call in message.tool_calls)) + for call, output in zip(message.tool_calls, results): + messages.append({"role": "tool", "tool_call_id": call.id, "content": output}) ``` - Adapt to the project's existing loop; don't rewrite it into this shape if it already has one. What matters: one `invoke_agent` span per agent run, all model calls and tool calls inside it, `litellm_session_id=` on every call (`metadata={"session_id": ...}` also works when the project already passes metadata). @@ -171,12 +173,13 @@ async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: - v2 default is `no_content`. `capture_message_content="span_only"` (or env `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only`) puts `gen_ai.input.messages` / `gen_ai.output.messages` JSON on `chat` spans. Maple's transcript needs them. - Never `event_only` / `span_and_event` for Maple: events are not read. +- Messages are OpenAI chat format (`{role, content, tool_calls}`). Maple's transcript ignores `tool_calls` inside messages: a call that only requested tools shows an empty reply, and tool calls render only from `execute_tool` spans (Step 5). So the `execute_tool` spans are required for tool calls to appear at all. - User wants content off → `no_content` + drop the tool args/result attributes in `run_tool`. `litellm.turn_off_message_logging = True` keeps structure but replaces text with `redacted-by-litellm`. ## Step 5: tools, errors, sub-agents ```py -def run_tool(agent: Agent, call) -> str: +async def run_tool(agent: Agent, call) -> str: name = call.function.name with tracer.start_as_current_span(f"execute_tool {name}") as span: span.set_attribute("gen_ai.operation.name", "execute_tool") @@ -185,6 +188,8 @@ def run_tool(agent: Agent, call) -> str: span.set_attribute("gen_ai.tool.call.arguments", call.function.arguments) try: result = agent.tools[name](**json.loads(call.function.arguments or "{}")) + if inspect.isawaitable(result): # async tools and sub-agents + result = await result except Exception as exc: span.set_status(StatusCode.ERROR, str(exc)) span.set_attribute("error.type", type(exc).__name__) @@ -194,15 +199,44 @@ def run_tool(agent: Agent, call) -> str: return output ``` -- Real tool name in `gen_ai.tool.name`, the model's `call.id` in `gen_ai.tool.call.id`. Async tools: `await` them inside the span. +- Real tool name in `gen_ai.tool.name`, the model's `call.id` in `gen_ai.tool.call.id`. Async tools are awaited inside the span (the `isawaitable` branch). - Tool failure: ERROR status + `error.type` on the tool span; the model still gets an error payload. Where the existing loop swallows exceptions into strings, add the status/`error.type` there (don't change what the model sees unless asked). - LiteLLM marks failed model calls ERROR + `error.type` itself. -- Multi-agent: each agent its own `name` (→ `gen_ai.agent.name`, one lane each). Run workers inside an outer `agent_span("orchestrator")` so the whole run is one trace; `asyncio.gather` keeps context. Same `conversation_id` to every `run_agent`. Delegation via a tool: call `run_agent` inside that tool's function (an `execute_tool` whose only child is `invoke_agent` renders as a delegation). +- Multi-agent: each agent its own `name` (→ `gen_ai.agent.name`, one lane each). Same `conversation_id` to every `run_agent`. + - Agents as tools (verified): the orchestrator's tool functions are `async def call(task): return await run_agent(worker, conversation_id, [{"role": "user", "content": task}])`. Each renders as `execute_tool ` → `invoke_agent ` (a delegation lane); parallel tool calls run in parallel via the `gather`. + - Code-driven orchestration: run workers inside an outer `agent_span("orchestrator")` so the whole run is one trace; `asyncio.gather` keeps context. - Thread pools (`run_in_executor`, `ThreadPoolExecutor`) lose OTel context: wrap with `contextvars.copy_context().run`. ## Step 6: cost (optional) and flush -Cost: Maple reads only `gen_ai.usage.cost`; LiteLLM writes `litellm.cost.total` (v2) / `hidden_params` (v1) → unpriced by default. If the user wants cost, sum `response._hidden_params.get("response_cost") or 0.0` per `run_agent` and `span.set_attribute("gen_ai.usage.cost", total)` on the `invoke_agent` span before returning (stream: last chunk's `usage.cost`). Session total becomes correct; per-model breakdown stays unpriced. Never add token pricing tables. +Cost: Maple reads only `gen_ai.usage.cost`; LiteLLM writes `litellm.cost.total` (v2) / `hidden_params` (v1) → unpriced by default. If the user wants cost, report the TURN total on the OUTERMOST agent span only. Maple subtracts a descendant's reported cost from its nearest cost-reporting ancestor, so cost on every nested `invoke_agent` undercounts the orchestrator. Replace `agent_span`: + +```py +from contextvars import ContextVar + +turn_costs: ContextVar[list | None] = ContextVar("turn_costs", default=None) + + +@contextmanager +def agent_span(name: str): + with tracer.start_as_current_span(f"invoke_agent {name}") as span: + span.set_attribute("gen_ai.operation.name", "invoke_agent") + span.set_attribute("gen_ai.agent.name", name) + if turn_costs.get() is not None: # a sub-agent: the outermost agent reports the cost + yield span + return + costs: list[float] = [] + token = turn_costs.set(costs) + try: + yield span + finally: + turn_costs.reset(token) + span.set_attribute("gen_ai.usage.cost", sum(costs)) +``` + +- After each non-stream `acompletion`: `turn_costs.get().append(response._hidden_params.get("response_cost") or 0.0)`. +- Stream: keep `cost = getattr(chunk.usage, "cost", None) or cost` for chunks whose `usage` is not None, append `cost` once after the loop. +- Session and turn totals are then exact (verified: equals the sum of `litellm.cost.total`); per-model breakdown stays unpriced. Never add token pricing tables. Flush (required for scripts, CLIs, Lambda/Cloud Run jobs, notebooks, workers; web servers only at shutdown). LiteLLM creates its span after the call via a background queue that drains at interpreter exit, after the provider shut down, so the last call is lost without this: @@ -224,16 +258,16 @@ Run one real conversation: 2+ messages with the same id, one tool call, one stre - [ ] Exactly one session per conversation, session id = the id passed; the second conversation is a separate session; no `trace:` sessions. - [ ] Framework shows **LiteLLM** (not Unidentified). -- [ ] One turn per top-level `invoke_agent`; transcript shows user prompts, assistant replies and tool calls. +- [ ] One turn per top-level `invoke_agent`; transcript shows user prompts, assistant replies and tool calls (tool rows come from `execute_tool` spans; a tool-only model reply shows empty, expected). - [ ] Spans: `invoke_agent ` (app scope), `chat ` (scope `litellm`) and `execute_tool ` inside it, same trace. No `litellm_request` / `raw_gen_ai_request` spans (those mean v1). - [ ] Every `chat` span has input and output tokens, including the streamed one; no call appears twice. - [ ] Tool spans have real names, arguments, results and call ids. - [ ] The failing tool is marked failed with its message; successful tools and model calls are not. - [ ] Sub-agents appear as separate lanes, all in the caller's session. -- [ ] Cost: unpriced, or the per-run total if Step 6 cost was added. +- [ ] Cost: unpriced, or (Step 6) `gen_ai.usage.cost` only on top-level `invoke_agent` spans, equal to the sum of the turn's `litellm.cost.total`. - [ ] Last call of a script run present (flush worked). - [ ] No attribute contains an API key, `Bearer ` or `sk-`. -- [ ] Proxy path: `chat` spans (service `litellm-proxy`) sit in the app's trace under `invoke_agent`, carrying the session id. +- [ ] Proxy path: `chat` spans (service `litellm-proxy`) sit in the app's trace under `invoke_agent` → `POST /chat/completions`, carrying the session id. LLM call count shows 2x (auth spans, known). Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` to the provider and confirm `gen_ai.conversation.id` on every `chat` span and the parent ids. @@ -248,5 +282,5 @@ Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExpor - Do not use `event_only` content capture. - Do not skip the flush in short-lived processes. - Do not create a second `TracerProvider` when one exists. -- Do not promise cost without the Step 6 recipe; do not add token pricing code. +- Do not promise cost without the Step 6 recipe; do not add token pricing code; do not put `gen_ai.usage.cost` on nested agent spans. - Do not print or commit real keys beyond the repo's convention. diff --git a/skills/maple-agent-tracing-llamaindex/SKILL.md b/skills/maple-agent-tracing-llamaindex/SKILL.md index 2405253de3..c2ac7af438 100644 --- a/skills/maple-agent-tracing-llamaindex/SKILL.md +++ b/skills/maple-agent-tracing-llamaindex/SKILL.md @@ -9,7 +9,7 @@ Goal: every conversation = one Maple Agent Session. Each `agent.run()` / `workfl Human guide with the reasoning: https://maple.dev/docs/agent-tracing/llamaindex -Mechanism: `openinference-instrumentation-llama-index` (scope `openinference.instrumentation.llama_index`) with `TraceConfig(enable_genai_semconv=True)`, which dual-writes `gen_ai.*` on span attributes. Maple reads `gen_ai.conversation.id` as the session key for these spans. Maple files them under "Unidentified" framework (generic GenAI/OpenInference buckets); that is expected. +Mechanism: `openinference-instrumentation-llama-index` (scope `openinference.instrumentation.llama_index`) with `TraceConfig(enable_genai_semconv=True)`, which dual-writes `gen_ai.*` on span attributes. Maple reads `gen_ai.conversation.id` as the session key. The `LlamaIndexForMaple` processor below stamps `llamaindex.instrumentor` so Maple labels the framework LlamaIndex (without it: "Unidentified"). ## Step 0: detect @@ -71,17 +71,22 @@ LLM_METHODS = (".chat", ".achat", ".stream_chat", ".astream_chat", class LlamaIndexForMaple(SpanProcessor): - """Sits in front of the exporter: one span per model call, agent names, no false HITL failures.""" + """Sits in front of the exporter: one span per model and tool call, agent names, no false HITL failures.""" def __init__(self, exporter_processor: SpanProcessor): self._next = exporter_processor self._open_llm_spans = {} def on_start(self, span, parent_context=None): + # Maple labels a span "LlamaIndex" by a llamaindex.* key; OpenInference writes none + span.set_attribute("llamaindex.instrumentor", "openinference") # instrument_tags({"gen_ai.agent.name": ...}) becomes an attribute, so sub-agents get lanes agent_name = active_instrument_tags.get().get("gen_ai.agent.name") if agent_name: span.set_attribute("gen_ai.agent.name", agent_name) + if span.name.endswith((".call_tool", ".aggregate_tool_results")): + # agent workflow steps around the tool span; by name alone Maple would count them as tool calls + span.set_attribute("gen_ai.operation.name", "invoke_workflow") if span.name.endswith(LLM_METHODS): self._open_llm_spans[span.context.span_id] = span self._next.on_start(span, parent_context) @@ -117,7 +122,7 @@ LlamaIndexInstrumentor().instrument( - Import `tracing` first in the entry point (app module, `main.py`, worker). `instrument()` must run before the first `agent.run()`. - Existing provider: skip `TracerProvider()`/`set_tracer_provider`; call `existing.add_span_processor(LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter())))` and pass `tracer_provider=existing`. -- The exporter MUST be added through `LlamaIndexForMaple`, never directly: without it every model call counts 2-3x in Maple (`_prepare_chat_with_tools` + nested same-name `astream_chat`/`achat` spans, all OpenInference kind LLM). +- The exporter MUST be added through `LlamaIndexForMaple`, never directly: without it every model call counts 2-3x in Maple (`_prepare_chat_with_tools` + nested same-name `astream_chat`/`achat` spans, all OpenInference kind LLM) and every tool call 3x (`call_tool` / `aggregate_tool_results` step spans are classified as tools by name). - No `service.name` default is acceptable: set `OTEL_SERVICE_NAME` (never `unknown_service`). ## Step 3: session id (required) @@ -169,9 +174,9 @@ async def run_agent(agent: FunctionAgent, message: str) -> str: return str(await handler) ``` - Parallel fan-out (`ctx.send_event` + `@step(num_workers=N)` + `ctx.collect_events`) stays in one trace with the session and agent tags. + Parallel fan-out (several `ctx.send_event` calls to separate steps or to a `@step(num_workers=N)`, joined with `ctx.collect_events`) stays in one trace with the session and agent tags. 3. Agents-as-tools: put the same `instrument_tags` block inside the tool function around the sub-agent's `run()`. -4. `AgentWorkflow` handoffs: one `AgentWorkflow.run` span, no per-agent spans, so no lanes. Tag the call with the root agent's name. If the user needs lanes, suggest running agents via workflow steps or tools (ask first; it changes app behavior). +4. `AgentWorkflow` handoffs: one `AgentWorkflow.run` span, no per-agent spans, so no lanes; every span carries the root agent's tag and the handoff is a `handoff` tool call. Tag the call with the root agent's name. If the user needs lanes, suggest running agents via workflow steps or tools (ask first; it changes app behavior). 5. Tool failures: a tool that raises → `FunctionTool.acall` status ERROR with the exception message (Maple counts it). Tools that `return "Error: ..."` look successful: convert to `raise` only where the user agrees. 6. HITL (`ctx.wait_for_event`): the first, suspended `FunctionTool.acall` ends ERROR `WaitingForEvent: ...`; the processor drops it. Nothing to add. 7. Known, unfixable here: `gen_ai.tool.call.arguments` = the tool's parameter schema (OpenInference GenAI mapping bug); no `gen_ai.tool.call.id` on tool spans. Real args are in the transcript's tool_call parts. @@ -201,11 +206,11 @@ Run one real conversation: 2+ messages with the same id, at least one tool call, - [ ] Exactly one session per conversation, id = the id you passed. A second conversation is a second session. Not `trace:` sessions. - [ ] One turn per `run()`; each trace roots at `FunctionAgent.run` / `AgentWorkflow.run` / `.run`. -- [ ] Framework shows "Unidentified" (expected for OpenInference LlamaIndex spans). +- [ ] Framework shows "LlamaIndex" ("Unidentified" = `llamaindex.instrumentor` missing: processor not installed or an old copy). - [ ] Transcript shows user messages, assistant replies and tool calls. - [ ] LLM call count ≈ real number of model calls (one `.astream_chat`/`.achat` span per call; no `_prepare_chat_with_tools` spans exported). - [ ] Every model span has a model and input/output tokens, including streamed calls. -- [ ] Tool spans `FunctionTool.acall` carry `gen_ai.tool.name` (real tool name) and a result; model and tool spans sit under the turn's agent span in the same trace. +- [ ] Tool call count = real tool calls (only `FunctionTool.acall`; `call_tool`/`aggregate_tool_results` carry `gen_ai.operation.name=invoke_workflow`). Tool spans carry `gen_ai.tool.name` (real tool name) and a result; model and tool spans sit under the turn's agent span in the same trace. - [ ] A failing tool is failed with its message; successful and HITL-approved tools are not. - [ ] Sub-agents appear as separate lanes with their names, under the caller's session. - [ ] Cost shows "unpriced". diff --git a/skills/maple-agent-tracing-mastra/SKILL.md b/skills/maple-agent-tracing-mastra/SKILL.md index c2c4517dd7..fa2f063aa5 100644 --- a/skills/maple-agent-tracing-mastra/SKILL.md +++ b/skills/maple-agent-tracing-mastra/SKILL.md @@ -11,13 +11,15 @@ Human guide with the reasoning: https://maple.dev/docs/agent-tracing/mastra Mechanism: Mastra's own tracing (`@mastra/observability`) converted to OTel GenAI semconv v1.38 by `@mastra/otel-exporter`, which runs its own BatchSpanProcessor. No OTel SDK or instrumentation package needed. Maple detects the vendor from resource `telemetry.sdk.name=@mastra/otel-exporter` and reads `gen_ai.conversation.id` as the session key. The exporter writes that key from span `metadata.threadId`, which Mastra sets from `memory.thread`. No thread = no session key. +Mastra 1.71 has three export gaps that a small span processor (Step 2) fixes; it is required in every setup: the `chat` span has no input messages (empty prompt side of the transcript), sub-agents get their own thread id (turn split, wrong session), and step spans carry the raw provider HTTP response (headers with cookies, full reply body) as `mastra.metadata.headers` / `mastra.metadata.body`, which `hideOutput` does not hide. + ## Step 0: detect 1. Versions: read `package.json` / lockfile for `@mastra/core`, `@mastra/observability`, `@mastra/otel-exporter`, `@mastra/memory`. - Need `@mastra/core` 1.x (written against 1.71) and Node >= 22.13. Mastra 0.x uses a different telemetry API: tell the user to upgrade; do not work around it. - `@mastra/core`, `@mastra/observability`, `@mastra/otel-exporter` must be from the same release train. Update all three together (`@latest`) if you add or bump one. 2. Find the `new Mastra({...})` instance (usually `src/mastra/index.ts`) and its current `observability` value. - - Already `new Observability({ configs: {...} })` → add the Maple exporter to the existing config's `exporters`; keep existing exporters (MastraStorageExporter, MastraPlatformExporter, Langfuse...). + - Already `new Observability({ configs: {...} })` → add the Maple exporter to the existing config's `exporters` and `mapleSpanProcessor` (Step 2) to its `spanOutputProcessors`; keep existing exporters (MastraStorageExporter, MastraPlatformExporter, Langfuse...) and processors. Do not add a second config: only one config is selected per request. - `observability: { default: { enabled: true } }` or any plain object → replace with `new Observability(...)` (a plain object silently installs a no-op). Keep `MastraStorageExporter` if the project uses Mastra Studio traces. - `@mastra/otel-bridge` already configured with an OTel SDK → do NOT add OtelExporter (double export). Point the existing OTel exporter at Maple instead. 3. Existing OTel NodeSDK / TracerProvider elsewhere in the app: leave it alone. OtelExporter does not use the global provider; Mastra spans become their own traces. @@ -37,10 +39,42 @@ Mechanism: Mastra's own tracing (`@mastra/observability`) converted to OTel GenA ## Step 2: install + init ```bash -npm install @mastra/observability@latest @mastra/otel-exporter@latest @opentelemetry/exporter-trace-otlp-proto +npm install @mastra/observability@latest @mastra/otel-exporter@latest ``` -Use the repo's package manager (pnpm/yarn/bun). Bump `@mastra/core` to latest in the same command if it is older than the other two. +Use the repo's package manager (pnpm/yarn/bun). Bump `@mastra/core` to latest in the same command if it is older than the other two. The OTLP protobuf exporter (`@opentelemetry/exporter-trace-otlp-proto`) is an optional dependency of `@mastra/otel-exporter` and installs with it; add it explicitly only if the project installs with `--omit=optional` / `--no-optional`. + +Create `src/mastra/maple-span-processor.ts` (next to the Mastra instance) with exactly this: + +```ts +import { SpanType, type SpanOutputProcessor } from "@mastra/core/observability" + +export const mapleSpanProcessor: SpanOutputProcessor = { + name: "maple-span-processor", + process(span) { + if (!span) return span + // One conversation id per trace: sub-agents get their own thread ids otherwise. + let root = span + while (root.parent) root = root.parent + const threadId = root.metadata?.threadId + if (threadId) span.metadata = { ...span.metadata, threadId } + // The model call span is created without its prompt: take the step's messages. + if (span.type === SpanType.MODEL_INFERENCE && span.input === undefined && span.parent?.input !== undefined) { + span.input = { messages: span.parent.input } + } + // Step spans carry the raw provider response (headers, cookies, full body) as metadata. + if (span.type === SpanType.MODEL_STEP && span.metadata) { + const { headers: _headers, body: _body, ...metadata } = span.metadata + span.metadata = metadata + } + return span + }, + async shutdown() {}, +} +``` + +- It must mutate and return the span it receives (Mastra drops a span when a processor returns a copy). +- Mastra appends its `SensitiveDataFilter` after user processors, so the copied prompt is still redacted. In the file that creates the Mastra instance: @@ -49,6 +83,7 @@ import { Mastra } from "@mastra/core/mastra" import { SpanType } from "@mastra/core/observability" import { Observability } from "@mastra/observability" import { OtelExporter } from "@mastra/otel-exporter" +import { mapleSpanProcessor } from "./maple-span-processor" export const mapleExporter = new OtelExporter({ provider: { @@ -69,7 +104,7 @@ export const mastra = new Mastra({ serviceName: "support-agent", exporters: [mapleExporter], excludeSpanTypes: [SpanType.MODEL_CHUNK], - spanOutputProcessors: [conversationIdFromRoot], // only if Step 5 applies + spanOutputProcessors: [mapleSpanProcessor], }, }, }), @@ -77,6 +112,7 @@ export const mastra = new Mastra({ ``` - `protocol: "http/protobuf"` is mandatory: the custom provider defaults to `http/json`, and a missing protocol package disables tracing with one console error. +- Keep `excludeSpanTypes: [SpanType.MODEL_CHUNK]`: without it every streamed chunk is its own span (~40% of all spans in a supervisor run). - Endpoint is the base URL; the exporter appends `/v1/traces` and `/v1/logs` (and strips them if present). - `serviceName` is required; use the project/service name, never leave `mastra-service`. Set `deployment.environment.name` from the app's env (e.g. `process.env.NODE_ENV`). - Logs: the exporter also sends Mastra log records (warn+) to `/v1/logs` by default. Keep that unless the user wants traces only: `signals: { logs: false }` on OtelExporter. @@ -113,33 +149,19 @@ HITL resumes (`approveToolCallGenerate` / `declineToolCallGenerate` / `approveTo ## Step 4: content -- On by default: `gen_ai.input.messages` / `gen_ai.output.messages` on `chat` spans, `gen_ai.system_instructions` on `invoke_agent`, tool args/results on `execute_tool`. Change nothing to keep it. -- Opt-out per request: `tracingOptions: { hideInput: true, hideOutput: true }` (whole trace). +- On by default: `gen_ai.output.messages` on `chat` spans, tool args/results on `execute_tool`. `gen_ai.input.messages` on `chat` spans (with the system prompt as the first message) only exists because of `mapleSpanProcessor`; without it the transcript has replies but no prompts. Tool calls inside earlier history are summarized by Mastra as `[tool: ]` text; the full calls are on the `execute_tool` spans. +- `gen_ai.system_instructions` on `invoke_agent` is plain text, which Maple does not decode; the system prompt shows through the system message in the `chat` input instead. +- Opt-out per request: `tracingOptions: { hideInput: true, hideOutput: true }` (whole trace). It hides span input/output only, not metadata; `mapleSpanProcessor` removes the one metadata copy of the reply (`mastra.metadata.body`). - `SensitiveDataFilter` is auto-applied (redacts values under keys like password/token/apiKey/authorization/secret). Do not disable it (`sensitiveDataFilter: false`) unless the user asks. - Serialization caps: 128 KiB/string, 50 items/array, 50 keys/object, depth 8. If agents keep > ~40 messages of history (`lastMessages` > 40 or custom history), add `serializationOptions: { maxArrayLength: 200 }` to the config, or the newest messages are cut from the transcript. ## Step 5: tools, errors, sub-agents - Tools: `createTool({ id, description, inputSchema, execute })`. Span name `execute_tool `, `gen_ai.tool.name` = id. Give tools real ids. -- Failures must throw (`throw new Error("...")`). Thrown → span ERROR + `error.type=TOOL_EXECUTION_FAILED` + message. Returning `{ error }` = counted as success in Maple. If a tool swallows errors into a return value and the user wants failures visible, rethrow. +- Failures must throw (`throw new Error("...")`). Thrown → span status ERROR with the message as status message, `error.type=unknown`, plus an `exception` event. Returning `{ error }` = counted as success in Maple. If a tool swallows errors into a return value and the user wants failures visible, rethrow. - Every `new Agent({ id, name, ... })` needs a distinct `name`: it is `gen_ai.agent.name`, which Maple uses for lanes. -- Supervisor agents (`agents: {...}` on an Agent), `agent.network()`, or workflow steps that call agents which have their own `memory`: sub-agents get their own thread ids, so one trace carries several `gen_ai.conversation.id` values and Maple may pick the wrong one or split the turn. Add this processor and list it in `spanOutputProcessors`: - -```ts -import type { SpanOutputProcessor } from "@mastra/core/observability" - -export const conversationIdFromRoot: SpanOutputProcessor = { - name: "conversation-id-from-root", - process(span) { - let root = span - while (root?.parent) root = root.parent - const threadId = root?.metadata?.threadId - if (span && threadId) span.metadata = { ...span.metadata, threadId } - return span - }, - async shutdown() {}, -} -``` +- Supervisor agents (`agents: {...}` on an Agent): each delegation is `execute_tool agent-` → `invoke_agent ` inside the supervisor's trace, one lane per sub-agent. Mastra gives each delegation a thread id `-`, which sorts after the real id, so without `mapleSpanProcessor` Maple picks a sub-agent's id as the session and splits the turn. Same for `agent.network()` and workflow steps calling agents with their own `memory`. The processor fixes all of them; do not remove it. +- HITL (`requireApproval: true` tools): the model call that requests the tool is exported twice, once without tokens or output when the run suspends and once with its tokens when the run resumes. Maple shows one extra model call with 0 tokens per approval; token totals are right. Expected, not a setup error. - Workflow steps that call an agent: pass `tracingContext` from the step's `execute` args into `agent.generate(prompt, { tracingContext })`, or the agent starts a separate trace. @@ -159,8 +181,9 @@ Run one conversation: 2+ user messages with the same thread id (one streamed), a - Framework shows **Mastra** (not Unidentified). - Exactly one session per conversation, id = the thread id; the second conversation is a separate session; no `trace:` sessions for chat turns. - One turn per `generate()` / `stream()` / workflow run, labeled with the user message. -- Transcript shows user prompts, assistant replies and tool calls. -- `chat ` spans have model, provider and input/output tokens, including the streamed turn. +- Transcript shows user prompts, assistant replies and tool calls. If it has replies but no prompts, `mapleSpanProcessor` is missing from `spanOutputProcessors`. +- `chat ` spans have model, provider, `gen_ai.input.messages` and input/output tokens, including the streamed turn. Exception: with HITL, one `chat` span per approval has no tokens (Step 5). +- No span has `mastra.metadata.headers` or `mastra.metadata.body`. - `execute_tool ` spans have the real tool name, arguments and result; a throwing tool is marked failed with its message; successful tools are not. - Supervisor/workflow: one session, one turn per run, one lane per sub-agent `name`, all spans in one trace. - Cost shows as unpriced (expected: Mastra emits no cost attribute). @@ -175,7 +198,8 @@ Run one conversation: 2+ user messages with the same thread id (one streamed), a - Do not call `generate()` / `stream()` without a stable `memory.thread` (or `tracingOptions.metadata.threadId`) in a multi-turn chat. - Do not enable `includeInternalSpans`. - Do not stamp `maple_ai.session.id` on Mastra spans; it re-vendors them away from Mastra decoding. Use the thread id. -- Do not mix `@mastra/observability` / `@mastra/otel-exporter` versions from different releases (spans arrive with no model calls or tokens). +- Do not mix `@mastra/core` / `@mastra/observability` / `@mastra/otel-exporter` versions from different releases (the exporter picks the model-call span from features both packages report). +- Do not leave out `mapleSpanProcessor`, and do not replace it with a processor that returns a copy of the span. - Do not return error objects from tools you want counted as failures; throw. - Do not exit a script without `await mastra.shutdown()`. - Do not promise cost or time-to-first-token in Maple: Mastra emits neither under keys Maple reads. Reasoning tokens are in the output total but not broken out. diff --git a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md index 12a196757d..a3de1a940d 100644 --- a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md +++ b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md @@ -128,6 +128,7 @@ sealed class ConversationIdProcessor : BaseProcessor - Default source names carry an `Experimental.` prefix; `AddSource("Microsoft.Agents.AI.*")` matches nothing. Keep the leading `*`. - `UseOpenTelemetry()` on the agent (1.22) also instruments its chat client. Do not add a second `UseOpenTelemetry()` on the `IChatClient`. +- Name every tool: `AIFunctionFactory.Create(GetWeather, name: "get_weather")`. Without `name:`, a local function in top-level `Program.cs` is exported as `_Main_g_GetWeather_0_3`. - Workflows: `.WithOpenTelemetry()` on the `WorkflowBuilder` (source `Microsoft.Agents.AI.Workflows`, matched by the wildcard). - In a hosted app, put the same sources, processor and exporter in `builder.Services.AddOpenTelemetry().WithTracing(...)`. @@ -144,8 +145,8 @@ async def handle_message(session: AgentSession, text: str) -> str: - Id: the app's chat/thread id, or `AgentSession.session_id` (create sessions with `agent.create_session(session_id=chat_id)` when the app has an id). Must be stable across all turns of one conversation and differ between conversations. - Streaming: the whole `async for update in agent.run(..., stream=True)` loop goes inside the `with`. -- Approval resumes (`request.to_function_approval_response(...)` passed back to `agent.run`) and workflow runs go inside the same `with`. -- Build workflows (`WorkflowBuilder(...).build()`, `SequentialBuilder`, `ConcurrentBuilder`, ...) inside the `with`; a build outside it emits a stray `workflow.build` trace with no id. +- Approval resumes (`request.to_function_approval_response(...)` passed back to `agent.run`) and workflow runs go inside the same `with`. 1.19 logs a WARN "Ignored an approval response ... did not match" on each resume even though the tool runs; ignore it if the `execute_tool` span is there once. +- Build workflows (`WorkflowBuilder(...).build()`, `SequentialBuilder`, `ConcurrentBuilder`, ...) inside the `with`. `build()` always emits a separate one-span `workflow.build` trace: inside the `with` it joins the session; outside it becomes a stray one-span session. Executors (fan-out included) inherit the id. - .NET: `ConversationIdProcessor.Current.Value = chatId;` in the request handler before `RunAsync`/`RunStreamingAsync`. ## Step 4: content @@ -224,7 +225,7 @@ Run one real conversation (2-3 turns, one tool call; a second conversation if ch - Spans arrive and the process exits cleanly (flush ran); `service.name` is yours, not `agent_framework` / `unknown_service`. - Every span of every turn in one conversation has the same `gen_ai.conversation.id`; a second conversation has a different one. In Maple: one session per conversation, not `trace:` sessions. -- Each turn: `invoke_agent ` root with `chat ` and `execute_tool ` descendants in the same trace (streamed turn included). +- Each turn: `invoke_agent ` root (.NET: `invoke_agent ()`) with `chat ` and `execute_tool ` descendants in the same trace (streamed turn included). - `chat` spans: `gen_ai.request.model`, `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` (also on the streamed turn), `gen_ai.input.messages` / `gen_ai.output.messages` as JSON arrays of `{role, parts}` (Python MAF; SK: on `invoke_agent` only). - `execute_tool`: real tool name, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`; a raising tool has ERROR status + `error.type`, successful ones don't. - Sub-agents have distinct `gen_ai.agent.name`s. diff --git a/skills/maple-agent-tracing-opentelemetry/SKILL.md b/skills/maple-agent-tracing-opentelemetry/SKILL.md index 07f38dd923..bad407e072 100644 --- a/skills/maple-agent-tracing-opentelemetry/SKILL.md +++ b/skills/maple-agent-tracing-opentelemetry/SKILL.md @@ -101,7 +101,8 @@ Escape hatch, only when a framework's spans carry a session key Maple ignores fo - Usage on `chat` spans only, never cumulative totals on `invoke_agent`. - Keys: `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` (ints). Not read: `total_tokens`, `reasoning_tokens`, `cache_read_input_tokens`. - Copy the provider's raw numbers. Maple interprets by `gen_ai.provider.name`: `anthropic` → input EXCLUDES cache (send Anthropic's raw `input_tokens`, NOT input+cache as the spec says, or cache is double counted); `gcp.gemini`/`gcp.vertex_ai` → input includes cache, output excludes thoughts (raw Gemini counts); `openai`/`openrouter`/other → input includes cache, output includes reasoning (OpenAI shape). -- Streaming OpenAI-compatible: `stream_options: {include_usage: true}`; usage is on the last chunk (empty `choices`). OpenRouter always sends usage. +- Streaming OpenAI-compatible: `stream_options: {include_usage: true}`. OpenAI sends usage in an extra last chunk with empty `choices`; OpenRouter always sends usage + `cost` on the chunk carrying `finish_reason`. Read `chunk.usage` before skipping chunks without choices. +- OpenRouter (Claude included): `prompt_tokens` already includes `cached_tokens` → provider `openrouter`, copy as is. Optional: `prompt_tokens_details.cache_write_tokens` → `gen_ai.usage.cache_write.input_tokens`. - Cost: `gen_ai.usage.cost` (double, USD) on `chat` spans. OpenRouter returns `usage.cost` → copy it. Other providers return none → compute only if the project has a price table; otherwise leave it (Maple shows "unpriced"; it never prices tokens). - `gen_ai.response.id` always (dedupes against gateway mirrors such as OpenRouter Broadcast). diff --git a/skills/maple-agent-tracing-opentelemetry/references/python.md b/skills/maple-agent-tracing-opentelemetry/references/python.md index 17e330b58b..97b03c207a 100644 --- a/skills/maple-agent-tracing-opentelemetry/references/python.md +++ b/skills/maple-agent-tracing-opentelemetry/references/python.md @@ -1,6 +1,6 @@ # Python reference (3.10+) -Tested pattern: `opentelemetry-sdk` 1.45, `opentelemetry-exporter-otlp-proto-http` 1.45, `openai` 3.20 against an OpenAI-compatible Chat Completions API (OpenRouter here). +Tested pattern: `opentelemetry-sdk` 1.45, `opentelemetry-exporter-otlp-proto-http` 1.45, `openai` 3.20 against an OpenAI-compatible Chat Completions API (OpenRouter here), Python 3.14. This is a complete loop. If the project already has a loop, keep its structure and copy only the span code: `invoke_agent` around one agent run, `chat` around each model call, `execute_tool` around each tool call, `to_semconv` for messages. diff --git a/skills/maple-agent-tracing-opentelemetry/references/typescript.md b/skills/maple-agent-tracing-opentelemetry/references/typescript.md index b578c53030..1c5780ee02 100644 --- a/skills/maple-agent-tracing-opentelemetry/references/typescript.md +++ b/skills/maple-agent-tracing-opentelemetry/references/typescript.md @@ -1,6 +1,6 @@ # TypeScript reference (Node.js 20+) -Tested pattern: OpenTelemetry JS SDK 2.11 (`@opentelemetry/sdk-trace-node`, `sdk-trace-base`, `resources` 2.11; `exporter-trace-otlp-proto` 0.222; `api` 1.9), `openai` 7.23 against an OpenAI-compatible Chat Completions API (OpenRouter here). +Tested pattern: OpenTelemetry JS SDK 2.11 (`@opentelemetry/sdk-trace-node`, `sdk-trace-base`, `resources` 2.11; `exporter-trace-otlp-proto` 0.222; `api` 1.9), `openai` 7.23 against an OpenAI-compatible Chat Completions API (OpenRouter here), Node.js 26, run with `tsx`. This is a complete loop. If the project already has a loop, keep its structure and copy only the span code: `invoke_agent` around one agent run, `chat` around each model call, `execute_tool` around each tool call, `toSemconv` for messages. @@ -274,6 +274,8 @@ try { } ``` +ESM (`"type": "module"`) for top-level `await`. Run with `npx tsx main.ts` or the project's bundler/tsc; plain `node main.ts` fails on the extensionless `./tracing` import. + ## Sub-agent (delegation through a tool) ```ts diff --git a/skills/maple-agent-tracing-provider-sdks/SKILL.md b/skills/maple-agent-tracing-provider-sdks/SKILL.md index d0c3ef2fc9..b95aba1f6c 100644 --- a/skills/maple-agent-tracing-provider-sdks/SKILL.md +++ b/skills/maple-agent-tracing-provider-sdks/SKILL.md @@ -85,7 +85,8 @@ Rules for both: - OpenAI Chat Completions streaming: ALWAYS pass `stream_options={"include_usage": True}` (Python) or use the helper's `onText` path (TS sets it). Otherwise the streamed call has 0 tokens. The final usage chunk has empty `choices`; skip it when reading text (`if chunk.choices`). - Anthropic / Gemini streams include usage; nothing to add. -- No instrumentation emits cost; Maple doesn't price tokens → sessions show "unpriced". Only if the user asks and the gateway returns cost (OpenRouter `usage.cost`): in the TS helper set `gen_ai.usage.cost` (USD) on the chat span. Don't add pricing tables. +- No Python instrumentation emits cost; Maple doesn't price tokens → sessions show "unpriced". The TS helper copies OpenRouter's `usage.cost` (USD) to `gen_ai.usage.cost`; direct OpenAI sends none. Don't add pricing tables. +- Claude/Gemini models via OpenRouter's OpenAI-compatible endpoint with the `openai` SDK → spans say `gen_ai.provider.name=openai` with OpenAI-shaped usage. Correct; don't override it. ## Step 7: flush @@ -105,7 +106,7 @@ Run one real conversation: 3+ user messages with the same id, one tool call, one - [ ] Tool spans have the real tool name, call id, arguments and result. - [ ] The failing tool is marked failed with its message; successful tools are not. - [ ] Sub-agents appear as separate lanes under their own names, inside the caller's session. -- [ ] Cost shows "unpriced" unless you set `gen_ai.usage.cost`. +- [ ] Cost shows "unpriced", except TS calls through OpenRouter (helper records `usage.cost`). - [ ] No attribute contains an API key, `Bearer `, `sk-` or `maple_sk_`. Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` and confirm the `invoke_agent` span has `gen_ai.conversation.id` and the model spans have `gen_ai.input.messages` and `gen_ai.usage.*`. @@ -113,8 +114,10 @@ Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExpor ## Known limitations (tell the user, don't work around) - Framework facet shows "Unidentified". -- Python Anthropic with prompt caching: the instrumentation reports `input_tokens` including cache, Maple treats Anthropic input as excluding cache → cache-read tokens counted twice in totals. -- No cost. +- Python Anthropic with prompt caching: the instrumentation reports `input_tokens` = raw + cache reads + cache writes, Maple treats Anthropic input as excluding cache → both cache buckets counted twice in totals. +- Python: no cost (unpriced). +- Anthropic instrumentation 1.2b0 records no time to first chunk for `messages.stream()`. +- Gemini path (Python instrumentation, TS mapping) not yet run end to end against a live model; OpenAI and Anthropic paths are verified. ## Do not diff --git a/skills/maple-agent-tracing-provider-sdks/references/python.md b/skills/maple-agent-tracing-provider-sdks/references/python.md index 31f9216575..3a745fc116 100644 --- a/skills/maple-agent-tracing-provider-sdks/references/python.md +++ b/skills/maple-agent-tracing-provider-sdks/references/python.md @@ -139,7 +139,9 @@ with agent_span("support_agent", conversation_id): ]}) ``` -`client.messages.stream(...)` is instrumented too; usage arrives without extra options. +`client.messages.stream(...)` is instrumented too (`with client.messages.stream(**kw) as s: response = s.get_final_message()`); usage arrives without extra options. + +Anthropic SDK through OpenRouter: `anthropic.Anthropic(base_url="https://openrouter.ai/api", api_key=OPENROUTER_API_KEY)` (no `/v1`; `auth_token=` also works), model ids like `anthropic/claude-haiku-4.5`. Spans say `gen_ai.provider.name=anthropic`. ## Gemini diff --git a/skills/maple-agent-tracing-provider-sdks/references/typescript.md b/skills/maple-agent-tracing-provider-sdks/references/typescript.md index b1065cd426..9228ba325a 100644 --- a/skills/maple-agent-tracing-provider-sdks/references/typescript.md +++ b/skills/maple-agent-tracing-provider-sdks/references/typescript.md @@ -125,6 +125,9 @@ export function tracedChat(client: OpenAI, params: ChatParams, onText?: (delta: "gen_ai.usage.reasoning.output_tokens": completion.usage.completion_tokens_details?.reasoning_tokens ?? 0, }) + // OpenRouter adds the call's price in USD to usage. Other providers don't send one. + const cost = (completion.usage as { cost?: number }).cost + if (cost !== undefined) span.setAttribute("gen_ai.usage.cost", cost) } if (captureContent) { const output = completion.choices.map((c) => ({ diff --git a/skills/maple-agent-tracing-spring-ai/SKILL.md b/skills/maple-agent-tracing-spring-ai/SKILL.md index 564b6eb6f3..8ff680c90e 100644 --- a/skills/maple-agent-tracing-spring-ai/SKILL.md +++ b/skills/maple-agent-tracing-spring-ai/SKILL.md @@ -39,13 +39,9 @@ Keep the existing `spring-ai-bom` import and model starter. Add: org.springframework.boot spring-boot-starter-opentelemetry - - org.springframework.boot - spring-boot-starter-actuator - ``` -Gradle: `implementation("org.springframework.boot:spring-boot-starter-opentelemetry")` and `...:spring-boot-starter-actuator`. +Gradle: `implementation("org.springframework.boot:spring-boot-starter-opentelemetry")`. Actuator is not needed on Boot 4 (verified); keep it if the app already has it. ### 2b. Properties @@ -63,13 +59,12 @@ management.otlp.metrics.export.url=https://ingest.maple.dev/v1/metrics management.otlp.metrics.export.headers.Authorization=Bearer ${MAPLE_INGEST_KEY} maple.ai.capture-content=true -spring.ai.openai.chat.stream-options.include-usage=true ``` - The endpoint property takes the FULL URL including `/v1/traces` (Boot does not append it). - `management.tracing.sampling.probability=1.0` is REQUIRED. Default is `0.1`: 90% of turns silently missing. - The starter also exports metrics, to `localhost:4318` by default. Either point them at Maple (above) or set `management.otlp.metrics.export.enabled=false`. Never leave the default. -- `include-usage` only for the OpenAI starter (and OpenAI-compatible gateways): without it streamed calls report no tokens. Per call alternative: `OpenAiChatOptions.builder().streamUsage(true)`. +- Streamed tokens: Spring AI 2.0's OpenAI model requests usage on streams by default. If the app sets ANY `spring.ai.openai.chat.stream-options.*` property, also set `spring.ai.openai.chat.stream-options.include-usage=true` (once stream options exist, unset means false and streamed turns report no tokens). - Boot 4.1+ also maps `OTEL_EXPORTER_OTLP_ENDPOINT` (base URL, Boot appends `/v1/traces`), `OTEL_EXPORTER_OTLP_HEADERS`, `OTEL_SERVICE_NAME`, `OTEL_RESOURCE_ATTRIBUTES`. Use them if the repo configures via env; keep the sampling property. - Do NOT set `management.opentelemetry.tracing.limits.max-attribute-value-length`: truncated JSON content no longer parses and Maple drops it. - Leave `spring.ai.chat.observations.log-prompt/log-completion` and `spring.ai.chat.client.observations.*` as they are. They only log; they never put content on spans. @@ -77,15 +72,20 @@ spring.ai.openai.chat.stream-options.include-usage=true ### 2c. OTel Java agent present -The agent does not convert Micrometer Observations into spans, so Steps 2a-2b are still required. With both: -- add `-Dotel.instrumentation.openai-java.enabled=false` (the agent instruments the OpenAI Java SDK under Spring AI 2.0's OpenAI starter: duplicate `chat` spans); -- add `management.observations.enable.http.server.requests=false` (duplicate server spans); -- point the agent's `OTEL_EXPORTER_OTLP_*` at Maple too. -Tell the user this combination is untested by Maple; recommend removing the agent for this service if nothing else needs it. +The agent does not turn Micrometer Observations into spans, so Steps 2a-2b and 3 are still required. But agent + starter as-is is BROKEN: Boot's SDK and the agent don't share context, so every Spring AI span becomes its own trace (no hierarchy, model calls split from their session). Verified with agent 2.31.1. Either remove the agent for this service, or hand Micrometer the agent's `OpenTelemetry` (add ONLY while the agent is attached; without it `GlobalOpenTelemetry.get()` is a no-op and nothing is traced): + +```java +@Bean +io.opentelemetry.api.OpenTelemetry openTelemetry() { + return io.opentelemetry.api.GlobalOpenTelemetry.get(); +} +``` + +Then the agent exports everything: set `OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev`, `OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer `, `OTEL_SERVICE_NAME`, `OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=` for the agent. Boot's `management.opentelemetry.*` export and sampling properties no longer apply (agent default sampler records everything). Also add `-Dotel.instrumentation.openai-java.enabled=false` as a precaution (no duplicate `chat` spans seen with Spring AI 2.0.1, but the agent ships OpenAI SDK instrumentation). HTTP client spans appear twice (Boot + agent); they are not AI spans and don't affect sessions. ### 2d. Spring AI 1.1 on Boot 3.5 -- Deps: `io.micrometer:micrometer-tracing-bridge-otel` + `io.opentelemetry:opentelemetry-exporter-otlp` + actuator (no `spring-boot-starter-opentelemetry`). +- Deps: `io.micrometer:micrometer-tracing-bridge-otel` + `io.opentelemetry:opentelemetry-exporter-otlp` + `spring-boot-starter-actuator` (required on Boot 3.5; no `spring-boot-starter-opentelemetry`). - Properties: `management.otlp.tracing.endpoint=https://ingest.maple.dev/v1/traces`, `management.otlp.tracing.headers.Authorization=Bearer ...`; sampling property unchanged. Stream usage: `spring.ai.openai.chat.options.stream-usage=true`. - Step 3 class: replace `tools.jackson.databind.json.JsonMapper.shared().writeValueAsString(...)` with a Jackson 2 `com.fasterxml.jackson.databind.ObjectMapper` (wrap the checked `JsonProcessingException`), and delete the two `getToolCallId()` lines (not in 1.1). @@ -103,7 +103,6 @@ import java.util.Map; import io.micrometer.common.KeyValue; import io.micrometer.observation.Observation; import io.micrometer.observation.ObservationFilter; -import io.micrometer.observation.ObservationPredicate; import io.micrometer.observation.ObservationRegistry; import tools.jackson.databind.json.JsonMapper; @@ -127,13 +126,6 @@ public class MapleAiObservationConfig { /** Agent name for a ChatClient: `.defaultAdvisors(a -> a.param(AGENT_NAME, "support_agent"))`. */ public static final String AGENT_NAME = "gen_ai.agent.name"; - // Advisor spans carry nothing Maple reads, and their names ("tool _calling ", - // "message_chat_memory") would be counted as extra tool and LLM calls. - @Bean - ObservationPredicate skipAdvisorObservations() { - return (name, context) -> !(context instanceof AdvisorObservationContext); - } - @Bean ObservationFilter mapleGenAiAttributes(@Value("${maple.ai.capture-content:false}") boolean captureContent) { return context -> { @@ -144,6 +136,11 @@ public class MapleAiObservationConfig { client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.agent.name", agent)); } } + else if (context instanceof AdvisorObservationContext advisor) { + // Advisor span names ("tool _calling ", "message_chat_memory") would be + // counted as extra tool and LLM calls. A neutral name keeps them as plumbing. + advisor.setContextualName("spring_ai advisor"); + } else if (context instanceof ToolCallingObservationContext tool) { tool.addLowCardinalityKeyValue(KeyValue.of("gen_ai.tool.name", tool.getToolDefinition().name())); if (tool.getToolCallId() != null) { @@ -204,7 +201,9 @@ public class MapleAiObservationConfig { } ``` -Kotlin project: translate one to one (e.g. `ObservationPredicate { _, context -> context !is AdvisorObservationContext }`), same beans, same keys. +Kotlin project: translate one to one (e.g. `ObservationFilter { context -> ...; context }`), same beans, same keys. + +Do NOT drop advisor observations with an `ObservationPredicate` instead of renaming them: on `.stream()` calls Spring AI takes the model span's parent from the Reactor context, which then holds the skipped (no-op) advisor observation, so the streamed `chat` span becomes its own trace outside the session (verified). ## Step 4: Session id (one conversation = one session) @@ -229,7 +228,7 @@ Maple reads ONLY `spring.ai.chat.client.conversation.id` for Spring AI, on the ` ## Step 6: Flush - Web app: nothing. Boot shuts down the `SdkTracerProvider` on context close, which flushes. -- `CommandLineRunner` / batch: let `run()` return or use `SpringApplication.exit(context)`. `System.exit()` is fine (shutdown hook); `Runtime.halt()` / SIGKILL lose the last batch. +- `CommandLineRunner` / batch: exit explicitly with `System.exit(SpringApplication.exit(SpringApplication.run(App.class, args)))` in `main`. If `main` just returns, the OpenAI starter's HTTP client keeps non-daemon threads alive ~60 s; the context (and the final span flush) only closes after that (verified). `Runtime.halt()` / SIGKILL lose the last batch. - Serverless (Spring Cloud Function on Lambda etc.): inject `io.opentelemetry.sdk.trace.SdkTracerProvider` and call `tracerProvider.forceFlush().join(10, TimeUnit.SECONDS)` at the end of EVERY invocation; never shut it down. - Batch delay is 5 s; wait before checking Maple. @@ -241,7 +240,7 @@ Run one real conversation: 2-3 turns with the same conversation id including one - [ ] The two conversations are two different sessions. - [ ] Framework shows **Spring AI**. - [ ] Every turn arrived (count = number of top-level `ChatClient` calls). Missing turns = sampling property not applied. -- [ ] Spans: `spring_ai chat_client` (agent, with your agent name), `chat `, `execute_tool `. NO `tool _calling `, `call`, `message_chat_memory` spans (else the predicate bean isn't loaded). +- [ ] Spans: `spring_ai chat_client` (agent, with your agent name), `chat `, `execute_tool `. Advisor spans are named `spring_ai advisor`; NO spans named `tool _calling `, `call`, `stream`, `message_chat_memory` (else the filter isn't loaded). - [ ] LLM call count = number of `chat ` spans (not doubled); tool call count = number of real tool invocations. - [ ] Transcript shows user messages, replies, tool calls with arguments and results (else `maple.ai.capture-content` isn't `true` or the class isn't scanned). - [ ] Input and output tokens on every model call, including the streamed one. @@ -257,6 +256,7 @@ Run one real conversation: 2-3 turns with the same conversation id including one - Do not stamp `maple_ai.session.id` on Spring AI spans; the conversation id param is the supported path. - Do not put a constant or per-request conversation id on calls. - Do not register a second `ToolExecutionExceptionProcessor` next to an existing one (ambiguous bean). -- Do not run the OTel Java agent's OpenAI instrumentation alongside Spring AI (duplicate `chat` spans). +- Do not attach the OTel Java agent next to the starter without the `GlobalOpenTelemetry` bean from Step 2c (every span becomes its own trace). - Do not set an attribute length limit (breaks content JSON). +- Do not use an `ObservationPredicate` to drop Spring AI observations (breaks the streamed turn's trace). - Do not use this skill for LangChain4j; use the generic OpenTelemetry GenAI guide: https://maple.dev/docs/agent-tracing/opentelemetry From 5981b96ddf0e9ace12802da877b5cc7eab9b4c66 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Mon, 28 Sep 2026 19:34:13 +0200 Subject: [PATCH 03/39] docs(agent-tracing): editorial pass, cross-links from instrumentation and onboarding --- .../src/content/docs/agent-tracing.mdx | 2 +- .../src/content/docs/agent-tracing/litellm.md | 2 +- .../content/docs/agent-tracing/spring-ai.md | 4 +++- skills/maple-agent-tracing-langchain/SKILL.md | 16 +++++++-------- skills/maple-agent-tracing-litellm/SKILL.md | 20 +++++++++---------- .../maple-agent-tracing-llamaindex/SKILL.md | 18 ++++++++--------- skills/maple-agent-tracing-mastra/SKILL.md | 16 +++++++-------- .../SKILL.md | 16 +++++++-------- .../maple-agent-tracing-openrouter/SKILL.md | 18 ++++++++--------- .../SKILL.md | 20 +++++++++---------- .../SKILL.md | 18 ++++++++--------- .../maple-agent-tracing-pydantic-ai/SKILL.md | 16 +++++++-------- skills/maple-agent-tracing-strands/SKILL.md | 16 +++++++-------- skills/maple-agent-tracing/SKILL.md | 7 ++++--- skills/maple-onboard/SKILL.md | 2 ++ 15 files changed, 98 insertions(+), 93 deletions(-) diff --git a/apps/landing/src/content/docs/agent-tracing.mdx b/apps/landing/src/content/docs/agent-tracing.mdx index fa8b2280dc..0556c16cfa 100644 --- a/apps/landing/src/content/docs/agent-tracing.mdx +++ b/apps/landing/src/content/docs/agent-tracing.mdx @@ -36,7 +36,7 @@ Use your key from **Settings → Ingestion**. Without one, the agent uses a plac ))} -If you use a gateway like OpenRouter or LiteLLM **and** a framework, pick one of the two for model calls. Instrumenting both records every call twice; Maple dedupes calls that share a response id, but the framework guide is the one that gives you turns, tools and sub-agents. +If you use a gateway like OpenRouter or LiteLLM **and** a framework, pick one of the two for model calls. Instrumenting both records every call twice. Maple can merge the two copies only when both carry the same response id, and many frameworks don't record one, so treat the framework guide as the source of truth: it's the one that gives you turns, tools and sub-agents. The [OpenRouter guide](/docs/agent-tracing/openrouter) covers the one setup where both work together. ## What every guide sets up diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md index e60342ebac..8a30280567 100644 --- a/apps/landing/src/content/docs/agent-tracing/litellm.md +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -282,7 +282,7 @@ async def stream_agent(agent: Agent, conversation_id: str, messages: list): messages.append({"role": "assistant", "content": text}) ``` -Cost shows as unpriced by default. LiteLLM prices every call, but the v2 logger writes the price to `litellm.cost.total` (and v1 buries it in the `hidden_params` JSON), and Maple reads cost only from `gen_ai.usage.cost` and never prices tokens itself. +Cost shows as unpriced by default. LiteLLM prices every call, but the v2 logger writes the price to `litellm.cost.total` (and v1 buries it in the `hidden_params` JSON), and Maple reads cost only from `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or `llm.cost.total`, and never prices tokens itself. To get cost into Maple, add up LiteLLM's price for every call of a turn, sub-agents included, and put the total on the outermost `invoke_agent` span. Only the outermost one: Maple subtracts a sub-agent's reported cost from the agent above it, on the assumption that the parent's figure already includes it, so a per-agent total on every level undercounts the orchestrator. Replace `agent_span` in `agent.py`: diff --git a/apps/landing/src/content/docs/agent-tracing/spring-ai.md b/apps/landing/src/content/docs/agent-tracing/spring-ai.md index 1240f134a6..9d78aba8ff 100644 --- a/apps/landing/src/content/docs/agent-tracing/spring-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/spring-ai.md @@ -9,7 +9,9 @@ icon: "spring" Spring AI instruments itself with Micrometer Observations. Add Spring Boot's OpenTelemetry starter and every `ChatClient` call becomes a `spring_ai chat_client` span, with a `chat ` span per model call and an `execute_tool ` span per tool call. The model spans follow the OpenTelemetry GenAI conventions (model, token counts, cache tokens, finish reasons, response id), and Maple recognizes all of it as Spring AI without a separate instrumentation library. -Four defaults work against you. Spring Boot samples 10% of traces, so nine turns out of ten never arrive. Prompts and replies are never written to spans: `log-prompt` and `log-completion` send them to the application log. A tool that throws ends its span as a success, because Spring AI hands the error message back to the model. And the advisor spans (`tool _calling `, `message_chat_memory`) have names Maple reads as extra tool and model calls. This guide fixes all four with a handful of properties and one configuration class. It covers Spring AI 2.0 (tested on 2.0.1) on Spring Boot 4.1 (4.1.1) and Java 21, with notes for Spring AI 1.1 on Boot 3.5. +Four defaults work against you. Spring Boot samples 10% of traces, so nine turns out of ten never arrive. Prompts and replies are never written to spans: `log-prompt` and `log-completion` send them to the application log. A tool that throws ends its span as a success, because Spring AI hands the error message back to the model. And the advisor spans (`tool _calling `, `message_chat_memory`) have names Maple reads as extra tool and model calls. This guide fixes all four with a handful of properties and one configuration class. + +This guide covers Spring AI 2.0 (tested on 2.0.1) on Spring Boot 4.1 (4.1.1) and Java 21, with notes for Spring AI 1.1 on Boot 3.5. ## Quick setup with a coding agent diff --git a/skills/maple-agent-tracing-langchain/SKILL.md b/skills/maple-agent-tracing-langchain/SKILL.md index 77e8b0fbb9..597c8a040d 100644 --- a/skills/maple-agent-tracing-langchain/SKILL.md +++ b/skills/maple-agent-tracing-langchain/SKILL.md @@ -11,7 +11,7 @@ Human guide with the reasoning: https://maple.dev/docs/agent-tracing/langchain Mechanism: `openinference-instrumentation-langchain` (scope `openinference.instrumentation.langchain`) with `TraceConfig(enable_genai_semconv=True)`, which dual-writes `gen_ai.*` (incl. `gen_ai.conversation.id` from run metadata `session_id` > `conversation_id` > `thread_id`, and `gen_ai.input/output.messages` in `{role, parts}` form). Maple classifies these as generic GenAI (framework facet "Unidentified") and reads `gen_ai.conversation.id` as the session key. This beats LangSmith's OTel export for Maple: readable transcript, interrupts not marked ERROR, no middleware noise spans, normal flush. Python only; LangChain.js/LangGraph.js are not covered (tell the user and stop). -## Step 0: detect +## Step 0: Detect 1. Versions: `python -c "import langchain, langgraph, langchain_core; print(langchain.__version__, langchain_core.__version__)"` and `pip show langgraph` (or read `pyproject.toml` / `uv.lock` / `requirements*.txt`). Tested: langchain 1.4.2, langgraph 1.2.12, langchain-core 1.6.5, langchain-openai 1.6.6, Python 3.12. Python must be >= 3.10. 2. Existing OTel setup. Search for `TracerProvider(`, `set_tracer_provider`, `logfire.configure`, `sentry_sdk.init`, `opentelemetry-instrument`, `Traceloop.init`, `LangChainInstrumentor`, `OpenAIInstrumentor`, `LANGSMITH_OTEL_ENABLED`, `LANGSMITH_TRACING_MODE`. @@ -21,7 +21,7 @@ Mechanism: `openinference-instrumentation-langchain` (scope `openinference.instr 3. Find: every `create_agent(`, `create_react_agent(`, `StateGraph(`/`.compile(`, every `.invoke(`/`.ainvoke(`/`.stream(`/`.astream(`/`Command(resume=` call on an agent/graph/chain, where the app's chat/thread id lives, every `ChatOpenAI(` (note `base_url`), and every tool that invokes another agent. 4. LangGraph Server (`langgraph.json` present): `import tracing` at the top of the module(s) `langgraph.json`'s `graphs` points to (the dir holding `tracing.py` must be in `dependencies`); `OTEL_*` env goes in the server env file/container. Server threads already carry `configurable.thread_id`: one Maple session per thread, no code. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -41,7 +41,7 @@ OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer If you pass `OTLPSpanExporter(endpoint=...)` in code, it must end in `/v1/traces` (used verbatim). -## Step 2: install + init +## Step 2: Install + init Add with the repo's package manager (uv/poetry/pip): @@ -99,7 +99,7 @@ LangChainInstrumentor().instrument( - Set a real `service.name` (never `unknown_service`). - Streaming with `ChatOpenAI(base_url=...)` or `OPENAI_BASE_URL` set: add `stream_usage=True` to the `ChatOpenAI(...)` constructor. ChatOpenAI only requests streamed usage from api.openai.com; servers that don't send it unasked (vLLM, many gateways) give streamed calls no tokens. OpenRouter sends it anyway; set it regardless. -## Step 3: session id (required) +## Step 3: Session id (required) Pass the app's conversation id as `thread_id` on EVERY agent/graph call, including streams and HITL resumes: @@ -120,13 +120,13 @@ agent.invoke(Command(resume={"decisions": [{"type": "approve"}]}), {"configurabl - The id must be stable per conversation and unique across conversations: no constants, no `uuid4()` per request. No id in the app → ask where the conversation boundary is; single-shot script → one uuid per conversation, reused. - Do not set `session.id`/`gen_ai.conversation.id`/`maple_ai.session.id` by hand. -## Step 4: content +## Step 4: Content - On by default: `gen_ai.input.messages` / `gen_ai.output.messages` (system prompt included as the first input message) on chat model spans; `gen_ai.tool.call.result` on tool spans. Maple's transcript needs these. - User wants content off → `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` (or `OPENINFERENCE_HIDE_INPUTS=true` / `OPENINFERENCE_HIDE_OUTPUTS=true`). The gen_ai copies are built from masked values, so they're empty too. Tell them the transcript will be empty; turns, tools, tokens and failures remain. - Narrower: `hide_input_text`, `hide_output_text`. Pattern redaction → OTel Collector `redaction` processor. -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents 1. Tool exceptions mark the tool span ERROR with the message automatically. `create_agent` re-raises them and aborts the run; if the app should continue, add (ask the user if behaviour changes matter): @@ -163,14 +163,14 @@ def ask_weather_worker(city: str) -> str: 4. Never put "agent" in a tool name (`ask_weather_agent`): OpenInference then marks the span AGENT, and it stops counting as a tool call. 5. Python 3.10 + async: pass the node's `config` to nested `ainvoke()` calls. Own thread pools: `from langchain_core.runnables.config import ContextThreadPoolExecutor`. -## Step 6: flush +## Step 6: Flush - `BatchSpanProcessor` exports every 5 s; the SDK flushes at normal interpreter exit. Long-running servers (incl. LangGraph Server) need nothing. - AWS Lambda / Cloud Functions / Cloud Run jobs: `provider.force_flush()` in a `finally` in the handler. - Scripts, CLIs, one-shot jobs: `provider.shutdown()` at the end (`finally`). - Celery/RQ workers, notebooks: `provider.force_flush()` after each task/cell that runs an agent. -## Step 7: verify +## Step 7: Verify Run one real conversation: 2+ messages with the same id, at least one tool call, one streamed message if the app streams, a sub-agent call if the app delegates. Then check (Maple → Agent Sessions, filter by service name; wait up to ~1 min): diff --git a/skills/maple-agent-tracing-litellm/SKILL.md b/skills/maple-agent-tracing-litellm/SKILL.md index 3bb5c0bfac..20dae5d591 100644 --- a/skills/maple-agent-tracing-litellm/SKILL.md +++ b/skills/maple-agent-tracing-litellm/SKILL.md @@ -14,7 +14,7 @@ Mechanism: - Use LiteLLM's **v2** OTel logger (`OpenTelemetryV2`). It writes `gen_ai.operation.name=chat`, provider, model, usage, `gen_ai.response.id`, TTFT, `error.type`, and `gen_ai.conversation.id` from `litellm_session_id`. Maple reads `gen_ai.conversation.id` as the session key for LiteLLM. - The default v1 logger (`litellm.callbacks=["otel"]` without v2) is wrong for Maple: op `acompletion`, no conversation id ever, and under an open parent span it writes onto the (ended) parent and the data is dropped. -## Step 0: detect +## Step 0: Detect 1. Versions: `python -c "import importlib.metadata as m; print(m.version('litellm'), m.version('opentelemetry-api'))"` (or read `pyproject.toml` / lock files). - LiteLLM 1.103.x (tested 1.103.0, latest stable on 2026-09-28): OpenTelemetry must be **<= 1.43.0**. On >= 1.44, `from litellm.integrations.otel.logger import OpenTelemetryV2` raises `ModuleNotFoundError: No module named 'opentelemetry._events'`; the `callbacks: ["otel"]` form (proxy) only logs `Error initializing custom logger` and exports nothing. Fixed in LiteLLM 1.104 (1.104.0rc1 verified with OTel 1.45); once 1.104 stable exists, use it and drop the pin. @@ -29,7 +29,7 @@ Mechanism: - An OpenAI/LiteLLM client instrumentor (OpenInference `LiteLLMInstrumentor`/`OpenAIInstrumentor`, OpenLLMetry, openai-v2) → duplicates LiteLLM's spans. Keep one; ask before removing. 5. Find: the agent loop (where `acompletion` results' `tool_calls` are executed), every tool function, where the chat/thread/conversation id lives, and any multi-agent orchestration. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -90,7 +90,7 @@ tracer = trace.get_tracer("support-agent") - If the project appends to `litellm.callbacks` elsewhere (other loggers), append the instance instead of overwriting the list. - Set a real `service.name`. -## Step 2b: proxy path (self-hosted LiteLLM Proxy) +## Step 2b: Proxy path (self-hosted LiteLLM Proxy) Proxy `config.yaml`: add `litellm_settings: { callbacks: ["otel"] }` (merge with existing callbacks). Proxy environment: @@ -129,11 +129,11 @@ async def call_model(conversation_id: str, messages: list, tools: list | None): - Never also instrument the app's OpenAI client when the proxy traces: double LLM calls and tokens. Pick gateway OR in-app. - Do not set `OTEL_IGNORE_CONTEXT_PROPAGATION=true` on the proxy. -## Step 2c: sync-only fallback (v1, only if Step 0.3 says so) +## Step 2c: Sync-only fallback (v1, only if Step 0.3 says so) `litellm.callbacks = ["otel"]` after `trace.set_tracer_provider(provider)` (v1 reuses the global SDK provider), plus env `USE_OTEL_LITELLM_REQUEST_SPAN=true` and `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental`. v1 has no conversation id: set `gen_ai.conversation.id` on each `invoke_agent` span. Consequence: framework shows **Unidentified** in Maple; tell the user. v1 content is on by default. -## Step 3: agent loop spans + session id (required) +## Step 3: Agent loop spans + session id (required) Wrap each agent run in `invoke_agent` and pass `litellm_session_id=` on EVERY `acompletion`: (imports: `asyncio`, `inspect`, `json`, `contextlib.contextmanager`, `opentelemetry.trace.StatusCode`, `tracer` from `tracing.py`): @@ -169,14 +169,14 @@ async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: - Do NOT put `gen_ai.conversation.id` or `maple_ai.session.id` on your own spans in the v2 setup: Maple labels the session with the vendor of the earliest session-bearing span, so the session would show **Unidentified**. LiteLLM's `chat` spans carrying the id is enough (Maple groups whole traces). - Streaming: `stream=True, stream_options={"include_usage": True}`, and consume the stream inside `agent_span`; LiteLLM closes its span when the stream ends. -## Step 4: content +## Step 4: Content - v2 default is `no_content`. `capture_message_content="span_only"` (or env `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only`) puts `gen_ai.input.messages` / `gen_ai.output.messages` JSON on `chat` spans. Maple's transcript needs them. - Never `event_only` / `span_and_event` for Maple: events are not read. - Messages are OpenAI chat format (`{role, content, tool_calls}`). Maple's transcript ignores `tool_calls` inside messages: a call that only requested tools shows an empty reply, and tool calls render only from `execute_tool` spans (Step 5). So the `execute_tool` spans are required for tool calls to appear at all. - User wants content off → `no_content` + drop the tool args/result attributes in `run_tool`. `litellm.turn_off_message_logging = True` keeps structure but replaces text with `redacted-by-litellm`. -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents ```py async def run_tool(agent: Agent, call) -> str: @@ -207,9 +207,9 @@ async def run_tool(agent: Agent, call) -> str: - Code-driven orchestration: run workers inside an outer `agent_span("orchestrator")` so the whole run is one trace; `asyncio.gather` keeps context. - Thread pools (`run_in_executor`, `ThreadPoolExecutor`) lose OTel context: wrap with `contextvars.copy_context().run`. -## Step 6: cost (optional) and flush +## Step 6: Cost (optional) and flush -Cost: Maple reads only `gen_ai.usage.cost`; LiteLLM writes `litellm.cost.total` (v2) / `hidden_params` (v1) → unpriced by default. If the user wants cost, report the TURN total on the OUTERMOST agent span only. Maple subtracts a descendant's reported cost from its nearest cost-reporting ancestor, so cost on every nested `invoke_agent` undercounts the orchestrator. Replace `agent_span`: +Cost: Maple reads `gen_ai.usage.cost` (also `gen_ai.usage.total_cost`, `llm.cost.total`), never `litellm.cost.total`; LiteLLM writes `litellm.cost.total` (v2) / `hidden_params` (v1) → unpriced by default. If the user wants cost, report the TURN total on the OUTERMOST agent span only. Maple subtracts a descendant's reported cost from its nearest cost-reporting ancestor, so cost on every nested `invoke_agent` undercounts the orchestrator. Replace `agent_span`: ```py from contextvars import ContextVar @@ -252,7 +252,7 @@ async def flush_tracing() -> None: Await it in a `finally` inside the event loop (end of `main()`, end of each handler invocation, FastAPI lifespan shutdown); scripts then call `provider.shutdown()` after `asyncio.run(...)`. -## Step 7: verify +## Step 7: Verify Run one real conversation: 2+ messages with the same id, one tool call, one streamed reply if the app streams, a failing tool if one exists, a second conversation with a different id, and a multi-agent run if the app has one. Check (Maple → Agent Sessions, filter by service; wait ~30 s): diff --git a/skills/maple-agent-tracing-llamaindex/SKILL.md b/skills/maple-agent-tracing-llamaindex/SKILL.md index c2ac7af438..22db34c38b 100644 --- a/skills/maple-agent-tracing-llamaindex/SKILL.md +++ b/skills/maple-agent-tracing-llamaindex/SKILL.md @@ -11,7 +11,7 @@ Human guide with the reasoning: https://maple.dev/docs/agent-tracing/llamaindex Mechanism: `openinference-instrumentation-llama-index` (scope `openinference.instrumentation.llama_index`) with `TraceConfig(enable_genai_semconv=True)`, which dual-writes `gen_ai.*` on span attributes. Maple reads `gen_ai.conversation.id` as the session key. The `LlamaIndexForMaple` processor below stamps `llamaindex.instrumentor` so Maple labels the framework LlamaIndex (without it: "Unidentified"). -## Step 0: detect +## Step 0: Detect 1. Versions: `python -c "import llama_index.core as c; print(c.__version__)"` (or `pyproject.toml` / `uv.lock` / `requirements*.txt`). - Need llama-index-core >= 0.14.19 (tested 0.14.25). Older: the instrumentor logs `DependencyConflict` and does nothing. Tell the user to upgrade. @@ -24,7 +24,7 @@ Mechanism: `openinference-instrumentation-llama-index` (scope `openinference.ins 4. Find: every `agent.run(` / `workflow.run(` / `AgentWorkflow(` call, where the chat/thread id lives in the request, every `FunctionAgent(`/`ReActAgent(`/`CodeActAgent(` construction, every tool that runs another agent, every `ctx.wait_for_event(` (HITL). 5. LLM class: `OpenAILike`/`OpenRouter` need `is_function_calling_model=True` or tools silently never run (no tool spans). Check it's set. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -45,7 +45,7 @@ OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf If you pass `OTLPSpanExporter(endpoint=...)` in code, it must end in `/v1/traces` (no auto-append). -## Step 2: install + init +## Step 2: Install + init Add with the repo's package manager (uv/poetry/pip): @@ -125,7 +125,7 @@ LlamaIndexInstrumentor().instrument( - The exporter MUST be added through `LlamaIndexForMaple`, never directly: without it every model call counts 2-3x in Maple (`_prepare_chat_with_tools` + nested same-name `astream_chat`/`achat` spans, all OpenInference kind LLM) and every tool call 3x (`call_tool` / `aggregate_tool_results` step spans are classified as tools by name). - No `service.name` default is acceptable: set `OTEL_SERVICE_NAME` (never `unknown_service`). -## Step 3: session id (required) +## Step 3: Session id (required) Nothing in LlamaIndex sets a conversation id; `Context` and `llamaindex.run_id` never reach Maple as one. Wrap every `agent.run(...)` / `workflow.run(...)` CALL in `using_session()`; add `instrument_tags` with the agent name in the same `with`: @@ -156,13 +156,13 @@ async def handle_message(conversation_id: str, text: str): - HITL: keep `handler.ctx.send_event(HumanResponseEvent(...))` on the same handler; the resumed step stays in the same trace and session. - Do not use `session.id`/`maple_ai.session.id` attributes of your own; `using_session` + `enable_genai_semconv` already writes `session.id` and `gen_ai.conversation.id`. -## Step 4: content +## Step 4: Content - On by default: `gen_ai.input.messages` (system + history + tool results), `gen_ai.output.messages` (incl. tool_call parts), tool results. Maple's transcript needs them. - User wants content off → `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` (or `hide_input_text`/`hide_output_text` to keep structure). Env equivalents `OPENINFERENCE_HIDE_*`. Tell them the transcript will be empty. - Pattern redaction (emails, cards): recommend an OTel Collector `redaction` processor. -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents 1. Give every agent a `name=` and wrap each agent's `run()` in `instrument_tags({"gen_ai.agent.name": agent.name})` (the processor copies it onto every span). No tag = no agent facet, no lanes. 2. Multi-agent via custom `Workflow`: run the whole workflow inside `using_session(...)`; inside steps call sub-agents like this: @@ -181,7 +181,7 @@ async def run_agent(agent: FunctionAgent, message: str) -> str: 6. HITL (`ctx.wait_for_event`): the first, suspended `FunctionTool.acall` ends ERROR `WaitingForEvent: ...`; the processor drops it. Nothing to add. 7. Known, unfixable here: `gen_ai.tool.call.arguments` = the tool's parameter schema (OpenInference GenAI mapping bug); no `gen_ai.tool.call.id` on tool spans. Real args are in the transcript's tool_call parts. -## Step 6: tokens, cost, streaming +## Step 6: Tokens, cost, streaming - Tokens come from the provider response on the model span. `FunctionAgent` streams by default; OpenAI-API streams only include usage when asked. For `OpenAI`/`OpenAILike` models not on OpenRouter add: @@ -193,14 +193,14 @@ llm = OpenAI(model="gpt-4o-mini", additional_kwargs={"stream_options": {"include - Streamed model spans end when the stream is handed back (~1 ms), so their duration is not model latency. If the app never streams tokens to users, `FunctionAgent(..., streaming=False)` gives real model-span durations. Ask before changing it. - Cost: never recorded. Sessions show "unpriced". Do not add pricing code. OpenRouter users can add OpenRouter Broadcast for cost. -## Step 7: flush +## Step 7: Flush - `BatchSpanProcessor` exports every 5 s; the provider flushes at normal interpreter exit. - Scripts/CLIs/one-shot jobs: `provider.shutdown()` in a `finally` at the end. - Serverless handlers, Celery/RQ tasks, notebooks: `provider.force_flush()` in a `finally` after each run (`from tracing import provider`). - FastAPI/long-running servers: nothing extra; optionally `provider.shutdown()` in the lifespan shutdown. -## Step 8: verify +## Step 8: Verify Run one real conversation: 2+ messages with the same id, at least one tool call, a failing tool if one exists, and a sub-agent run if the app delegates. Then check (Maple → Agent Sessions, filter by service name; wait ~30-60 s): diff --git a/skills/maple-agent-tracing-mastra/SKILL.md b/skills/maple-agent-tracing-mastra/SKILL.md index fa2f063aa5..019bf1bc11 100644 --- a/skills/maple-agent-tracing-mastra/SKILL.md +++ b/skills/maple-agent-tracing-mastra/SKILL.md @@ -13,7 +13,7 @@ Mechanism: Mastra's own tracing (`@mastra/observability`) converted to OTel GenA Mastra 1.71 has three export gaps that a small span processor (Step 2) fixes; it is required in every setup: the `chat` span has no input messages (empty prompt side of the transcript), sub-agents get their own thread id (turn split, wrong session), and step spans carry the raw provider HTTP response (headers with cookies, full reply body) as `mastra.metadata.headers` / `mastra.metadata.body`, which `hideOutput` does not hide. -## Step 0: detect +## Step 0: Detect 1. Versions: read `package.json` / lockfile for `@mastra/core`, `@mastra/observability`, `@mastra/otel-exporter`, `@mastra/memory`. - Need `@mastra/core` 1.x (written against 1.71) and Node >= 22.13. Mastra 0.x uses a different telemetry API: tell the user to upgrade; do not work around it. @@ -26,7 +26,7 @@ Mastra 1.71 has three export gaps that a small span processor (Step 2) fixes; it 4. Find: every `generate(` / `stream(` / `network(` call and where the chat/thread id lives in the request; every `new Agent(` (need `id` + `name`); every Agent with an `agents:` property (supervisor); every `createWorkflow` / `run.start(` / `createStep` that calls an agent; tools created with `createTool`. 5. Other tracers of the same model calls (OpenLLMetry `Traceloop.init`, OpenInference AI SDK instrumentation, Vercel AI SDK `experimental_telemetry` on the underlying model) → they double-trace. Keep Mastra's; ask before removing the others if they serve something else. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization: Bearer ` (passed as a headers object in code; no `%20` encoding). @@ -36,7 +36,7 @@ Mastra 1.71 has three export gaps that a small span processor (Step 2) fixes; it - Follow the repo's secret/env convention (`.env`, config module, secret manager) if it has one, e.g. `process.env.MAPLE_INGEST_KEY`. Otherwise inline is acceptable: ingest keys are write-only. - OtelExporter's `custom` provider reads NO env vars (`OTEL_EXPORTER_OTLP_*` are ignored). Endpoint, protocol and headers must be passed in code. -## Step 2: install + init +## Step 2: Install + init ```bash npm install @mastra/observability@latest @mastra/otel-exporter@latest @@ -120,7 +120,7 @@ export const mastra = new Mastra({ - Agents and workflows must be registered on this `Mastra` instance and called through it (`mastra.getAgent(...)`, `mastra.getWorkflow(...)`), or called with a `tracingContext` from a traced parent. An `Agent` used standalone has no observability. - Debugging delivery: `logLevel: "debug"` on OtelExporter prints `Export completed: N spans sent successfully` / `Export FAILED: ...`. Remove it afterwards. -## Step 3: session id (conversation id) +## Step 3: Session id (conversation id) Agents: pass the app's conversation id as the memory thread on EVERY call of a conversation: @@ -147,7 +147,7 @@ await run.start({ inputData, tracingOptions: { metadata: { threadId: conversatio HITL resumes (`approveToolCallGenerate` / `declineToolCallGenerate` / `approveToolCall` / `declineToolCall`): pass the same `memory` again. -## Step 4: content +## Step 4: Content - On by default: `gen_ai.output.messages` on `chat` spans, tool args/results on `execute_tool`. `gen_ai.input.messages` on `chat` spans (with the system prompt as the first message) only exists because of `mapleSpanProcessor`; without it the transcript has replies but no prompts. Tool calls inside earlier history are summarized by Mastra as `[tool: ]` text; the full calls are on the `execute_tool` spans. - `gen_ai.system_instructions` on `invoke_agent` is plain text, which Maple does not decode; the system prompt shows through the system message in the `chat` input instead. @@ -155,7 +155,7 @@ HITL resumes (`approveToolCallGenerate` / `declineToolCallGenerate` / `approveTo - `SensitiveDataFilter` is auto-applied (redacts values under keys like password/token/apiKey/authorization/secret). Do not disable it (`sensitiveDataFilter: false`) unless the user asks. - Serialization caps: 128 KiB/string, 50 items/array, 50 keys/object, depth 8. If agents keep > ~40 messages of history (`lastMessages` > 40 or custom history), add `serializationOptions: { maxArrayLength: 200 }` to the config, or the newest messages are cut from the transcript. -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents - Tools: `createTool({ id, description, inputSchema, execute })`. Span name `execute_tool `, `gen_ai.tool.name` = id. Give tools real ids. - Failures must throw (`throw new Error("...")`). Thrown → span status ERROR with the message as status message, `error.type=unknown`, plus an `exception` event. Returning `{ error }` = counted as success in Maple. If a tool swallows errors into a return value and the user wants failures visible, rethrow. @@ -165,7 +165,7 @@ HITL resumes (`approveToolCallGenerate` / `declineToolCallGenerate` / `approveTo - Workflow steps that call an agent: pass `tracingContext` from the step's `execute` args into `agent.generate(prompt, { tracingContext })`, or the agent starts a separate trace. -## Step 6: flush +## Step 6: Flush Batch interval is 5 s. Anything that can exit sooner must flush. @@ -173,7 +173,7 @@ Batch interval is 5 s. Anything that can exit sooner must flush. - Serverless / per-request handlers (Next.js route handlers, Vercel, Lambda, Workers): `await mastra.observability.flush()` in a `finally` at the end of each request; keep the instance. For streamed responses, flush after the stream completes (`after()`, `ctx.waitUntil()`, stream `onFinish`), not when the handler returns. - Long-running servers: nothing needed; optionally call `mastra.shutdown()` on SIGTERM. -## Step 7: verify +## Step 7: Verify Run one conversation: 2+ user messages with the same thread id (one streamed), at least one tool call, and a second conversation with a different thread id. If the project has a supervisor or workflow, run it once. Flush. Then check (Maple UI Agent Sessions, or debug log + Maple MCP `list_agent_sessions` filtered by service): diff --git a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md index a3de1a940d..adeed7afee 100644 --- a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md +++ b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md @@ -9,7 +9,7 @@ Goal: one user conversation = one Maple Agent Session, with a transcript, every The framework emits the spans itself. You add: an OTLP/HTTP exporter, content capture, a `gen_ai.conversation.id` span processor (the framework never sets one for local-history sessions), and a flush. -## Step 0: detect +## Step 0: Detect - Python MAF: `agent-framework`, `agent-framework-core` in `pyproject.toml` / `requirements*.txt` / `uv.lock`. Check the installed version (`python -c "import agent_framework; print(agent_framework.__version__)"`). Target ≥ 1.19.0; upgrade if older (1.13 lacks the `otlp_*` arguments used below). - .NET MAF: `Microsoft.Agents.AI` in `*.csproj`. Target ≥ 1.22.0. @@ -17,14 +17,14 @@ The framework emits the spans itself. You add: an OTLP/HTTP exporter, content ca - Existing OpenTelemetry: search for `TracerProvider(`, `set_tracer_provider`, `configure_azure_monitor`, `logfire.configure`, `configure_otel_providers`, `AddOpenTelemetry(`, `Sdk.CreateTracerProviderBuilder`. If a provider exists, add Maple's exporter and the conversation processor to it; do not create a second provider and do not call `configure_otel_providers()`. - Find every place a user message is handled (HTTP route, queue consumer, CLI loop) and what identifies the conversation there (chat id, thread id, `AgentSession`). You need it in Step 3. -## Step 1: key and region +## Step 1: Key and region - US endpoint `https://ingest.maple.dev`, EU `https://ingest.eu.maple.dev`. Header `Authorization=Bearer `. Protocol `http/protobuf`. - Key in the user's prompt: use it. No key: use the literal `MAPLE_TEST` (ingest accepts and discards it) and tell the user to replace it with their key from Settings → Ingestion. - Never put a private `maple_sk_` key in browser code. - Follow the repo's existing secret/env convention (`.env`, settings class, user-secrets). If there is none, inlining the ingest key is acceptable: ingest keys are write-only. -## Step 2: install and init +## Step 2: Install and init ### Python MAF @@ -132,7 +132,7 @@ sealed class ConversationIdProcessor : BaseProcessor - Workflows: `.WithOpenTelemetry()` on the `WorkflowBuilder` (source `Microsoft.Agents.AI.Workflows`, matched by the wildcard). - In a hosted app, put the same sources, processor and exporter in `builder.Services.AddOpenTelemetry().WithTracing(...)`. -## Step 3: conversation id +## Step 3: Conversation id Wrap every request/turn so all spans of that turn start inside `conversation()`: @@ -149,20 +149,20 @@ async def handle_message(session: AgentSession, text: str) -> str: - Build workflows (`WorkflowBuilder(...).build()`, `SequentialBuilder`, `ConcurrentBuilder`, ...) inside the `with`. `build()` always emits a separate one-span `workflow.build` trace: inside the `with` it joins the session; outside it becomes a stray one-span session. Executors (fan-out included) inherit the id. - .NET: `ConversationIdProcessor.Current.Value = chatId;` in the request handler before `RunAsync`/`RunStreamingAsync`. -## Step 4: content +## Step 4: Content - Python: `enable_sensitive_data=True` (or `ENABLE_SENSITIVE_DATA=true`). .NET: `EnableSensitiveData = true` (or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=true`). - If the process sets `OTEL_SEMCONV_STABILITY_OPT_IN`, it must include `gen_ai_latest_experimental` (e.g. `http,gen_ai_latest_experimental`); otherwise MAF moves content off the spans into log events Maple doesn't read. - `enable_message_events=False`: otherwise every message is also exported as OTLP log records (duplicate payload). - If the user wants no prompt content stored, leave sensitive data off and tell them the transcript and tool arguments/results will be empty. -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents - Give every `Agent` a distinct `name` (Maple lanes key on `gen_ai.agent.name`; unnamed agents get a UUID). - Tool failures: raising from the tool function is enough; MAF sets ERROR + `error.type` on `execute_tool`. Do not catch and return an error string from the tool body (that hides the failure). - Sub-agents: prefer `worker.as_tool()` in the orchestrator's `tools=[...]`, or MAF workflows/orchestrations. Do not also instrument the provider SDK (e.g. OpenInference/OpenLLMetry OpenAI instrumentors): that double-counts every call. -## Step 6: flush +## Step 6: Flush Scripts, CLIs, notebooks, tests, serverless: @@ -219,7 +219,7 @@ trace.set_tracer_provider(provider) - Flush with `provider.shutdown()` in `finally`. - .NET SK: same env vars (or `AppContext` switches `Microsoft.SemanticKernel.Experimental.GenAI.EnableOTelDiagnostics[Sensitive]`), `AddSource("Microsoft.SemanticKernel*")`, same `ConversationIdProcessor`. Shows as "Unidentified" framework in Maple. -## Step 7: verify +## Step 7: Verify Run one real conversation (2-3 turns, one tool call; a second conversation if cheap). With a real key, open Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`) after ~1 minute. With `MAPLE_TEST`, check the spans locally instead (add `ConsoleSpanExporter` temporarily, or point the endpoint at a local collector). Check: diff --git a/skills/maple-agent-tracing-openrouter/SKILL.md b/skills/maple-agent-tracing-openrouter/SKILL.md index 6816d1d2f0..1f16f57187 100644 --- a/skills/maple-agent-tracing-openrouter/SKILL.md +++ b/skills/maple-agent-tracing-openrouter/SKILL.md @@ -15,7 +15,7 @@ You cannot change the OpenRouter dashboard. Your job: (1) edit the app's OpenRou What Broadcast cannot give (tell the user, don't try to fix it here): tool spans, tool failures, agent names / sub-agent lanes. Those need the app's framework instrumentation (router skill `maple-agent-tracing`). Content arrives as wrapped JSON (`{"messages":[...]}`, `{"completion":...}`), which Maple shows as a raw JSON block, without turn labels. -## Step 0: detect +## Step 0: Detect 1. Find every OpenRouter call site: `openrouter.ai` base URLs (`https://openrouter.ai/api/v1`, `https://eu.openrouter.ai/api/v1`), `OPENROUTER_API_KEY`, `@openrouter/ai-sdk-provider`, `@openrouter/sdk`, the `openrouter` PyPI package, `openai` clients with an OpenRouter `baseURL` / `base_url`, LiteLLM `openrouter/` models, framework model classes pointed at OpenRouter. 2. Find where the conversation id lives per request: chat thread id, conversation id, session id, agent run id. It must be per conversation, never a process constant or a per-client default. @@ -24,7 +24,7 @@ What Broadcast cannot give (tell the user, don't try to fix it here): tool spans - No → Step 2 only (Broadcast-only). Mention that the app's framework skill adds tool spans and agent structure. 4. Which endpoint region the app calls (`openrouter.ai` vs `eu.openrouter.ai`), for the destination's data regions. -## Step 1: key and region (for the destination headers) +## Step 1: Key and region (for the destination headers) - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -33,7 +33,7 @@ What Broadcast cannot give (tell the user, don't try to fix it here): tool spans - Never put a private `maple_sk_` key in browser code. The key here goes into OpenRouter's dashboard, not the repo. - If the app also exports its own OTel to Maple, follow the repo's existing secret/env convention for that exporter. Ingest keys are write-only, so inline is acceptable if there is none. -## Step 2: send `session_id` on every OpenRouter request +## Step 2: Send `session_id` on every OpenRouter request Same id for every request of one conversation, new id per conversation, max 256 characters. If the app also has framework instrumentation, use the SAME value as the framework's conversation/session id (a trace with two different ids is assigned to the lexically larger one, silently). @@ -64,7 +64,7 @@ Other clients: find the framework's extra-body or per-request-headers option and Optional: `user` (<= 128 chars) is forwarded as `user.id`. Maple does not use it for sessions. Never put emails/names in `user`, `session_id` or `trace` metadata: Privacy Mode does not strip them. -## Step 3: nest Broadcast under the app's own traces (only if the app exports OTel to Maple) +## Step 3: Nest Broadcast under the app's own traces (only if the app exports OTel to Maple) Without this, every model call is recorded twice (app span + Broadcast trace) and Broadcast-only turns split one per model call. Put the active span's W3C ids in `trace.trace_id` (32 hex) and `trace.parent_span_id` (16 hex). @@ -109,21 +109,21 @@ This parents to the span current at the call site (turn/agent span), so the app' Don't set `trace_name` / `span_name` / `generation_name` unless the user asks: `span_name` creates an extra intermediate span. -## Step 4: content +## Step 4: Content Broadcast includes prompts and completions by default (`gen_ai.prompt`, `gen_ai.completion` on `LLM Generation`). The completion object has also been seen echoing the request body (tool definitions, `user`, `session_id`, `trace`). Nothing to change in code. If the user must keep content out of Maple: tell them to enable **Privacy Mode** on the destination (tokens, cost, timing and metadata still arrive). -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents - Broadcast has no tool spans and no agent names. Don't add fake tool spans. Point the user to their framework's skill for tools/lanes. - Provider fallbacks show as `provider attempt N: ` children; a failed attempt followed by a successful one is a retry and not counted as a failure. - A call where every provider failed: `LLM Generation` status Error, message `Provider returned error`, counted as `provider_error`. On this path OpenRouter drops the `trace` object, so it's its own trace (still in the right session via `session.id`). -## Step 6: flush +## Step 6: Flush Broadcast needs none: OpenRouter exports from its servers after each request. Delivery lag is about a minute. If the app exports its own spans, keep/ensure its normal shutdown flush (`sdk.shutdown()`, `provider.force_flush()`), otherwise nested Broadcast spans point at a missing parent. -## Step 7: hand the user the dashboard settings +## Step 7: Hand the user the dashboard settings Print these, filled in (you cannot apply them): @@ -138,7 +138,7 @@ Print these, filled in (you cannot apply them): - Leave **Additional generation metadata → Cost** off; Maple reads `gen_ai.usage.total_cost`, which is sent anyway. 3. Click **Test Connection**; it only saves if the test passes. The test creates a sessionless `openrouter-connection-test` trace (`trace:0000…0001` in Maple); ignore it. -## Step 8: verify +## Step 8: Verify Run one conversation (3+ turns, one with a tool call) with a fixed `session_id`, then a second conversation. Wait about a minute. In Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), filter service `openrouter` (plus the app's service if nested). Check: diff --git a/skills/maple-agent-tracing-opentelemetry/SKILL.md b/skills/maple-agent-tracing-opentelemetry/SKILL.md index bad407e072..62811c0970 100644 --- a/skills/maple-agent-tracing-opentelemetry/SKILL.md +++ b/skills/maple-agent-tracing-opentelemetry/SKILL.md @@ -11,7 +11,7 @@ Human guide with the reasoning: https://maple.dev/docs/agent-tracing/opentelemet Mechanism: you write the spans. Maple classifies a span only by `gen_ai.operation.name`, groups a trace by `gen_ai.conversation.id`, and reads content only from span attributes holding JSON strings. Hand-written spans show as framework "Unidentified" (vendor `unknown:genai`); that is expected. -## Step 0: detect +## Step 0: Detect 1. Language and entry points (web server, workers, scripts, serverless handlers). 2. Is a supported framework the real agent runtime? (`@mastra/core`, `ai`, `@openai/agents`/`openai-agents`, `langchain`/`langgraph`, `pydantic-ai`, `crewai`, `google-adk`, `llama-index`, `strands-agents`, `smolagents`, `agno`, `dspy`, `haystack-ai`, `agent-framework`, Spring AI, `litellm`, Claude Agent SDK). If yes, stop and use `maple-agent-tracing-` instead; use this skill only for the parts that framework doesn't cover, or for the `maple_ai.session.id` wrapper (Step 4). @@ -19,7 +19,7 @@ Mechanism: you write the spans. Maple classifies a span only by `gen_ai.operatio 4. Existing GenAI auto-instrumentation on the model client (`@opentelemetry/instrumentation-openai`, `opentelemetry-instrumentation-openai-v2`, OpenLLMetry `Traceloop.init`, OpenInference `OpenAIInstrumentor`, `logfire.instrument_openai`). Pick one source of `chat` spans: either keep that instrumentation (then see `maple-agent-tracing-provider-sdks`) or remove it and write `chat` spans here. Both = every model call twice. 5. Find in the code: the agent loop (where one user message is handled), every model call site, every tool dispatch, sub-agent calls, and where the conversation/chat/thread id lives in the request. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -36,7 +36,7 @@ OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf The exporters append `/v1/traces`. If an SDK rejects the space in the header, use `Bearer%20`. -## Step 2: install + init +## Step 2: Install + init Read the reference for the language and adapt it: - TypeScript/Node: `references/typescript.md` @@ -48,7 +48,7 @@ Rules: - Name the tracer after the app (e.g. `support-agent`). Never `openrouter`, `langsmith`, `litellm`, `haystack`, `ai`, `gen_ai`: Maple fingerprints frameworks by scope name and would read a different session key. - Keep the project's loop structure; add spans around its existing calls. Use the reference's complete loop only when there is no loop yet. -## Step 3: the three spans (exact keys) +## Step 3: The three spans (exact keys) `invoke_agent` (kind INTERNAL, name `invoke_agent `), around one agent run; for a user turn it is the trace root: - `gen_ai.operation.name`=`invoke_agent`, `gen_ai.agent.name`, `gen_ai.conversation.id` (Step 4) @@ -67,7 +67,7 @@ Message JSON (`input.messages`/`output.messages`): array of `{role, parts}`; par `provider.name` = the API actually called: `openai`, `anthropic`, `gcp.gemini`, `gcp.vertex_ai`, `aws.bedrock`, `azure.ai.openai`, `mistral_ai`, `groq`, `x_ai`, `deepseek`, or `openrouter` for OpenRouter. -## Step 4: session id (required) +## Step 4: Session id (required) - Set `gen_ai.conversation.id` on the turn's `invoke_agent` span from the app's conversation/chat/thread id. Same value for every message of a conversation; different across conversations. One classified span per trace is enough; every span in the trace joins. - Never: `uuid4()`/`randomUUID()` per request, the trace id, a module-level constant, a per-process default. No real id (single-shot script) → generate one per conversation, not per message, and reuse it. @@ -79,7 +79,7 @@ Escape hatch, only when a framework's spans carry a session key Maple ignores fo - Use the same value the framework would use for the session. - Not needed for hand-written spans: use `gen_ai.conversation.id`. -## Step 5: content +## Step 5: Content - Content = `gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages` (chat), `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result` (execute_tool). On span attributes only: span events, log records, and indexed keys (`gen_ai.prompt.0.content`, `llm.input_messages.0.*`) are not read. - Do not set `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` / `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT`; if the platform sets one, unset it. Truncated JSON is dropped whole. To cap size, drop the oldest messages whole before serializing. @@ -87,7 +87,7 @@ Escape hatch, only when a framework's spans carry a session key Maple ignores fo - User wants no content → skip those five attributes (everything else still works; transcript empty). Wants redaction → redact inside `toSemconv`/`to_semconv` before serializing, or an OTel Collector `redaction`/`transform` processor. - Never put API keys or `Authorization` headers in any attribute. -## Step 6: tools, errors, sub-agents +## Step 6: Tools, errors, sub-agents - Tool failure: set status ERROR with the error message as description, set `error.type` (exception class or error code), no `gen_ai.tool.call.result`, then return the error to the model as the tool result so the loop continues. Keep the message specific: Maple groups failures by it. - Tools that return `{"error": ...}` instead of raising: mark the span failed the same way when you detect it. @@ -96,7 +96,7 @@ Escape hatch, only when a framework's spans carry a session key Maple ignores fo - Sub-agent: call its loop inside the delegating tool's `execute_tool` span, so `execute_tool ask_x` → `invoke_agent x` → its `chat`/`execute_tool`. Distinct `gen_ai.agent.name` per agent (lanes need it). - Parallel tools: start each `execute_tool` span inside the turn's context (Node `Promise.all` keeps it; Python threads need `contextvars.copy_context().run`). -## Step 7: tokens and cost +## Step 7: Tokens and cost - Usage on `chat` spans only, never cumulative totals on `invoke_agent`. - Keys: `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` (ints). Not read: `total_tokens`, `reasoning_tokens`, `cache_read_input_tokens`. @@ -106,13 +106,13 @@ Escape hatch, only when a framework's spans carry a session key Maple ignores fo - Cost: `gen_ai.usage.cost` (double, USD) on `chat` spans. OpenRouter returns `usage.cost` → copy it. Other providers return none → compute only if the project has a price table; otherwise leave it (Maple shows "unpriced"; it never prices tokens). - `gen_ai.response.id` always (dedupes against gateway mirrors such as OpenRouter Broadcast). -## Step 8: flush +## Step 8: Flush - Node script/CLI: `await provider.shutdown()` in `finally`. Serverless: `await provider.forceFlush()` before returning (inside `waitUntil`/`after()` if available). - Python script: `provider.shutdown()` in `finally`. Lambda: `force_flush()` in `finally`. Notebooks/workers: `force_flush()` per cell/task. - Go: `defer tp.Shutdown(context.Background())` in `main`. -## Step 9: verify +## Step 9: Verify Run one real conversation: 2+ messages with the same id, one streamed reply if the app streams, one tool call, one failing tool if one exists, one sub-agent call if the app delegates; then a second conversation. Wait ~30 s; Maple → Agent Sessions, filter by service name. Check: diff --git a/skills/maple-agent-tracing-provider-sdks/SKILL.md b/skills/maple-agent-tracing-provider-sdks/SKILL.md index b95aba1f6c..8121733b8f 100644 --- a/skills/maple-agent-tracing-provider-sdks/SKILL.md +++ b/skills/maple-agent-tracing-provider-sdks/SKILL.md @@ -15,7 +15,7 @@ Mechanism: - Both: YOU add the `invoke_agent` span (with `gen_ai.conversation.id`) and `execute_tool` spans. Instrumentations can't see turns, conversations or your tools. - Maple files these spans under vendor `unknown:genai` (UI: "Unidentified") and reads `gen_ai.conversation.id` as the session key. Everything else is read in full. -## Step 0: detect +## Step 0: Detect 1. Confirm there is NO agent framework: `openai-agents`/`@openai/agents`, `langchain*`, `langgraph`, `pydantic-ai*`, `crewai`, `llama-index*`, `ai` (Vercel), `@mastra/core`, `google-adk`, `strands-agents`, `smolagents`, `agno`, `dspy`, `haystack-ai`, `agent-framework`, `litellm`. If one is present, stop and use that framework's skill (`npx skills add MapleTechLabs/maple/skills --skill maple-agent-tracing -y` routes). Instrumenting the provider SDK under a framework double-records every call. 2. Language and SDK versions: @@ -26,7 +26,7 @@ Mechanism: 4. Other instrumentations of the same SDK → duplicate model-call spans. Look for: `opentelemetry-instrumentation-openai` / `-anthropic` (OpenLLMetry, NOT the official ones), `opentelemetry-instrumentation-openai-v2` (deprecated), `openinference-instrumentation-*`, `@arizeai/openinference-*`, `@traceloop/*`, `logfire.instrument_openai/anthropic`, Sentry OpenAI/Anthropic integrations, `langfuse.openai`. Keep exactly one; ask before removing one that serves something else. 5. Find: every model call site, the agent loop(s), the tool dispatch, where the conversation/thread id lives per request, any agent that calls another agent (sub-agent), any streaming call. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -44,7 +44,7 @@ OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY ``` -## Step 2: install + init + helper +## Step 2: Install + init + helper Read the reference for the service's language and apply it exactly: @@ -56,7 +56,7 @@ Rules for both: - Set a real `service.name` (never `unknown_service`). - Do not set `OTEL_SEMCONV_STABILITY_OPT_IN`; the 1.x GenAI packages don't need it. -## Step 3: session id (required) +## Step 3: Session id (required) - Wrap each user turn (the whole model/tool loop) in `agent_span(, conversation_id)` / `agentSpan(...)`. It sets `gen_ai.operation.name=invoke_agent`, `gen_ai.agent.name`, `gen_ai.conversation.id`. - `conversation_id` = the app's stable conversation/thread/ticket id from the request. Same for every turn of one conversation, different across conversations. No `uuid4()` per request, no process-wide constant, no module-level default. @@ -65,14 +65,14 @@ Rules for both: - OpenAI Responses API with `conversation=`: the instrumentation also stamps `gen_ai.conversation.id=conv_...` on the model span. Pass that same id to `agent_span`. - Do not add `session.id` or `maple_ai.session.id`: Maple reads `gen_ai.conversation.id` for these spans. -## Step 4: content +## Step 4: Content - `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` puts `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.definitions` on span attributes. The helpers read the same variable for tool args/results (and, in TS, messages). - Values: `NO_CONTENT` (default, empty transcript), `SPAN_ONLY` (use), `EVENT_ONLY` (logs only, Maple shows nothing), `SPAN_AND_EVENT` (also works; duplicates into logs). Legacy `true` = events: wrong. - User wants content off → leave it unset in that environment; tell them the transcript will be empty but sessions, turns, tools, tokens and failures remain. - Redaction → in app code before the call, or an OTel Collector `redaction`/`transform` processor. Never lower `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` / `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` (truncated JSON is dropped). -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents 1. Route every tool execution through `run_tool(call_id, name, arguments_json, fn)` / `runTool(...)`, passing the model's tool call id. It catches the exception, marks the span failed (status ERROR + `error.type` + recorded exception) and returns `{"error": ...}` to the model. - If the existing loop already catches tool exceptions, move the catch into `run_tool` (or set status + `error.type` where it catches). A caught, unmarked failure shows as success. @@ -81,19 +81,19 @@ Rules for both: 2. Sub-agent (an agent run inside a tool): call it via `run_tool`, and inside wrap its loop in `agent_span("")` WITHOUT a conversation id (it's in the caller's trace). Every agent gets a unique `gen_ai.agent.name`; same names merge lanes. 3. Parallel tools: `asyncio.gather` / `Promise.all` keep context. `ThreadPoolExecutor`: submit `contextvars.copy_context().run(fn, ...)`, or tool spans become orphan traces. -## Step 6: tokens, streaming, cost +## Step 6: Tokens, streaming, cost - OpenAI Chat Completions streaming: ALWAYS pass `stream_options={"include_usage": True}` (Python) or use the helper's `onText` path (TS sets it). Otherwise the streamed call has 0 tokens. The final usage chunk has empty `choices`; skip it when reading text (`if chunk.choices`). - Anthropic / Gemini streams include usage; nothing to add. - No Python instrumentation emits cost; Maple doesn't price tokens → sessions show "unpriced". The TS helper copies OpenRouter's `usage.cost` (USD) to `gen_ai.usage.cost`; direct OpenAI sends none. Don't add pricing tables. - Claude/Gemini models via OpenRouter's OpenAI-compatible endpoint with the `openai` SDK → spans say `gen_ai.provider.name=openai` with OpenAI-shaped usage. Correct; don't override it. -## Step 7: flush +## Step 7: Flush - Python: the provider flushes at normal interpreter exit. Add `provider.force_flush()` after each turn in Lambda/Cloud Functions/Cloud Run jobs, notebooks, Celery/RQ tasks, and anything ending in `os._exit`. Scripts/CLIs: `provider.shutdown()` in `finally`. - TypeScript: `await sdk.shutdown()` before a CLI/script exits. Serverless: `await spanProcessor.forceFlush()` before returning (or in `waitUntil` after a streamed response). -## Step 8: verify +## Step 8: Verify Run one real conversation: 3+ user messages with the same id, one tool call, one streamed message if the app streams, one failing tool if you can trigger it, one sub-agent call if the app delegates; then a second conversation with a different id. Flush. In Maple → Agent Sessions (filter by service; allow ~30 s): diff --git a/skills/maple-agent-tracing-pydantic-ai/SKILL.md b/skills/maple-agent-tracing-pydantic-ai/SKILL.md index c4cef2ae53..7b0bd73a7a 100644 --- a/skills/maple-agent-tracing-pydantic-ai/SKILL.md +++ b/skills/maple-agent-tracing-pydantic-ai/SKILL.md @@ -11,7 +11,7 @@ Human guide with the reasoning: https://maple.dev/docs/agent-tracing/pydantic-ai Mechanism: Pydantic AI's native OTel instrumentation (scope `pydantic-ai`, GenAI semconv on span attributes). No extra instrumentation package. Maple reads `gen_ai.conversation.id` as the session key for this framework. -## Step 0: detect +## Step 0: Detect 1. Pydantic AI version: `python -c "import pydantic_ai; print(pydantic_ai.__version__)"` (or read `pyproject.toml` / `uv.lock` / `requirements*.txt`). - Need 2.x (tested 2.51.0). 1.x has no `conversation_id=`: tell the user to upgrade; do not work around it. @@ -23,7 +23,7 @@ Mechanism: Pydantic AI's native OTel instrumentation (scope `pydantic-ai`, GenAI 3. Find: every `agent.run(` / `run_sync(` / `run_stream(` / `iter(` call, where the chat/thread id lives in the request, every `Agent(` construction, and every tool that calls another agent's `run()`. 4. Other instrumentors on the same model client (`logfire.instrument_openai`, `OpenAIInstrumentor`, OpenLLMetry `Traceloop.init`) → they double-trace model calls. Keep Pydantic AI's; ask before removing the others if they serve something else. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -40,7 +40,7 @@ OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf ``` -## Step 2a: install + init (plain OpenTelemetry, default) +## Step 2a: Install + init (plain OpenTelemetry, default) Add with the repo's package manager (uv/poetry/pip): @@ -111,7 +111,7 @@ logfire.instrument_pydantic_ai() - Default scrubbing replaces tool args/results containing `session`, `auth`, `password`, `cookie`, `secret`, `api key`... with `[Scrubbed due to '']`. `gen_ai.conversation.id` and message attributes are exempt. If the project already has a scrubbing config, merge the callback into it instead of replacing it. - Do not also call `Agent.instrument_all(...)` with another provider. -## Step 3: session id (required) +## Step 3: Session id (required) Pydantic AI resolves `gen_ai.conversation.id` per run as: explicit `conversation_id=` > id on the last message of `message_history` > fresh UUID7. Pass it explicitly on EVERY run call, from the app's chat/thread/conversation id: @@ -130,14 +130,14 @@ async with agent.run_stream(text, conversation_id=chat_id, message_history=histo - No id available in the app → ask the user where the conversation boundary is; if it's a single-shot script, generate one uuid per conversation (not per run) and reuse it. - Streaming: keep the `async with` open until the stream is consumed (in FastAPI, inside the generator passed to `StreamingResponse`), or the `invoke_agent` span ends early. -## Step 4: content +## Step 4: Content - Content is on by default (`include_content=True`): `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions` on `chat` spans; tool args/results on `execute_tool` spans. Maple's transcript needs these. - Always set `include_binary_content=False` (base64 media repeats in every later `chat` span). - User wants content off → `include_content=False` globally, or per agent: `Agent(..., capabilities=[Instrumentation(settings=InstrumentationSettings(include_content=False))])` (`from pydantic_ai.capabilities import Instrumentation`; omit `tracer_provider` to use the global one). Tell them the transcript keeps roles but no text. - Logfire scrubbing does not redact message content. For pattern redaction of prompts, recommend an OTel Collector `redaction`/`transform` processor. -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents 1. Name every agent: `Agent(..., name="support")`. Unnamed agents become `agent` and share one lane. 2. Tool failures the model should see: `raise ToolFailed("message")` (`from pydantic_ai import ToolFailed`, >= 2.16). Span ERROR, message recorded as `gen_ai.tool.call.result`, run continues, no retry budget used. @@ -161,7 +161,7 @@ async def research_weather(ctx: RunContext[None], city: str) -> str: Without `conversation_id=ctx.conversation_id` each delegate mints a UUID7, so one trace carries several ids and Maple may file the trace under the wrong session or split the turn. 4. Sequential pipelines of top-level runs (orchestrator run, then summary run): pass the same `conversation_id=` to each; each run is its own trace/turn in the same session. -## Step 6: flush +## Step 6: Flush - The SDK flushes at normal interpreter exit. Add an explicit flush where that doesn't happen: - AWS Lambda / Cloud Functions / Cloud Run jobs: `trace.get_tracer_provider().force_flush()` in a `finally` in the handler. @@ -169,7 +169,7 @@ async def research_weather(ctx: RunContext[None], city: str) -> str: - Celery/RQ/multiprocessing workers, notebooks: `force_flush()` after each task/cell that runs an agent. - Logfire: `logfire.force_flush()` / `logfire.shutdown()`. -## Step 7: verify +## Step 7: Verify Run one real conversation: 2+ messages with the same id, at least one tool call, one streamed message if the app streams, and a sub-agent call if the app delegates. Then check (Maple → Agent Sessions, filter by the service name; wait ~30 s): diff --git a/skills/maple-agent-tracing-strands/SKILL.md b/skills/maple-agent-tracing-strands/SKILL.md index 1d541f7d65..3ca84360fb 100644 --- a/skills/maple-agent-tracing-strands/SKILL.md +++ b/skills/maple-agent-tracing-strands/SKILL.md @@ -11,7 +11,7 @@ Human guide with the reasoning: https://maple.dev/docs/agent-tracing/strands Mechanism: Strands' native OTel tracer (scope `strands.telemetry.tracer`, `gen_ai.provider.name=strands-agents`). No extra instrumentation package. Maple reads `session.id` as the session key for Python Strands, and reads span ATTRIBUTES only (never span events). -## Step 0: detect +## Step 0: Detect 1. Language and version. - Python: `python -c "from importlib.metadata import version; print(version('strands-agents'))"` or read `pyproject.toml` / `uv.lock` / `requirements*.txt`. Need >= 1.54 (tested 1.57.1): span-attribute content needs 1.48, tool args/results 1.51, Maple-readable cache token names 1.54. Older → upgrade; do not work around it. @@ -23,7 +23,7 @@ Mechanism: Strands' native OTel tracer (scope `strands.telemetry.tracer`, `gen_a 3. Find: every `Agent(` construction (and whether it is module-level/shared), every place the agent is invoked, where the chat/thread/conversation id lives in the request, every `session_manager=`, every `.as_tool(`, `Swarm(`, `GraphBuilder(` / `Graph(`. 4. Other instrumentors on the same model calls (OpenLIT, OpenLLMetry `Traceloop.init`, OpenInference, `opentelemetry-instrumentation-openai*`, botocore/Bedrock GenAI instrumentation) → they double-trace model calls. Keep Strands' spans; ask before removing the others if they serve something else. -## Step 1: key and region +## Step 1: Key and region - US: `https://ingest.maple.dev`. EU: `https://ingest.eu.maple.dev`. - Header: `Authorization=Bearer `. @@ -32,7 +32,7 @@ Mechanism: Strands' native OTel tracer (scope `strands.telemetry.tracer`, `gen_a - Never put a private `maple_sk_` key in browser code. - Follow the repo's secret/env convention (`.env`, settings module, secret manager, container env) if it has one. Otherwise inline is acceptable: ingest keys are write-only. -## Step 2: install + init +## Step 2: Install + init Python. Add with the repo's package manager, keeping existing extras (`openai`, `anthropic`, `litellm`, ...): @@ -80,7 +80,7 @@ import { setupTracer } from "@strands-agents/sdk/telemetry" export const provider = setupTracer({ exporters: { otlp: true } }) // before the first Agent; reads OTEL_EXPORTER_OTLP_* ``` -## Step 3: session id +## Step 3: Session id Python: pass the conversation id as `session.id` in `trace_attributes` on the agent that handles the turn. `gen_ai.conversation.id` is ignored for Python Strands; don't rely on it. @@ -123,13 +123,13 @@ const agent = new Agent({ - TS: construct the agent PER REQUEST (restore history with the repo's `SessionManager` storage). The TS `invoke_agent` span always carries the agent instance's accumulated usage (no `gen_ai_use_latest_invocation_tokens`), so a reused agent re-reports every earlier turn and Maple's totals inflate. - TS stamps `traceAttributes` on `invoke_agent` only (not chat/tool spans). That is enough. -## Step 4: content +## Step 4: Content - Content capture is ON by default; Step 2's tokens only move it onto attributes. Nothing else to enable. - Redaction, only if the user asks or the repo handles regulated data: append `gen_ai_unredacted_attributes=` to `OTEL_SEMCONV_STABILITY_OPT_IN`. `;`-separated, single trailing `*` only. Covered keys: `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result`. Empty list (`gen_ai_unredacted_attributes=`) redacts all to `[REDACTED]`. Example keeping replies only: `...,gen_ai_unredacted_attributes=gen_ai.output.*;gen_ai.tool.call.result`. - Not redactable: `gen_ai.tool.description`, `gen_ai.tool.json_schema`, `trace_attributes` values. -## Step 5: tools, errors, sub-agents +## Step 5: Tools, errors, sub-agents - Nothing to add for tool failures: a raising `@tool` (or one returning `{"status": "error"}`) gets span status ERROR with the exception message and `gen_ai.tool.status=error`. Do not catch exceptions inside tools just to return friendly text; that hides the failure unless you return `status: "error"`. - Sub-agents: `sub.as_tool(description=...)` nests `invoke_agent ` under `execute_tool `; Maple shows a delegation lane. Give each sub-agent a distinct `name`. @@ -137,7 +137,7 @@ const agent = new Agent({ - Interrupt/resume (HITL): the resume is a new trace in the same session (same `trace_attributes`). Interrupted tool spans appear twice with the same `gen_ai.tool.call.id` (first ends OK with no result, second has the real outcome); Maple counts both as tool calls. Expected, framework-level; do not try to filter spans. - TS: failed Graph nodes end with status OK (upstream bug harness-sdk#4166). Tool failures are fine. -## Step 6: flush +## Step 6: Flush - Long-running server: nothing. - Script / CLI / job / notebook / test: in `finally`: @@ -151,7 +151,7 @@ telemetry.tracer_provider.shutdown() - Existing provider (Step 0): flush that provider instead. - TS: `await provider.forceFlush(); await provider.shutdown()` before exit. `setupTracer`'s own `beforeExit` flush does not run after `process.exit()`. -## Step 7: verify +## Step 7: Verify Run one short conversation (2-3 messages, one tool call; plus a failing tool if easy) with the real key, flush, then check in Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), or via the Maple MCP (`list_agent_sessions`, `get_agent_session`): diff --git a/skills/maple-agent-tracing/SKILL.md b/skills/maple-agent-tracing/SKILL.md index 2847c70b9f..dc38243870 100644 --- a/skills/maple-agent-tracing/SKILL.md +++ b/skills/maple-agent-tracing/SKILL.md @@ -25,7 +25,8 @@ Then read the installed `SKILL.md` and follow it. If `npx skills` is unavailable | --- | --- | | `@mastra/core` | `maple-agent-tracing-mastra` | | `@openai/agents`, `openai-agents` | `maple-agent-tracing-openai-agents` | -| `@langchain/langgraph`, `langchain`, `@langchain/core`, `langgraph`, `langchain-core` | `maple-agent-tracing-langchain` | +| `langchain`, `langgraph`, `langchain-core` (Python) | `maple-agent-tracing-langchain` | +| LangChain.js / LangGraph.js (`@langchain/core`, `@langchain/langgraph`) | None yet: tell the user it isn't covered. Offer `maple-agent-tracing-opentelemetry` only if they want hand-written spans. | | `@anthropic-ai/claude-agent-sdk`, `claude-agent-sdk`, or the `claude` CLI itself | `maple-agent-tracing-claude-agent-sdk` | | `ai` (Vercel AI SDK) | `maple-agent-tracing-vercel-ai-sdk` | | `pydantic-ai`, `pydantic-ai-slim` | `maple-agent-tracing-pydantic-ai` | @@ -37,14 +38,14 @@ Then read the installed `SKILL.md` and follow it. If `npx skills` is unavailable | `agno` | `maple-agent-tracing-agno` | | `dspy` | `maple-agent-tracing-dspy` | | `haystack-ai` | `maple-agent-tracing-haystack` | -| `agent-framework`, `Microsoft.Agents.AI`, `semantic-kernel`, `Microsoft.SemanticKernel` | `maple-agent-tracing-microsoft-agent-framework` | +| `agent-framework`, `agent-framework-core`, `Microsoft.Agents.AI`, `semantic-kernel`, `Microsoft.SemanticKernel` | `maple-agent-tracing-microsoft-agent-framework` | | `spring-ai-*` (Maven/Gradle) | `maple-agent-tracing-spring-ai` | | `litellm` (SDK or proxy) | `maple-agent-tracing-litellm` | | Requests go through OpenRouter (`openrouter.ai` base URL) and the user wants gateway-side traces | `maple-agent-tracing-openrouter` | | Only a provider SDK: `openai`, `@anthropic-ai/sdk`, `anthropic`, `google-genai`, `@google/genai` | `maple-agent-tracing-provider-sdks` | | Anything else, a hand-rolled agent loop, or another language | `maple-agent-tracing-opentelemetry` | -Order matters: a framework row wins over the provider-SDK row, because frameworks depend on provider SDKs and instrumenting both records every model call twice. Mastra and LangChain JS depend on `ai` or `openai` too; match the framework first. +Order matters: a framework row wins over the provider-SDK row, because frameworks depend on provider SDKs and instrumenting both records every model call twice. Mastra depends on `ai` and LangChain.js on `openai` too; match the framework first, and don't route a LangChain.js service to the provider-SDK skill. ## Step 3: Hand-off diff --git a/skills/maple-onboard/SKILL.md b/skills/maple-onboard/SKILL.md index 204b3f0220..2e145a5772 100644 --- a/skills/maple-onboard/SKILL.md +++ b/skills/maple-onboard/SKILL.md @@ -131,6 +131,8 @@ Get the meter once at module level, create instruments at module level, incremen If the project calls OpenAI / Anthropic / Google / any LLM provider, follow `maple-onboarding-style` "LLM calls": provider instrumentation (OpenInference) where it exists, `maple_ai.session.id` on each turn so Agent Sessions can group a conversation, and `gen_ai.usage.cost` only when the provider reports a billed cost. Maple does not price tokens. +If the service runs an agent framework (Vercel AI SDK, OpenAI Agents SDK, LangChain/LangGraph, Mastra, Pydantic AI, CrewAI, Google ADK and others) or routes calls through OpenRouter or LiteLLM, use the `maple-agent-tracing` skill for that service instead: it installs a per-framework skill with the switches each framework needs for sessions, transcripts, tool failures and token counts. + ## Step 4: Verify the app still works and telemetry arrives Per service: From 0d342ccce4adbe810e10f631d993c61aae001881 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Mon, 28 Sep 2026 23:59:31 +0200 Subject: [PATCH 04/39] docs(agent-tracing): align guides with what Agent Sessions shows for each framework --- apps/landing/src/content/docs/agent-tracing/agno.md | 4 +++- .../src/content/docs/agent-tracing/claude-agent-sdk.md | 2 ++ apps/landing/src/content/docs/agent-tracing/crewai.md | 3 ++- apps/landing/src/content/docs/agent-tracing/dspy.md | 5 ++++- .../src/content/docs/agent-tracing/google-adk.md | 5 +++-- .../landing/src/content/docs/agent-tracing/haystack.md | 5 ++++- .../src/content/docs/agent-tracing/langchain.md | 8 +++++--- apps/landing/src/content/docs/agent-tracing/litellm.md | 4 ++-- .../src/content/docs/agent-tracing/llamaindex.md | 2 +- apps/landing/src/content/docs/agent-tracing/mastra.md | 6 ++++-- .../docs/agent-tracing/microsoft-agent-framework.md | 4 ++-- .../src/content/docs/agent-tracing/openai-agents.md | 6 ++++-- .../src/content/docs/agent-tracing/openrouter.md | 10 ++++++---- .../src/content/docs/agent-tracing/provider-sdks.md | 2 +- .../src/content/docs/agent-tracing/pydantic-ai.md | 2 +- .../src/content/docs/agent-tracing/smolagents.md | 6 +++--- apps/landing/src/content/docs/agent-tracing/strands.md | 8 +++++--- .../src/content/docs/agent-tracing/vercel-ai-sdk.md | 7 +++++-- skills/maple-agent-tracing-agno/SKILL.md | 2 +- skills/maple-agent-tracing-claude-agent-sdk/SKILL.md | 2 +- skills/maple-agent-tracing-crewai/SKILL.md | 1 + skills/maple-agent-tracing-dspy/SKILL.md | 5 +++-- skills/maple-agent-tracing-google-adk/SKILL.md | 2 +- skills/maple-agent-tracing-haystack/SKILL.md | 4 +++- skills/maple-agent-tracing-langchain/SKILL.md | 4 ++-- skills/maple-agent-tracing-litellm/SKILL.md | 4 ++-- skills/maple-agent-tracing-llamaindex/SKILL.md | 2 +- skills/maple-agent-tracing-mastra/SKILL.md | 3 ++- .../SKILL.md | 2 +- skills/maple-agent-tracing-openai-agents/SKILL.md | 6 +++--- skills/maple-agent-tracing-openrouter/SKILL.md | 4 ++-- skills/maple-agent-tracing-provider-sdks/SKILL.md | 2 +- skills/maple-agent-tracing-smolagents/SKILL.md | 6 +++--- skills/maple-agent-tracing-strands/SKILL.md | 3 ++- skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md | 4 ++-- 35 files changed, 88 insertions(+), 57 deletions(-) diff --git a/apps/landing/src/content/docs/agent-tracing/agno.md b/apps/landing/src/content/docs/agent-tracing/agno.md index de3de5483b..4000ddb2ac 100644 --- a/apps/landing/src/content/docs/agent-tracing/agno.md +++ b/apps/landing/src/content/docs/agent-tracing/agno.md @@ -252,6 +252,8 @@ Reasoning tokens are not broken out into their own attribute, so Maple can't sho Cost appears when the model provider returns a price with the response. OpenRouter does, and the instrumentor records it as `llm.cost.total` in USD. Maple never prices tokens itself, so calls to providers that don't return a cost, such as OpenAI or Anthropic called directly, show as unpriced. +For now, only the **Agent Sessions** list shows that cost. The session's own page doesn't read `llm.cost.total` yet and shows the cost as not reported. Tokens, models and call counts match on both. + ## Flush before short-lived processes exit `BatchSpanProcessor` exports every 5 seconds. A long-running server needs nothing extra: the SDK flushes on normal shutdown. Scripts, CLIs, notebooks, queue workers and serverless handlers need an explicit flush, or the last spans are lost: @@ -278,7 +280,7 @@ Run one conversation of two or three turns with the same `session_id`, including - **Model calls** named after the Agno model class (`OpenRouter.invoke`, `OpenAIChat.ainvoke`, `Claude.invoke_stream`), with the model id (`openai/gpt-4o-mini`) as the model. - **Tool calls** named after your functions, with arguments and results, and failed ones marked. - **Agents** named after your `Agent(name=...)` and `Team(name=...)`, with a lane per team member. -- **Tokens** on every model call, and a cost if your provider returns one. +- **Tokens** on every model call, and a cost in the sessions list if your provider returns one. The session page shows cost as not reported for Agno. A human-in-the-loop approval shows up as two turns in the same session: the run that paused (`support_agent.run`) and the resumed run (`support_agent.continue_run`), each in its own trace. diff --git a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md index c767d41872..615ff64bfd 100644 --- a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md @@ -293,6 +293,8 @@ What you don't get is a lane per sub-agent. Maple opens a lane for each distinct Each `claude_code.llm_request` span carries `input_tokens`, `output_tokens`, `cache_read_tokens` and `cache_creation_tokens`, plus `ttft_ms` and the stop reason. Maple maps them to its token buckets and time to first token. Anthropic's `input_tokens` excludes both cache buckets, and Maple counts it that way, so total input is the sum of the three. The CLI always streams and still records usage, so there is no streaming gap to work around. +Claude Code reports the cache buckets on every call, even when they're zero. If your prompts are shorter than the model's minimum cacheable length (a short custom `systemPrompt` often is), nothing is cached and the session's **Prompt cache** check warns with a 0% hit rate. That warning is accurate, not a tracing problem. + Maple doesn't show cost for these sessions. Maple never prices tokens itself, and Claude Code puts cost only on the `api_request` log event (`cost_usd`) and the `claude_code.cost.usage` metric, never on a span. Sessions read as **unpriced**. To track spend anyway, keep the logs and metrics exporters on. The `claude_code.api_request` records are searchable under **Logs** with their `cost_usd` and `session.id`, and `claude_code.cost.usage` can go on a dashboard. Both are Claude Code's client-side estimate at list price, unless your organization sets `modelPricing` in managed settings. In an SDK app, the result message's `total_cost_usd` is the same estimate for one `query()`. diff --git a/apps/landing/src/content/docs/agent-tracing/crewai.md b/apps/landing/src/content/docs/agent-tracing/crewai.md index eebf6f6481..ad7792d0e4 100644 --- a/apps/landing/src/content/docs/agent-tracing/crewai.md +++ b/apps/landing/src/content/docs/agent-tracing/crewai.md @@ -270,7 +270,8 @@ Call `provider.shutdown()` instead when the process is about to exit and won't t Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: - **One session** for the conversation, with one turn per `kickoff()`. Each turn's trace starts at `support.kickoff` (your crew's name), or `support_flow.kickoff` for a flow named `support_flow`. -- **The transcript**: the system message built from the agent's role, goal and backstory, `Current Task: …` with your message, and the model's replies. +- **Framework: CrewAI** in the session list. +- **The transcript**: the system message built from the agent's role, goal and backstory, `Current Task: …` with your message, and the model's replies. Turns are labeled `Current Task: `. - **Model calls** named `ChatCompletion` (from the OpenAI instrumentor), each with a model and input and output tokens. - **Tool calls** named `get_weather.run` and `calculate.run`, with results. - **Agents**: one lane per role, from `assistant.reply._execute_core` and its siblings. diff --git a/apps/landing/src/content/docs/agent-tracing/dspy.md b/apps/landing/src/content/docs/agent-tracing/dspy.md index 1528db99cb..41782a0985 100644 --- a/apps/landing/src/content/docs/agent-tracing/dspy.md +++ b/apps/landing/src/content/docs/agent-tracing/dspy.md @@ -314,12 +314,15 @@ Run one conversation of two or three messages through `handle_message` with the - **One session** for the conversation, framework **DSPy**, with one turn per call to your module. Each turn's trace starts at `ChatAssistant.forward` (your module's class name), titled with the question you asked. - **Model calls** named `LM.__call__`, each with a model, input and output tokens. Every model call sits under `Predict.forward`, `Predict(StringSignature).forward` and `ChatAdapter.__call__` spans; those are DSPy's steps, not extra calls. - **The transcript**: the messages DSPy sent, in its `[[ ## field ## ]]` format, and the replies. -- **Tool calls** named `get_weather.__call__` and `finish.__call__`, with arguments and results. +- **Tool calls** named `get_weather.__call__` and `finish.__call__`, with arguments and results. `finish` is ReAct's end-of-loop tool and counts as a tool call, so a turn that used one tool shows two. - **Agents**: one per module class you wrote, with a lane for each worker module. - **Cost** per call where DSPy has a price for the model. +- **Checks**: a tool that raised fails the session's **Tool availability** check, with the exception as its headline. A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. +Expect a **Prompt cache** warning on a short test conversation. OpenRouter reports cached input tokens even when they're zero, so Maple judges the cache hit rate, and providers only cache long prompts (OpenAI from 1,024 tokens; Anthropic models only with explicit cache markers). In our test the prompts peaked at 870 tokens and none were cached. + ## Troubleshooting - **No spans at all.** `instrument()` never ran, or the exporter can't reach Maple. Import `tracing` first in the entry point and look for an `OTLPSpanExporter` error in the logs. diff --git a/apps/landing/src/content/docs/agent-tracing/google-adk.md b/apps/landing/src/content/docs/agent-tracing/google-adk.md index a68c3338d3..fc3a4b5558 100644 --- a/apps/landing/src/content/docs/agent-tracing/google-adk.md +++ b/apps/landing/src/content/docs/agent-tracing/google-adk.md @@ -222,7 +222,8 @@ Avoid wrapping agents in `AgentTool`. It runs the sub-agent in a new in-memory s `generate_content` spans carry `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.reasoning.output_tokens` when the model reports them. ADK counts cached tokens inside the input and thinking tokens inside the output, which is how Maple adds them up. - Streamed turns (`RunConfig(streaming_mode=StreamingMode.SSE)`) report usage too. `LiteLlm` requests it with `stream_options.include_usage`. -- The `call_llm` span above each `generate_content` repeats the same usage. Maple nets a parent's usage against its children, so each call is counted once. +- The `call_llm` span above each `generate_content` repeats the same usage. Maple nets a parent's usage against its children, so tokens and the LLM call count cover each call once. +- The session checks don't net yet. Headlines such as "All 16 model calls were answered first time" count the `call_llm` and `generate_content` spans of each call separately, so they show twice the LLM call count. - ADK doesn't record cost, and Maple doesn't price tokens, so ADK sessions show as **unpriced**. LiteLLM computes a cost, but it never reaches ADK's spans. ## Flush spans before a short-lived process exits @@ -255,7 +256,7 @@ In a long-running server, call `provider.shutdown()` from your shutdown hook (Fa Run one conversation of two or three turns, one of which calls a tool, then open **Agent Sessions** in Maple. Spans leave the process within 5 seconds and usually show up within a minute. You should see: -- **One session per ADK session id**, with the framework shown as **Google ADK** and one turn per `run_async()` call. +- **One session per ADK session id**, with the framework shown as **Google ADK** and one turn per `run_async()` call. The `run_async()` that sends a confirmation is a turn of its own, labeled with the original request. - **A transcript** on the session page: the user messages, the assistant replies, and each tool call with its arguments and result. - **LLM calls and tokens** for each `generate_content {model}` span, with the model from `gen_ai.request.model` (for LiteLLM, the full id such as `openrouter/openai/gpt-4o-mini`). - **Tool calls** named after your functions, with failed ones counted as errors. diff --git a/apps/landing/src/content/docs/agent-tracing/haystack.md b/apps/landing/src/content/docs/agent-tracing/haystack.md index dfe796add3..02064d5bdc 100644 --- a/apps/landing/src/content/docs/agent-tracing/haystack.md +++ b/apps/landing/src/content/docs/agent-tracing/haystack.md @@ -135,7 +135,8 @@ class MapleSpan(OpenTelemetrySpan): self._span.set_attribute("error.type", "ToolInvocationError") if self._content: attribute = "gen_ai.tool.call.arguments" if key.endswith(".input") else "gen_ai.tool.call.result" - self._span.set_attribute(attribute, json.dumps(value, default=str)) + # Tool results arrive as strings: write them as-is, so a JSON result stays a JSON object + self._span.set_attribute(attribute, value if isinstance(value, str) else json.dumps(value, default=str)) class MapleHaystackTracer(OpenTelemetryTracer): @@ -257,6 +258,8 @@ With the default `content=True`, the tracer writes: Maple builds the transcript and the turn labels from the `gen_ai.*` messages. The `haystack.*` blobs only show up in the raw span attributes. +Maple currently shows a tool span's result only when it is a JSON object or array. A tool that returns plain text, like `"Sunny, 21°C"`, still has its result in the transcript, as the tool result the next model call receives, but its tool call row shows no result. Return a dict from tools whose results you want on the tool pages. + `HAYSTACK_CONTENT_TRACING_ENABLED` has no effect with this tracer: `MapleHaystackTracer` decides on its own. Without the tracer, that variable is read once, at the first `import haystack`, and setting it any later silently records nothing. To keep prompts and responses out of Maple, turn content off: diff --git a/apps/landing/src/content/docs/agent-tracing/langchain.md b/apps/landing/src/content/docs/agent-tracing/langchain.md index b2639f009b..8b40c5743b 100644 --- a/apps/landing/src/content/docs/agent-tracing/langchain.md +++ b/apps/landing/src/content/docs/agent-tracing/langchain.md @@ -135,7 +135,9 @@ If you skip it, every `invoke()` shows up in **Agent Sessions** as its own one-t ## Record prompts, responses and tool calls -Content capture is on by default. Every chat model span carries the full message list sent to the model and the reply, as `gen_ai.input.messages` and `gen_ai.output.messages` in `{role, parts}` form, and Maple renders them as the session transcript, with the user's message as the turn's label. The system prompt is the first input message. Tool spans carry the tool's result in `gen_ai.tool.call.result`. +Content capture is on by default. Every chat model span carries the full message list sent to the model and the reply, as `gen_ai.input.messages` and `gen_ai.output.messages` in `{role, parts}` form, and Maple renders them as the session transcript. The system prompt is the first input message. Tool spans carry the tool's result in `gen_ai.tool.call.result`, as LangChain's serialized `ToolMessage`, so the result reads as a small JSON object with the output under `data.content`. + +Turn labels are the one part that goes wrong. Maple labels a turn with the user message on the first span of the turn that has messages, and in a `create_agent` graph that's the `model` node's span, where the instrumentor records only the thread's first message. With a checkpointer, every turn of a conversation is labeled with its opening message. The transcript inside each turn still starts with the right message. With a checkpointer, every model span repeats the thread's whole history, so a long conversation gets large. Maple has no per-attribute limit, and ingest accepts requests up to 20 MiB. @@ -151,7 +153,7 @@ config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=T Each tool call is a span named after the tool, with `gen_ai.operation.name` `execute_tool`, `gen_ai.tool.name`, `gen_ai.tool.description` and the result. The arguments aren't on the tool span, and neither is `gen_ai.tool.call.id`, because the instrumentor doesn't record them there. The arguments are still in the transcript, in the model's tool call just before. -A tool that raises is marked failed without extra code: its span ends with status `ERROR` and the exception as the status message, for example `RuntimeError('transport data service unavailable (503)')`, and Maple counts it on the session and on the tool's page. +A tool that raises is marked failed without extra code: its span ends with status `ERROR` and the exception plus its traceback as the status message, starting with `RuntimeError('transport data service unavailable (503)')`, and Maple counts it on the session and on the tool's page. What happens to the run next is up to LangChain. `create_agent` re-raises any exception from a tool by default, so one broken tool fails the whole `invoke()`. To hand the error to the model and keep going, add a `wrap_tool_call` middleware: @@ -256,7 +258,7 @@ Run one conversation of two or three messages through `handle_message` with the - **One session** for the conversation, with one turn per `invoke()`. A second conversation with a different id is a second session. - **Framework: Unidentified.** Maple recognizes LangChain by LangSmith's exporter, and the OpenInference spans read as generic GenAI spans. Everything else on the session page works. -- **The transcript**: your messages as turn labels, the model's replies, and its tool calls. +- **The transcript**: your messages, the model's replies, and its tool calls. Every turn's label repeats the conversation's first message (see [Record prompts, responses and tool calls](#record-prompts-responses-and-tool-calls)); the second conversation's single turn is labeled correctly. - **Agents**: `assistant`, plus one lane per sub-agent in `AGENT_NAMES`. Each turn's trace starts at the agent span, with `model` and `tools` node spans below it. - **Model calls** named `ChatOpenAI` (or `ChatAnthropic`, and so on), each with a model, input and output tokens, including streamed ones. - **Tool calls** named after your tools, like `get_weather`, with results, and a failing tool marked failed with its message. diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md index 8a30280567..f64127a273 100644 --- a/apps/landing/src/content/docs/agent-tracing/litellm.md +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -392,7 +392,7 @@ Don't add an OpenAI client instrumentor (OpenInference, OpenLLMetry, `openteleme Each request then shows up in your app's trace as `invoke_agent` → `POST /chat/completions` (the proxy's server span) → `chat gpt-4o-mini`, next to an `auth /chat/completions` span for the proxy's key check. Depending on its setup, the proxy adds more housekeeping spans (database, Redis, guardrails). -Maple currently counts each `auth /chat/completions` span as an extra LLM call, because it comes from LiteLLM and its name contains "chat". On the proxy path the LLM call count in Agent Sessions is double the real number. Tokens, cost, the transcript and the session grouping are not affected. +Maple currently counts each `auth /chat/completions` span as an extra LLM call, because it comes from LiteLLM and its name contains "chat". On the proxy path the LLM call count in the sessions list is double the real number. The session page also counts the proxy's `POST /chat/completions` server span, which carries `gen_ai.request.model`, so it shows three times the real number. Tokens, cost, the transcript and the session grouping are not affected. ## Short-lived processes @@ -449,7 +449,7 @@ Run one conversation with at least two messages and a tool call, then open **Age - **The last call of a script or Lambda is missing.** The process ended before LiteLLM's logging queue ran. Await `flush_tracing()` before the loop ends. - **Every model call appears twice.** Two loggers (the v2 instance plus `"otel"` in `litellm.callbacks`), or the proxy and an in-app OpenAI instrumentor both tracing the same call. Keep one. - **Proxy spans land in their own traces, apart from the app's agent span.** The request carried no `traceparent`. Inject it with `propagate.inject(headers)` inside the agent span, and don't set `OTEL_IGNORE_CONTEXT_PROPAGATION` on the proxy. -- **The LLM call count is twice what the app made, on the proxy path only.** Maple counts the proxy's `auth /chat/completions` spans as calls. Token and cost totals are correct. +- **The LLM call count is twice what the app made (three times on the session page), on the proxy path only.** Maple counts the proxy's `auth /chat/completions` spans as calls, and the session page also counts its `POST /chat/completions` server span. Token and cost totals are correct. - **A model call shows an empty reply in the transcript.** That call only requested tools. The tool calls appear as their own rows from your `execute_tool` spans. - **Session cost is lower than LiteLLM's spend.** Sub-agents report their own `gen_ai.usage.cost` and Maple subtracts it from the parent agent's. Report cost only on the outermost agent span, as in the cost section. - **Nothing arrives from the proxy.** It is still on the default `console` exporter because no endpoint reached it. Check that `OTEL_EXPORTER_OTLP_ENDPOINT` is set in the proxy's environment, not only in your app's. diff --git a/apps/landing/src/content/docs/agent-tracing/llamaindex.md b/apps/landing/src/content/docs/agent-tracing/llamaindex.md index 432bce860b..b205c3e984 100644 --- a/apps/landing/src/content/docs/agent-tracing/llamaindex.md +++ b/apps/landing/src/content/docs/agent-tracing/llamaindex.md @@ -325,7 +325,7 @@ A second conversation with a different id is a second session. If a turn is miss - **One session per message.** `agent.run()` isn't inside `using_session(...)`, or the id changes per request. `Context` and `llamaindex.run_id` are not session ids. - **Each model or tool call counted two or three times.** `LlamaIndexForMaple` isn't in front of the exporter. Add the exporter through it, not directly with `add_span_processor(BatchSpanProcessor(...))`. - **Framework shows "Unidentified".** The spans lack a `llamaindex.*` attribute: `LlamaIndexForMaple` is missing, or it's an older copy without the `llamaindex.instrumentor` line. -- **Model calls take 1 ms.** Streamed model spans end when LlamaIndex hands back the stream, not when the last token arrives, so their duration isn't the model's latency. The enclosing `BaseWorkflowAgent.run_agent_step` span has the real time. If you don't stream tokens to users, `FunctionAgent(..., streaming=False)` records model spans with their full duration. +- **Model calls take 1 ms.** Streamed model spans end when LlamaIndex hands back the stream, not when the last token arrives, so their duration isn't the model's latency, and the session's inference time adds up to a few milliseconds. The enclosing `BaseWorkflowAgent.run_agent_step` span has the real time. If you don't stream tokens to users, `FunctionAgent(..., streaming=False)` records model spans with their full duration. - **No tokens on streamed calls.** The provider didn't send usage on the stream. Add `stream_options={"include_usage": True}` through `additional_kwargs`. - **Tool arguments show the tool's schema.** A known gap in the instrumentor's GenAI output; see [Tools, errors and sub-agents](#tools-errors-and-sub-agents). - **An approved tool call shows as failed first.** `wait_for_event()` suspends the tool by raising `WaitingForEvent`. `LlamaIndexForMaple` drops that attempt; check it's installed. diff --git a/apps/landing/src/content/docs/agent-tracing/mastra.md b/apps/landing/src/content/docs/agent-tracing/mastra.md index 6c6f1eea16..7ab44d8103 100644 --- a/apps/landing/src/content/docs/agent-tracing/mastra.md +++ b/apps/landing/src/content/docs/agent-tracing/mastra.md @@ -245,6 +245,8 @@ Approvals leave one artifact. The model call that asks for the tool is exported Mastra's multi-agent idiom is a supervisor: an agent with an `agents` property calls each sub-agent through a tool named `agent-`. Each delegation runs the sub-agent inside the supervisor's trace, under that tool span. Give every agent a `name`; it becomes `gen_ai.agent.name`, and Maple draws one lane per agent name. An `execute_tool agent-weather_worker` span whose only child is `invoke_agent weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. +The `agent-` calls also count as tool calls in the session's totals, next to the sub-agents' own tools. A sub-agent's prompt in the transcript starts with the supervisor's instructions and the user's original message, then the delegated task: Mastra passes the supervisor's conversation along to each sub-agent. + ```ts export const orchestrator = new Agent({ id: "orchestrator", @@ -276,7 +278,7 @@ Steps in `.parallel([...])` run concurrently, and their lanes overlap in time in ## Tokens and cost -The `chat` span of each model call carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. Only that span carries usage, so the enclosing agent, step and generation spans add nothing to the total. +The `chat` span of each model call carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. The input count includes the cached tokens; Maple shows the cached part separately and counts it once. Only that span carries usage, so the enclosing agent, step and generation spans add nothing to the total. Reasoning tokens are exported as `gen_ai.usage.reasoning_tokens`, a key Maple doesn't read. They're still inside the output token count, so totals are right; only the reasoning breakdown is missing. @@ -324,7 +326,7 @@ Run one conversation with at least two messages and a tool call, then open **Age - one turn per `generate()` or `stream()` call, labeled with the user's message, and a transcript with the prompts, replies and tool calls; - `invoke_agent ` spans for runs, `chat ` spans for model calls, and `execute_tool ` spans for tool calls, with `agent_step` and `model_generation` spans in between; - `invoke_workflow ` as the turn for workflow runs; -- a lane per sub-agent, named after its `name`; +- a lane per sub-agent, named after its `name`, with each `agent-` delegation counted as a tool call; - token counts on every model call, including streamed ones, except one 0-token call per tool approval; - failed tool calls marked as failed, with the thrown message; - cost shown as unpriced. diff --git a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md index d8f8ef001c..ad8fc62930 100644 --- a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md +++ b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md @@ -263,7 +263,7 @@ with conversation(session.session_id): Workflows (`WorkflowBuilder`, or the `SequentialBuilder`, `ConcurrentBuilder`, `HandoffBuilder` and `MagenticBuilder` orchestrations) trace each executor as `executor.process ` with the agent's `invoke_agent` span inside it. Fan-in is recorded as span links, not parent-child, so parallel workers appear as siblings under `workflow.run`. The executors run as `asyncio` tasks and inherit the conversation id from the block. -Build the workflow inside the `conversation()` block too. `WorkflowBuilder.build()` emits its own one-span `workflow.build` trace. Inside the block it joins the session as a short extra trace with no model calls; built once at import time, it becomes a one-span session of its own. +Build the workflow inside the `conversation()` block too. `WorkflowBuilder.build()` emits its own one-span `workflow.build` trace. Inside the block it joins the session as a short extra trace with no model calls. Maple shows it as an empty first turn with no label or agent, the real run is turn 2, and the session has no title. Built once at import time, it becomes a one-span session of its own instead. Tools that need approval (`@tool(approval_mode="always_require")`) end the run with a pending request. The resume is a new `agent.run()` and a new trace; with the processor in place, it joins the same session. A rejected call emits no `execute_tool` span at all; the rejection only appears in the next `chat` span's input messages. @@ -367,7 +367,7 @@ Run one conversation of two or three turns, one of which calls a tool. Sessions - **One session per conversation**, labeled **Microsoft Agent Framework** (or **Semantic Kernel**) for Python, whose id is your conversation id, not `trace:…`. A second conversation is a second session. - **One turn per `agent.run()`**, each rooted at `invoke_agent support_agent` with `chat gpt-4o-mini` and `execute_tool get_weather` spans under it. An approval resume is its own turn in the same session. In .NET the root is `invoke_agent support_agent()`. -- **A transcript** with the system instructions, your messages and the replies, labeled by the first line of each user message. +- **A transcript** with the system instructions, your messages and the replies, labeled by the first line of each user message. Semantic Kernel shows only each turn's message and the agent's reply, with tool calls as tool spans. - **Tool calls** with arguments and results, and failed tools counted under **Tool errors**. - **Tokens** on every model call, including streamed ones. Cost shows as unpriced. - For multi-agent runs, a lane per agent name. diff --git a/apps/landing/src/content/docs/agent-tracing/openai-agents.md b/apps/landing/src/content/docs/agent-tracing/openai-agents.md index 5321a32644..f3b05b11af 100644 --- a/apps/landing/src/content/docs/agent-tracing/openai-agents.md +++ b/apps/landing/src/content/docs/agent-tracing/openai-agents.md @@ -196,6 +196,8 @@ OpenInference's own switches (`TraceConfig(hide_inputs=True, hide_outputs=True)` Each function tool call is a span named after the tool, with `gen_ai.operation.name` `execute_tool`, `gen_ai.tool.name`, the tool's description, its arguments (with `MapleSpanFixes`) and its result in `gen_ai.tool.call.result`. +Maple shows that result on the tool span only when it's JSON, as it is for a tool that returns a dict or a list. A plain-text result shows as not captured there: a tool that returns a string, an agent called as a tool, a handoff, or a failed tool's error text. The transcript still shows it, as the tool message the model received on its next call. + A tool that raises is marked failed without extra code. The SDK catches the exception, sends the model `An error occurred while running the tool. Please try again. Error: ...` as the tool result, and records the error on its span. The bridge turns that into status `ERROR` with a message like `Error running tool (non-fatal): {'tool_name': 'fetch_transport_data', 'error': 'transport data service unavailable (503)'}`, and Maple counts the call as failed on the session and on the tool's page. A tool that returns an error string instead of raising counts as a success. Every agent the run enters gets a span named after it, with `gen_ai.agent.name`, and Maple opens a lane for each agent whose name differs from its caller's. Give every `Agent` a distinct `name`. The spans in between are the bridge's bookkeeping: a `CHAIN` span named after the workflow per `Runner.run`, and a `turn` span per step of the agent loop. @@ -259,7 +261,7 @@ Run one conversation of two or three messages through `handle_message` with the - **One session** for the conversation, framework **OpenAI Agents SDK**, with one turn per `Runner.run`. Each turn's trace starts at a span named after your `workflow_name`. - **The transcript**: the agent's instructions as the system message, your messages, the model's replies (streamed ones included) and its tool calls. - **Model calls** named `generation` (Chat Completions) or `response` (Responses API), as many as the app made, each with a model and input and output tokens, streamed turns included. -- **Tool calls** named after your tools, such as `get_weather`, with arguments and results. A tool that raised is marked failed. +- **Tool calls** named after your tools, such as `get_weather`, with arguments, and results where the tool returned JSON. A tool that raised is marked failed, and the session's verdict names it under **Tool availability**. - **Agents**: one lane per agent name, such as `assistant`, plus sub-agents and handoff targets. - **Cost**: unpriced. @@ -267,7 +269,7 @@ A second conversation with a different id is a second session. If a turn is miss ## TypeScript (`@openai/agents`) -The same bridge exists for the TypeScript SDK as `@arizeai/openinference-instrumentation-openai-agents`. We ran it with `@openai/agents` 0.18 and bridge 0.2.15, and it works with Maple with three differences. Maple doesn't identify it as the OpenAI Agents SDK, so the framework shows as **Unidentified**. The TypeScript bridge has no GenAI dual-write: the transcript is built from the OpenInference `input.value` and `output.value` JSON, so model replies show as the raw API response, tool calls have no separate arguments and result fields, and there's no `gen_ai.agent.name` for lanes. And the session id has to be `gen_ai.conversation.id`, because Maple doesn't read `session.id` for unidentified OpenInference spans: +The same bridge exists for the TypeScript SDK as `@arizeai/openinference-instrumentation-openai-agents`. We ran it with `@openai/agents` 0.18 and bridge 0.2.15, and it works with Maple with three differences. Maple doesn't identify it as the OpenAI Agents SDK, so the framework shows as **Unidentified**. The TypeScript bridge has no GenAI dual-write, so the transcript is built from the OpenInference `input.value` and `output.value` JSON. Inputs render as messages, but `output.value` is the raw API response, which Maple can't read yet. Each model call's reply and the tool calls it asked for are missing from that call; an earlier reply only appears as history in the next call's input, so the last reply of every turn is missing. Tool spans show their input and output as messages but have no arguments and result fields, and there's no `gen_ai.agent.name`, so no agent lanes. And the session id has to be `gen_ai.conversation.id`, because Maple doesn't read `session.id` for unidentified OpenInference spans: ```ts import * as agents from "@openai/agents" diff --git a/apps/landing/src/content/docs/agent-tracing/openrouter.md b/apps/landing/src/content/docs/agent-tracing/openrouter.md index a0846e5de7..33dd9cf396 100644 --- a/apps/landing/src/content/docs/agent-tracing/openrouter.md +++ b/apps/landing/src/content/docs/agent-tracing/openrouter.md @@ -208,7 +208,7 @@ For tools and agent structure, instrument the app with its framework guide from Model failures do show up. Each request's trace has an `LLM Generation` root and a `provider attempt N: ` child per upstream provider OpenRouter tried: - A failed attempt that OpenRouter recovered from by falling back to another provider is a retry. Maple does not count it as a failure. -- If every attempt fails, `LLM Generation` has status Error with the message `Provider returned error`, and Maple counts it as a `provider_error` in the session. +- If every attempt fails, `LLM Generation` has status Error with the message `Provider returned error`, and Maple counts it as a `provider_error` in the session. Such a call carries no token usage, and Maple then counts the root and each failed attempt as separate model calls, so one failed request with one attempt shows as two LLM calls. On that error path, OpenRouter currently drops the `trace` object, so the failed call lands in its own trace. It keeps `session.id`, so it still joins the right session. @@ -237,6 +237,8 @@ Not read by Maple: `gen_ai.provider.name` is the model's author (`openai`, `anthropic`), not the provider that served the request. That one is in `trace.metadata.openrouter.provider_name` (for example `Amazon Bedrock`). +One known gap for Claude models: OpenRouter's input count includes cached tokens for every model, but Maple reads an `anthropic` input count as excluding them, so cache reads are counted twice in Claude sessions' token totals. In our test, a Claude session with 26,730 input tokens (4,248 of them cached) showed 32,573 total tokens instead of 28,325. Cost is unaffected, because it comes from OpenRouter's charge. + ## Short-lived processes Broadcast needs no flush. OpenRouter sends traces from its own servers after each request completes, so a script, a serverless function or a CLI that exits right after the response loses nothing. @@ -250,10 +252,10 @@ If your app also exports its own spans, those still need the usual flush on exit Run one conversation of three or more turns, with a tool call, sending the same `session_id` on every request. Wait about a minute, then open **Agent Sessions** in Maple and filter by service `openrouter`. - **One session for the conversation**, named by your `session_id`, with vendor **OpenRouter**. A second conversation is a second session. -- **Model calls.** Each call has an `LLM Generation` span with one or more `provider attempt N: ` children, and sometimes `generation` or `moderation` children. Only `LLM Generation` counts as a model call. +- **Model calls.** Each call has an `LLM Generation` span with one or more `provider attempt N: ` children, and sometimes `generation` or `moderation` children. Only `LLM Generation` counts as a model call, except on a request where every provider attempt failed (see above). - **Tokens and cost** on the session and per model, with cost in USD. - **Transcript**: each call's prompt and completion as a JSON block, unless Privacy Mode is on. -- **Turns**: one per trace. Broadcast-only, that's one per model call. Nested under your own traces, it's one per turn of your agent. +- **Turns**: one per trace. Broadcast-only, that's one per model call, listed as unlabeled segments (Segment 1, Segment 2, ...). Nested under your own traces, it's one per turn of your agent. - **Tools**: none from Broadcast. Tool spans come from your app's instrumentation. The service name on Broadcast spans is always `openrouter`, and there is no environment attribute. Custom keys in the `trace` object arrive as `trace.metadata.` attributes, which you can search in [Traces](/docs/explore/traces) but which don't set Maple's environment or service. @@ -263,7 +265,7 @@ The service name on Broadcast spans is always `openrouter`, and there is no envi - **Test Connection fails.** The endpoint must be the full `https://ingest.maple.dev/v1/traces` URL and the headers valid JSON with `"Authorization": "Bearer YOUR_INGEST_KEY"`. EU organizations use `ingest.eu.maple.dev`. - **Test Connection passes but nothing arrives.** Check the destination's API key filter and data regions against the key and endpoint your app actually uses, and that **Enable Broadcast** is on for the account or organization your app's key belongs to. A placeholder key such as `MAPLE_TEST` passes the test, but Maple discards everything it sends. - **Every call is its own session, named `trace:`.** The request has no `session_id`. Check the outgoing body, not just your code: a wrapper or framework may drop unknown fields. -- **A session named `trace:00000000000000000000000000000001` with one span.** That's the `openrouter-connection-test` span from **Test Connection**. Ignore it. +- **A session named `trace:00000000000000000000000000000001` with no model calls.** That's the `openrouter-connection-test` span from **Test Connection**. Every test reuses that trace id, so the session gains a span per click. Ignore it. - **A session with an `openai/gpt-4-turbo` call you never made, trace name `Test Trace - OpenRouter Observability`.** OpenRouter's sample trace from the destination settings, with sample tokens and cost. - **Tokens or LLM calls are about double what you expect.** Your app's instrumentation and Broadcast both report the call, in different traces and without a shared response id. Nest Broadcast with `trace.trace_id` and `trace.parent_span_id`, or filter that service's API key out of the destination. - **Broadcast spans show up as a separate trace despite `trace.trace_id`.** The id isn't W3C hex, or the call failed: on the all-providers-failed path OpenRouter drops the `trace` object. diff --git a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md index a90a2b0403..3616c26754 100644 --- a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md +++ b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md @@ -468,7 +468,7 @@ Run one conversation of two or three messages with at least one tool call, flush - **A failing tool shows as successful.** The tool's exception was caught outside `run_tool`. Let `run_tool` catch it, or set the span status and `error.type` where you catch it. - **Sub-agent calls land in the orchestrator's lane.** The sub-agent's `agent_span` has the same name as the orchestrator's, or it was never wrapped. Give each agent its own name. - **Two conversation ids in one trace.** With the OpenAI Responses API and a `conversation` parameter, the instrumentation also sets `gen_ai.conversation.id` to that `conv_...` id. Pass the same id to `agent_span`, or Maple picks one of the two and can split the turn. -- **Cached Anthropic calls show more input tokens than you were billed for (Python).** The Anthropic GenAI instrumentation reports `gen_ai.usage.input_tokens` as the raw input plus cache reads plus cache writes (a call that sent 330 new tokens and wrote 7,581 to the cache reports 7,911), while Maple reads Anthropic's figure as excluding the cache and adds both buckets again. Calls without prompt caching are unaffected. +- **Cached Anthropic calls show more input tokens than you were billed for (Python).** The Anthropic GenAI instrumentation reports `gen_ai.usage.input_tokens` as the raw input plus cache reads plus cache writes (a call that sent 330 new tokens and wrote 7,581 to the cache reports 7,911), while Maple reads Anthropic's figure as excluding the cache and adds both buckets again. In our test, a three-turn session that processed 40,239 tokens showed 78,144 in Maple. Calls without prompt caching are unaffected. - **Using the Anthropic SDK through OpenRouter.** Point it at `https://openrouter.ai/api` (no `/v1`): `Anthropic(base_url="https://openrouter.ai/api", api_key=OPENROUTER_API_KEY)`. OpenRouter accepts the key as `api_key` or `auth_token`. Model ids are OpenRouter's, such as `anthropic/claude-haiku-4.5`, and the spans say `gen_ai.provider.name=anthropic`. - **Nothing arrives, and the exporter logs 401.** The key or region is wrong. EU keys only work with `ingest.eu.maple.dev`. diff --git a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md index d2a8dd0546..cd08b4b0d5 100644 --- a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md @@ -204,7 +204,7 @@ def fetch_transport_data(city: str) -> dict: Returning an error payload keeps the span green, and Maple counts the call as a success. -Approval-gated tools (`requires_approval=True`) pause the run without a tool span. The call only gets an `execute_tool` span when the resumed run executes it, so pass the same `conversation_id=` to the run that sends `DeferredToolResults`. +Approval-gated tools (`requires_approval=True`) pause the run without a tool span. The call only gets an `execute_tool` span when the resumed run executes it, so pass the same `conversation_id=` to the run that sends `DeferredToolResults`. Maple then shows the paused run and the resumed run as two turns of the same session, both labeled with the original request. ### Sub-agents: pass the conversation id down diff --git a/apps/landing/src/content/docs/agent-tracing/smolagents.md b/apps/landing/src/content/docs/agent-tracing/smolagents.md index 58ce1a22b2..0af22f2afe 100644 --- a/apps/landing/src/content/docs/agent-tracing/smolagents.md +++ b/apps/landing/src/content/docs/agent-tracing/smolagents.md @@ -166,7 +166,7 @@ When a `ToolCallingAgent` model asks for several tools or managed agents in one `CodeAgent` calls tools from the Python code the model writes, run by `LocalPythonExecutor` in your process. Those calls produce the same tool spans; positional arguments are recorded as a JSON array, like `["Berlin"]`. With a remote executor (`executor_type="e2b"`, `"docker"`, `"modal"` and others), the code and its tool calls run outside your process, so there are no tool spans, only the model and step spans. -Two gaps remain in what Maple can show for smolagents tools. Tool spans have no `gen_ai.tool.call.id`, because smolagents doesn't pass the model's tool call id to the tool, so Maple can't link a tool span to the exact call in the model's reply. And the step's failure and the tool's failure are both on the trace, so a session with one broken tool has two failed spans. +Two gaps remain in what Maple can show for smolagents tools. Tool spans have no `gen_ai.tool.call.id`, because smolagents doesn't pass the model's tool call id to the tool, so Maple can't link a tool span to the exact call in the model's reply. And the step's failure and the tool's failure are both on the trace, so a session with one broken tool has two failed spans. Maple counts one failed tool call, and reports the failed `Step` span separately as an "Other errors" warning with the `AgentToolExecutionError` message. ## Tokens and cost @@ -201,10 +201,10 @@ Call `provider.shutdown()` instead of `force_flush()` when the process is about Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: -- **One session** for the conversation, framework **smolagents**, with one turn per `agent.run()`. Turn traces start at `assistant.run` (your agent's name). +- **One session** for the conversation, framework **smolagents**, with one turn per `agent.run()`. Turn traces start at `assistant.run` (your agent's name). Every turn, and the session title, is labeled `New task:`, not with your message: Maple labels a turn with the first line of its user message, and smolagents puts `New task:` on that line. - **The transcript**: your messages and the model's replies. smolagents sends each task to the model as `New task:` followed by your text, and tool results come back as `tool-response` messages. - **Model calls** named `OpenAIModel.generate` (or `LiteLLMModel.generate`, `InferenceClientModel.generate`), each with a model, input and output tokens. -- **Tool calls** named `execute_tool get_weather` and `execute_tool final_answer`, with arguments and results. +- **Tool calls** named `execute_tool get_weather` and `execute_tool final_answer`, with arguments and results. The tool-call count includes one `final_answer` per agent run, managed agents included. - **Agents**: `assistant`, plus one lane per managed agent if you use them. - **Cost**: unpriced. diff --git a/apps/landing/src/content/docs/agent-tracing/strands.md b/apps/landing/src/content/docs/agent-tracing/strands.md index 45690ed40d..3ea4589be9 100644 --- a/apps/landing/src/content/docs/agent-tracing/strands.md +++ b/apps/landing/src/content/docs/agent-tracing/strands.md @@ -203,9 +203,11 @@ The `invoke_agent` span also carries usage, and this is where counts go wrong: A per-request agent (as in the session example above) never accumulates across turns, which is another reason to create agents per request. +One Maple issue remains even with a correct setup: the **Agent Sessions** list currently shows about twice the real tokens for Strands sessions, in Python and TypeScript. The list only recognizes a roll-up whose model calls sit directly below it, and Strands puts an `execute_event_loop_cycle` span in between. The session's own page shows the right totals, the sum of the `chat` spans. + Strands doesn't emit cost, so Maple shows sessions as unpriced. Maple never prices tokens itself. -Two more gaps: Strands records time to first token as `gen_ai.server.time_to_first_token` in milliseconds, which Maple doesn't read, and `chat` spans carry no `gen_ai.response.id`. The missing response id means Maple can't recognize the same call reported twice, so don't also enable a gateway export such as OpenRouter Broadcast for the same traffic. +More gaps: Strands records time to first token as `gen_ai.server.time_to_first_token` in milliseconds, which Maple doesn't read, and `chat` spans carry no `gen_ai.response.id`. The finish reason is only inside the output messages, so Maple's reply-length check shows as skipped. The missing response id means Maple can't recognize the same call reported twice, so don't also enable a gateway export such as OpenRouter Broadcast for the same traffic. ## Flush before short-lived processes exit @@ -282,7 +284,7 @@ Run one conversation of two or three messages, including one that calls a tool, - **One session per conversation id**, with the framework shown as Strands Agents and one turn per `agent(...)` call. A list of one-turn sessions means `session.id` is missing. - **The transcript**: each user message, the assistant's replies and the tool calls with their arguments and results. - **Spans named** `invoke_agent support_agent` (your agent's `name`), `execute_event_loop_cycle`, `chat` and `execute_tool get_weather`. -- **Tokens** on every model call, including streamed ones, and the model id you passed (for example `gpt-4o-mini`, or a Bedrock id such as `us.anthropic.claude-sonnet-4-5-20250929-v1:0`). +- **Tokens** on every model call, including streamed ones (the session page has the right total; the list currently shows about twice that), and the model id you passed (for example `gpt-4o-mini`, or a Bedrock id such as `us.anthropic.claude-sonnet-4-5-20250929-v1:0`). - **Sub-agents** as separate lanes named after each agent, with failed tool calls counted under the tool's name. - **Cost** shown as unpriced. @@ -291,7 +293,7 @@ Run one conversation of two or three messages, including one that calls a tool, - **Transcript is empty, but tokens and tools show up.** Content is still in span events. Add `gen_ai_span_attributes_only` (and `gen_ai_latest_experimental`) to `OTEL_SEMCONV_STABILITY_OPT_IN`, make sure it's set before the first `Agent` is created, and upgrade to 1.48 or newer. - **Every message is its own session.** No `session.id` on the trace. Pass `trace_attributes={"session.id": conversation_id}` to the agent, or to the `Swarm` / `Graph` that runs it. Setting only the session manager's `session_id` isn't enough. - **Two users' messages land in one session.** A shared `Agent` instance carries one `trace_attributes` dict for everyone. Create the agent per request. -- **Token totals look several times too high.** The `invoke_agent` span reports the agent's lifetime usage. Add `gen_ai_use_latest_invocation_tokens`, or create the agent per request. +- **Token totals look several times too high.** The `invoke_agent` span reports the agent's lifetime usage. Add `gen_ai_use_latest_invocation_tokens`, or create the agent per request. If only the sessions list shows about twice the session page's total, that's the known list issue from [Tokens, cost and the agent-level roll-up](#tokens-cost-and-the-agent-level-roll-up); trust the session page. - **Cache tokens are zero with prompt caching on.** Versions before 1.54 use `cache_read_input_tokens`, which Maple doesn't read. Upgrade. - **Every sub-agent is called "Strands Agents".** The agents have no `name`. Set `Agent(name=...)` on each one. - **Graph session splits from the rest of the conversation.** `GraphBuilder.build()` drops trace attributes. Set `graph.trace_attributes` after building. diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md index a73204dda1..f444b110c1 100644 --- a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -291,7 +291,9 @@ A pipeline of separate top-level calls (an orchestrator, then a summary agent) p Each `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. With `usage: true`, reasoning tokens are added as `ai.usage.outputTokenDetails.reasoningTokens`. Maple reads all of them. -The `invoke_agent` span repeats the call's total. Maple counts each model call once: a span's usage is netted against the usage its descendants already reported, so the total isn't doubled, and a sub-agent's tokens stay on the sub-agent's spans. +The `invoke_agent` span repeats the total of its own `chat` spans, two levels up (`invoke_agent` → `step` → `chat`). Maple doesn't net that repeat out yet, so a session's token total, in the session list and on the session page, is twice what the model calls used. The token counts on each `chat` span and the per-model breakdown on the session page are correct. Use those until this is fixed. + +A sub-agent's tokens stay on the sub-agent's spans: the orchestrator's `invoke_agent` repeats only its own model calls, not the workers'. Streaming needs nothing extra. The AI SDK normalizes provider usage, so `streamText` and `agent.stream()` calls have token counts too, and the streamed `chat` span records time to first chunk, which Maple shows as TTFT. @@ -378,7 +380,7 @@ Run one conversation with at least two messages and a tool call, then open **Age - one turn per `generate()` or `stream()` call, each labeled with the user's message, and a transcript with the prompts, replies and tool calls; - per turn, an `invoke_agent ` span, a `step ` span per loop iteration, a `chat ` span per model request and an `execute_tool ` span per tool call; - the agent name from `functionId` in the agent filter, and a lane per sub-agent; -- token counts on every model call, including streamed ones, and TTFT on streamed calls; +- token counts on every model call, including streamed ones, and TTFT on streamed calls (the session's token total reads double; see [Tokens and cost](#tokens-and-cost)); - failed tool calls marked as failed, with the error message; - cost shown as unpriced. @@ -389,6 +391,7 @@ Span names carry the model id, not the agent name, so two agents on the same mod - **No AI spans at all.** `registerTelemetry()` was never called, or ran in a module that isn't loaded. AI SDK 7 emits nothing without it, and `experimental_telemetry: { isEnabled: true }` alone does nothing. Import `@ai-sdk/otel` and register `new OpenTelemetry()` at startup. - **Every message is its own session.** No `gen_ai.conversation.id` on the spans. Check all three parts: `enrichSpan` on the integration, `runtimeContext: { conversationId }` on the call (or `prepareCall` for an agent), and `includeRuntimeContext: { conversationId: true }`. Without the last one, `enrichSpan` receives an empty object. - **Sessions show framework "Unidentified".** The spans carry no `ai.*` attributes. Set `usage: true` and `runtimeContext: true` on `new OpenTelemetry()`, and don't pass a custom `tracer` with another name. +- **The session's token total is twice the sum of its model calls.** Expected for now: Maple counts the total on each `invoke_agent` span on top of its `chat` spans. The per-call counts and the per-model breakdown are correct. If the spans themselves are duplicated, see the next item. - **Every span shows up twice.** `registerTelemetry()` ran twice, both `OpenTelemetry` and `LegacyOpenTelemetry` are registered, or a second OpenTelemetry SDK exports the same spans (Sentry without `skipOpenTelemetrySetup: true` next to `@vercel/otel`, for example). Register one integration and one exporter. - **One call has no spans while others do.** It passes `telemetry.integrations`, which replaces the globally registered integrations for that call. Add the `OpenTelemetry` instance to that list, or remove the option. - **Nothing arrives from a script or function.** The process ended before the batch was exported. Call `sdk.shutdown()` or `spanProcessor.forceFlush()` in a `finally`. diff --git a/skills/maple-agent-tracing-agno/SKILL.md b/skills/maple-agent-tracing-agno/SKILL.md index 4e58e3f759..eb0b9f622d 100644 --- a/skills/maple-agent-tracing-agno/SKILL.md +++ b/skills/maple-agent-tracing-agno/SKILL.md @@ -164,7 +164,7 @@ Run one real conversation: 2-3 turns with the same `session_id` including one to - [ ] Model calls (`.invoke|ainvoke|invoke_stream|ainvoke_stream`) show the model id and non-zero input/output tokens, including streamed turns. - [ ] Each tool call appears once with its real name; a raised tool error is marked failed and nothing else is. Structured results render as JSON (not a Python repr). - [ ] Teams: one lane per named member; the team run and all member spans are in one trace, same session; a run started after a team run is NOT inside the team's trace (else see Step 5 context leak). -- [ ] Cost shown if the provider returns it (OpenRouter does); otherwise "unpriced" is expected. +- [ ] Cost shown in the Agent Sessions list if the provider returns it (OpenRouter does); otherwise "unpriced" is expected. The session detail page shows cost as not reported for Agno even then (it doesn't read `llm.cost.total` yet); tell the user, don't try to fix it. - [ ] No span attribute contains the model provider API key or `Bearer`. Raw span check (optional, e.g. with a console exporter in a scratch run): run spans `.run` have `session.id` + `gen_ai.operation.name=invoke_agent` + `gen_ai.agent.name`; model spans have `gen_ai.operation.name=chat`, `gen_ai.input.messages`, `gen_ai.usage.input_tokens`; tool spans have `gen_ai.operation.name=execute_tool`. diff --git a/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md b/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md index a7082b96d2..450a40ddde 100644 --- a/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md +++ b/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md @@ -201,7 +201,7 @@ Then in Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `ap - Tool calls listed by real name (`mcp____`, `Bash`, ...), args for Bash/file tools, results when `OTEL_LOG_TOOL_CONTENT=1`. - A tool that threw is marked failed; successful tools are not. - Sub-agent model/tool calls appear under the `Agent` tool call in the same trace, and no turn is titled `` (if one is, see background sub-agents in Step 5). -- Expected and not bugs: cost "unpriced"; no assistant text; no sub-agent lanes. +- Expected and not bugs: cost "unpriced"; no assistant text; no sub-agent lanes; a "Prompt cache" warning with 0% hit rate when prompts are below the model's minimum cacheable length (Claude Code reports zero cache buckets explicitly). If spans exist but the turn nests under an unrelated trace, an inherited `TRACEPARENT` survived: fix the env stripping. diff --git a/skills/maple-agent-tracing-crewai/SKILL.md b/skills/maple-agent-tracing-crewai/SKILL.md index 9bc2e4d83b..55831219f5 100644 --- a/skills/maple-agent-tracing-crewai/SKILL.md +++ b/skills/maple-agent-tracing-crewai/SKILL.md @@ -194,6 +194,7 @@ def stream_message(conversation_id: str, text: str, history: str, send) -> None: Run one real conversation (2-3 messages, same conversation id, at least one tool call), and one message in a second conversation. If the user gave no key (`MAPLE_TEST`), you can't see results in Maple; say so and list what they should check. Otherwise check in Maple **Agent Sessions** (`https://app.maple.dev/agent-sessions`, EU `app.eu.maple.dev`), filtered to the service name: - Exactly one session per conversation id (two here), not one per message and no `trace:` sessions. +- Framework shows CrewAI in the session list (model spans themselves are tagged `openinference-openai`; that's fine). - One turn per kickoff; each turn's root span is `.kickoff` (or `.kickoff`, or your `invoke_agent` wrapper when streaming). No empty extra turns. - Transcript is non-empty (system message from role/goal/backstory, `Current Task: …`, replies). - Model calls (`ChatCompletion` for the OpenAI instrumentor) have a model and non-zero input/output tokens, including streamed calls. Each call appears once. diff --git a/skills/maple-agent-tracing-dspy/SKILL.md b/skills/maple-agent-tracing-dspy/SKILL.md index 8d6929fab9..324197408c 100644 --- a/skills/maple-agent-tracing-dspy/SKILL.md +++ b/skills/maple-agent-tracing-dspy/SKILL.md @@ -241,8 +241,9 @@ Run one conversation of 2-3 messages with the same conversation id (one using a - Model calls named `LM.__call__`, each with model, input and output tokens, and cost where DSPy knows the price. The LLM call count equals the real number of model calls (not double). - If the app streams: the streamed turn is in the same session and its `LM.__call__` spans have tokens. - Transcript non-empty (DSPy's `[[ ## field ## ]]` prompt format is expected). -- Tool calls named `.__call__` with `gen_ai.tool.name`, arguments and results. `finish.__call__` (ReAct's end-of-loop tool) is expected. -- A tool that raised is counted as failed; no successful tool or model call is marked failed. +- Tool calls named `.__call__` with `gen_ai.tool.name`, arguments and results. `finish.__call__` (ReAct's end-of-loop tool) is expected and counts as a tool call. +- A tool that raised is counted as failed (session check **Tool availability** fails with the exception); no successful tool or model call is marked failed. +- A **Prompt cache** warning on a short test conversation is expected: the callback records the provider's cached input tokens (0 counts), and providers only cache long prompts (OpenAI from 1,024 tokens, Anthropic only with cache markers). - Worker modules appear as separate agents/lanes; with `dspy.Parallel`, all workers are in the same trace as the orchestrator, none orphaned. - The process exited cleanly and the last turn is present (flush ran). diff --git a/skills/maple-agent-tracing-google-adk/SKILL.md b/skills/maple-agent-tracing-google-adk/SKILL.md index 190461d113..a4bd4e6840 100644 --- a/skills/maple-agent-tracing-google-adk/SKILL.md +++ b/skills/maple-agent-tracing-google-adk/SKILL.md @@ -157,7 +157,7 @@ Run one conversation of 2-3 turns with the same session id, one turn calling a t Without a real key (`MAPLE_TEST`), verify locally: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` to the provider, run one turn, and confirm `generate_content` spans have `gen_ai.conversation.id`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.input_tokens`; `execute_tool` spans have `gen_ai.tool.call.arguments`. Remove the console exporter afterwards. -Tell the user about the known gaps: cost is unpriced; the model is the requested id, not the served one, and there is no `gen_ai.response.id`; with `StreamingMode.SSE` the transcript shows the streamed chunks and then the full reply (ADK records each chunk). +Tell the user about the known gaps: cost is unpriced; the model is the requested id, not the served one, and there is no `gen_ai.response.id`; with `StreamingMode.SSE` the transcript shows the streamed chunks and then the full reply (ADK records each chunk); session check headlines (provider errors, prompt cache) count `call_llm` and `generate_content` separately, so they show twice the LLM call count (tokens and LLM calls are netted correctly). ## Do not diff --git a/skills/maple-agent-tracing-haystack/SKILL.md b/skills/maple-agent-tracing-haystack/SKILL.md index fd9c83c0bb..c8018fa336 100644 --- a/skills/maple-agent-tracing-haystack/SKILL.md +++ b/skills/maple-agent-tracing-haystack/SKILL.md @@ -120,7 +120,8 @@ class MapleSpan(OpenTelemetrySpan): self._span.set_attribute("error.type", "ToolInvocationError") if self._content: attribute = "gen_ai.tool.call.arguments" if key.endswith(".input") else "gen_ai.tool.call.result" - self._span.set_attribute(attribute, json.dumps(value, default=str)) + # Tool results arrive as strings: write them as-is, so a JSON result stays a JSON object + self._span.set_attribute(attribute, value if isinstance(value, str) else json.dumps(value, default=str)) class MapleHaystackTracer(OpenTelemetryTracer): @@ -204,6 +205,7 @@ history = [m for m in result["assistant"]["messages"] if not m.is_from("system") ## Step 4: Content - Default `MapleHaystackTracer(..., content=True)` writes `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments`/`.result`, plus Haystack's own `haystack.*` content tags. +- Maple shows a tool span's `gen_ai.tool.call.result` only when it is a JSON object/array. Plain-text tool results still appear in the transcript (as the next model call's tool message) but not on the tool call row. Don't change tools for this; mention it if the user's tools return strings. - `content=False` keeps model, tokens, cost, finish reason, tool names and failures; drops messages, tool args/results, `haystack.*` content tags and the ungated `haystack.pipeline.input_data` tag. Use it if the user asks for no prompts/PII in telemetry. The failed tool's status message still quotes its arguments (Haystack's `Failed to invoke Tool ... with parameters {...}`); if arguments can hold PII, redact them in `_tool()` before `set_status`. - `HAYSTACK_CONTENT_TRACING_ENABLED` is irrelevant with this tracer; don't add it. diff --git a/skills/maple-agent-tracing-langchain/SKILL.md b/skills/maple-agent-tracing-langchain/SKILL.md index 597c8a040d..b86fe07c72 100644 --- a/skills/maple-agent-tracing-langchain/SKILL.md +++ b/skills/maple-agent-tracing-langchain/SKILL.md @@ -176,7 +176,7 @@ Run one real conversation: 2+ messages with the same id, at least one tool call, - [ ] Exactly one session per conversation, id = the `thread_id` you passed. A second conversation is a different session. - [ ] Framework shows **Unidentified** (expected for this path). -- [ ] One turn per `invoke()`; transcript shows user messages as turn labels, assistant replies, tool calls (not a raw JSON blob). +- [ ] One turn per `invoke()`; transcript shows each turn's user message, assistant replies, tool calls (not a raw JSON blob). Turn labels all repeat the conversation's first message: expected, see Known gaps. - [ ] Each turn trace starts at the agent span (`name=`), with `model`/`tools` node spans, `ChatOpenAI` (or other chat model class) spans and tool spans named after the tools, all in one trace. - [ ] Tool call count = the tools the model actually called (a higher count means a tool node is missing from `STEP_NAMES`). - [ ] Every chat model span has input and output tokens, including the streamed one. @@ -185,7 +185,7 @@ Run one real conversation: 2+ messages with the same id, at least one tool call, - [ ] Cost shows "unpriced" (expected: nothing records cost). - [ ] No attribute contains an API key or `Bearer ` token. -Known gaps (not setup bugs, don't try to fix): tool spans have no `gen_ai.tool.call.id` or arguments (arguments are in the model's tool call in the transcript); chat model spans have no `gen_ai.response.id`. +Known gaps (not setup bugs, don't try to fix): tool spans have no `gen_ai.tool.call.id` or arguments (arguments are in the model's tool call in the transcript); tool results are LangChain's serialized `ToolMessage` JSON (output under `data.content`); chat model spans have no `gen_ai.response.id`; with a checkpointer every turn's label is the thread's first user message (Maple takes it from the `model` node span, where the instrumentor records only the first message; the transcript inside each turn is right). Local check without Maple: add `SimpleSpanProcessor(ConsoleSpanExporter())` temporarily and confirm `gen_ai.conversation.id` is identical on every span of every turn of one conversation. diff --git a/skills/maple-agent-tracing-litellm/SKILL.md b/skills/maple-agent-tracing-litellm/SKILL.md index 20dae5d591..fc589b4fa8 100644 --- a/skills/maple-agent-tracing-litellm/SKILL.md +++ b/skills/maple-agent-tracing-litellm/SKILL.md @@ -125,7 +125,7 @@ async def call_model(conversation_id: str, messages: list, tools: list | None): ``` - Alternative session carrier: body `metadata: {"session_id": ...}` (`extra_body={"metadata": {...}}`), verified. -- Resulting tree per request: `invoke_agent` → `POST /chat/completions` (proxy FastAPI server span) → `chat ` + `auth /chat/completions`. Known Maple limitation: the `auth /chat/completions` span is counted as an extra LLM call (litellm scope + "chat" in the name), so LLM call counts double on the proxy path; tokens, cost, transcript and sessions are correct. Tell the user; nothing to fix app-side. +- Resulting tree per request: `invoke_agent` → `POST /chat/completions` (proxy FastAPI server span) → `chat ` + `auth /chat/completions`. Known Maple limitation: the `auth /chat/completions` span is counted as an extra LLM call (litellm scope + "chat" in the name), so LLM call counts double on the proxy path in the sessions list; the session detail page also counts the FastAPI `POST /chat/completions` span (it carries `gen_ai.request.model`), so it shows 3x. Tokens, cost, transcript and sessions are correct. Tell the user; nothing to fix app-side. - Never also instrument the app's OpenAI client when the proxy traces: double LLM calls and tokens. Pick gateway OR in-app. - Do not set `OTEL_IGNORE_CONTEXT_PROPAGATION=true` on the proxy. @@ -267,7 +267,7 @@ Run one real conversation: 2+ messages with the same id, one tool call, one stre - [ ] Cost: unpriced, or (Step 6) `gen_ai.usage.cost` only on top-level `invoke_agent` spans, equal to the sum of the turn's `litellm.cost.total`. - [ ] Last call of a script run present (flush worked). - [ ] No attribute contains an API key, `Bearer ` or `sk-`. -- [ ] Proxy path: `chat` spans (service `litellm-proxy`) sit in the app's trace under `invoke_agent` → `POST /chat/completions`, carrying the session id. LLM call count shows 2x (auth spans, known). +- [ ] Proxy path: `chat` spans (service `litellm-proxy`) sit in the app's trace under `invoke_agent` → `POST /chat/completions`, carrying the session id. LLM call count shows 2x in the list, 3x on the session page (auth + FastAPI spans, known). Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` to the provider and confirm `gen_ai.conversation.id` on every `chat` span and the parent ids. diff --git a/skills/maple-agent-tracing-llamaindex/SKILL.md b/skills/maple-agent-tracing-llamaindex/SKILL.md index 22db34c38b..6abdffa8d6 100644 --- a/skills/maple-agent-tracing-llamaindex/SKILL.md +++ b/skills/maple-agent-tracing-llamaindex/SKILL.md @@ -190,7 +190,7 @@ llm = OpenAI(model="gpt-4o-mini", additional_kwargs={"stream_options": {"include ``` (LlamaIndex strips it from non-streaming requests.) OpenRouter sends usage without it. -- Streamed model spans end when the stream is handed back (~1 ms), so their duration is not model latency. If the app never streams tokens to users, `FunctionAgent(..., streaming=False)` gives real model-span durations. Ask before changing it. +- Streamed model spans end when the stream is handed back (~1 ms), so their duration is not model latency (Maple's session inference time totals a few ms; `FunctionAgent` streams by default, so this is every call). If the app never streams tokens to users, `FunctionAgent(..., streaming=False)` gives real model-span durations. Ask before changing it. - Cost: never recorded. Sessions show "unpriced". Do not add pricing code. OpenRouter users can add OpenRouter Broadcast for cost. ## Step 7: Flush diff --git a/skills/maple-agent-tracing-mastra/SKILL.md b/skills/maple-agent-tracing-mastra/SKILL.md index 019bf1bc11..8628c42456 100644 --- a/skills/maple-agent-tracing-mastra/SKILL.md +++ b/skills/maple-agent-tracing-mastra/SKILL.md @@ -161,6 +161,7 @@ HITL resumes (`approveToolCallGenerate` / `declineToolCallGenerate` / `approveTo - Failures must throw (`throw new Error("...")`). Thrown → span status ERROR with the message as status message, `error.type=unknown`, plus an `exception` event. Returning `{ error }` = counted as success in Maple. If a tool swallows errors into a return value and the user wants failures visible, rethrow. - Every `new Agent({ id, name, ... })` needs a distinct `name`: it is `gen_ai.agent.name`, which Maple uses for lanes. - Supervisor agents (`agents: {...}` on an Agent): each delegation is `execute_tool agent-` → `invoke_agent ` inside the supervisor's trace, one lane per sub-agent. Mastra gives each delegation a thread id `-`, which sorts after the real id, so without `mapleSpanProcessor` Maple picks a sub-agent's id as the session and splits the turn. Same for `agent.network()` and workflow steps calling agents with their own `memory`. The processor fixes all of them; do not remove it. +- Maple counts each `execute_tool agent-` delegation as a tool call, next to the sub-agents' own tools. A sub-agent's `chat` input starts with the supervisor's system prompt and the user's original message (Mastra forwards the supervisor's conversation); expected. - HITL (`requireApproval: true` tools): the model call that requests the tool is exported twice, once without tokens or output when the run suspends and once with its tokens when the run resumes. Maple shows one extra model call with 0 tokens per approval; token totals are right. Expected, not a setup error. - Workflow steps that call an agent: pass `tracingContext` from the step's `execute` args into `agent.generate(prompt, { tracingContext })`, or the agent starts a separate trace. @@ -185,7 +186,7 @@ Run one conversation: 2+ user messages with the same thread id (one streamed), a - `chat ` spans have model, provider, `gen_ai.input.messages` and input/output tokens, including the streamed turn. Exception: with HITL, one `chat` span per approval has no tokens (Step 5). - No span has `mastra.metadata.headers` or `mastra.metadata.body`. - `execute_tool ` spans have the real tool name, arguments and result; a throwing tool is marked failed with its message; successful tools are not. -- Supervisor/workflow: one session, one turn per run, one lane per sub-agent `name`, all spans in one trace. +- Supervisor/workflow: one session, one turn per run, one lane per sub-agent `name`, all spans in one trace; tool calls include the `agent-` delegations. - Cost shows as unpriced (expected: Mastra emits no cost attribute). ## Do not diff --git a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md index adeed7afee..f6e9e71f3f 100644 --- a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md +++ b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md @@ -146,7 +146,7 @@ async def handle_message(session: AgentSession, text: str) -> str: - Id: the app's chat/thread id, or `AgentSession.session_id` (create sessions with `agent.create_session(session_id=chat_id)` when the app has an id). Must be stable across all turns of one conversation and differ between conversations. - Streaming: the whole `async for update in agent.run(..., stream=True)` loop goes inside the `with`. - Approval resumes (`request.to_function_approval_response(...)` passed back to `agent.run`) and workflow runs go inside the same `with`. 1.19 logs a WARN "Ignored an approval response ... did not match" on each resume even though the tool runs; ignore it if the `execute_tool` span is there once. -- Build workflows (`WorkflowBuilder(...).build()`, `SequentialBuilder`, `ConcurrentBuilder`, ...) inside the `with`. `build()` always emits a separate one-span `workflow.build` trace: inside the `with` it joins the session; outside it becomes a stray one-span session. Executors (fan-out included) inherit the id. +- Build workflows (`WorkflowBuilder(...).build()`, `SequentialBuilder`, `ConcurrentBuilder`, ...) inside the `with`. `build()` always emits a separate one-span `workflow.build` trace: inside the `with` it joins the session (Maple shows it as an empty, unlabeled turn 1, the run is turn 2, and the session has no title); outside it becomes a stray one-span session. Executors (fan-out included) inherit the id. - .NET: `ConversationIdProcessor.Current.Value = chatId;` in the request handler before `RunAsync`/`RunStreamingAsync`. ## Step 4: Content diff --git a/skills/maple-agent-tracing-openai-agents/SKILL.md b/skills/maple-agent-tracing-openai-agents/SKILL.md index 04d6d9c1f1..f85747bec5 100644 --- a/skills/maple-agent-tracing-openai-agents/SKILL.md +++ b/skills/maple-agent-tracing-openai-agents/SKILL.md @@ -140,7 +140,7 @@ Without it the SDK sends no `stream_options` for non-OpenAI clients and every `r ### 2e. TypeScript (`@openai/agents`) only -Verified with `@openai/agents` 0.18.0, `@arizeai/openinference-instrumentation-openai-agents` 0.2.15, `@opentelemetry/sdk-trace-node` 2.11. Use a `NodeTracerProvider({ resource: detectResources({ detectors: [envDetector] }), spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())] })` (`@opentelemetry/resources`, `@opentelemetry/exporter-trace-otlp-proto`; the 2.x provider does NOT read `OTEL_SERVICE_NAME` without `envDetector`, you get `unknown_service:node`), `provider.register()`, then `new OpenAIAgentsInstrumentation({ tracerProvider: provider }).manuallyInstrument(agents)`. Session: wrap each run in `context.with(setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }), () => agents.run(agent, text))` (`setAttributes` from `@arizeai/openinference-core`). Tell the user the limits: framework shows as Unidentified, model replies render as the raw API response JSON (inputs render as messages), no tool arguments/results fields, no agent lanes. `await provider.forceFlush()` before a script exits. Maple ignores `session.id`/`setSession` for this scope. +Verified with `@openai/agents` 0.18.0, `@arizeai/openinference-instrumentation-openai-agents` 0.2.15, `@opentelemetry/sdk-trace-node` 2.11. Use a `NodeTracerProvider({ resource: detectResources({ detectors: [envDetector] }), spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())] })` (`@opentelemetry/resources`, `@opentelemetry/exporter-trace-otlp-proto`; the 2.x provider does NOT read `OTEL_SERVICE_NAME` without `envDetector`, you get `unknown_service:node`), `provider.register()`, then `new OpenAIAgentsInstrumentation({ tracerProvider: provider }).manuallyInstrument(agents)`. Session: wrap each run in `context.with(setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }), () => agents.run(agent, text))` (`setAttributes` from `@arizeai/openinference-core`). Tell the user the limits: framework shows as Unidentified, inputs render as messages but model replies and the model's tool-call requests are MISSING from each call (`output.value` is the raw `chat.completion` response, which Maple can't decode; earlier replies only show as history in the next call's input, so each turn's final reply is absent), no tool arguments/results fields, no agent lanes, the reply-length check is skipped (no finish reason read). `await provider.forceFlush()` before a script exits. Maple ignores `session.id`/`setSession` for this scope. ## Step 3: Session id (one conversation = one session) @@ -174,7 +174,7 @@ async def handle_message(conversation_id: str, text: str) -> str: ## Step 5: Tools, errors, sub-agents -- Function tools: span named after the tool, `execute_tool`, `gen_ai.tool.name`, description, arguments (via `MapleSpanFixes`), result. A tool that RAISES: the SDK catches it and the span gets status ERROR with message `Error running tool (non-fatal): {...}`; Maple counts it failed. A tool that RETURNS an error string counts as success; point it out, don't change behavior unasked. +- Function tools: span named after the tool, `execute_tool`, `gen_ai.tool.name`, description, arguments (via `MapleSpanFixes`), result. A tool that RAISES: the SDK catches it and the span gets status ERROR with message `Error running tool (non-fatal): {...}`; Maple counts it failed. A tool that RETURNS an error string counts as success; point it out, don't change behavior unasked. Maple shows `gen_ai.tool.call.result` on the tool span only when it is JSON (dict/list return); plain-text results (string returns, agents as tools, handoffs, failed tools' error text) read as "not captured" on the tool span but still appear in the transcript as the tool message of the next model call. Maple-side gap; don't work around it. - Give every `Agent` a distinct `name=`. Agent spans carry `gen_ai.agent.name`; Maple draws one lane per name. - Agents as tools (`agent.as_tool(...)`): nested run appears inside the calling tool span. Nothing to add. - Handoffs: a `handoff to ` span (counted as a tool call named `transfer_to_` via `MapleSpanFixes`) and the target agent span as a sibling of the source agent's. Nothing to add. @@ -209,7 +209,7 @@ Run one real conversation: 2-3 turns with the same conversation id including one - [ ] One turn per `Runner.run`; turn labels are the user messages; root span named after `workflow_name`. - [ ] Transcript shows instructions, user and assistant messages, tool calls, including the streamed turn's reply (missing there = `MapleSpanFixes` absent or not first). Empty transcript with non-zero tokens = `enable_genai_semconv` not active. - [ ] Model calls (`generation` for Chat Completions, `response` for Responses API) show the model and non-zero input/output tokens, INCLUDING the streamed turn (zero there = Step 2d missing). LLM call count equals real model calls (higher = `chat`/`completion` in `workflow_name`). -- [ ] Each tool call appears once, with its real name and real arguments (not a JSON schema); a raised tool error is marked failed and nothing else is. +- [ ] Each tool call appears once, with its real name and real arguments (not a JSON schema); a raised tool error is marked failed (verdict: **Tool availability** failed for it) and nothing else is. Only `needs_approval` tools appear twice (pause + resume). - [ ] Multi-agent: one lane per agent name; all sub-agent spans in one trace with the same session. - [ ] Each model call appears once (no second `ChatCompletion`-style span from another instrumentor). - [ ] Cost shows "unpriced" (expected; nothing emits cost). diff --git a/skills/maple-agent-tracing-openrouter/SKILL.md b/skills/maple-agent-tracing-openrouter/SKILL.md index 1f16f57187..b5bbaa7965 100644 --- a/skills/maple-agent-tracing-openrouter/SKILL.md +++ b/skills/maple-agent-tracing-openrouter/SKILL.md @@ -136,7 +136,7 @@ Print these, filled in (you cannot apply them): - Data regions: include Europe if the app calls `eu.openrouter.ai`. - Privacy Mode: off unless content must not leave OpenRouter. - Leave **Additional generation metadata → Cost** off; Maple reads `gen_ai.usage.total_cost`, which is sent anyway. -3. Click **Test Connection**; it only saves if the test passes. The test creates a sessionless `openrouter-connection-test` trace (`trace:0000…0001` in Maple); ignore it. +3. Click **Test Connection**; it only saves if the test passes. The test creates a sessionless `openrouter-connection-test` trace (`trace:0000…0001` in Maple, 0 model calls, one span per test click); ignore it. ## Step 8: Verify @@ -151,7 +151,7 @@ Run one conversation (3+ turns, one with a tool call) with a fixed `session_id`, - Tool counts from Broadcast are 0 (expected). Tool spans only come from app instrumentation. - No attribute contains `sk-or-`, `Bearer ` or `maple_sk_`. -Tell the user: cost is shown (OpenRouter's charge); cache writes, TTFT, environment, tool calls and agent lanes are not available from Broadcast; transcript is raw JSON. +Tell the user: cost is shown (OpenRouter's charge); cache writes, TTFT, environment, tool calls and agent lanes are not available from Broadcast; transcript is raw JSON and Broadcast-only turns are unlabeled segments. Claude models (`gen_ai.provider.name=anthropic`): Maple applies Anthropic's input-excludes-cache rule to OpenRouter's cache-inclusive input, so cache reads count twice in token totals (cost unaffected). ## Do not diff --git a/skills/maple-agent-tracing-provider-sdks/SKILL.md b/skills/maple-agent-tracing-provider-sdks/SKILL.md index 8121733b8f..85c42c7804 100644 --- a/skills/maple-agent-tracing-provider-sdks/SKILL.md +++ b/skills/maple-agent-tracing-provider-sdks/SKILL.md @@ -114,7 +114,7 @@ Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExpor ## Known limitations (tell the user, don't work around) - Framework facet shows "Unidentified". -- Python Anthropic with prompt caching: the instrumentation reports `input_tokens` = raw + cache reads + cache writes, Maple treats Anthropic input as excluding cache → both cache buckets counted twice in totals. +- Python Anthropic with prompt caching: the instrumentation reports `input_tokens` = raw + cache reads + cache writes, Maple treats Anthropic input as excluding cache → both cache buckets counted twice in totals (verified in Maple: 3-turn session, 40,239 real tokens shown as 78,144). - Python: no cost (unpriced). - Anthropic instrumentation 1.2b0 records no time to first chunk for `messages.stream()`. - Gemini path (Python instrumentation, TS mapping) not yet run end to end against a live model; OpenAI and Anthropic paths are verified. diff --git a/skills/maple-agent-tracing-smolagents/SKILL.md b/skills/maple-agent-tracing-smolagents/SKILL.md index 75c5f5ca88..ba72fd8281 100644 --- a/skills/maple-agent-tracing-smolagents/SKILL.md +++ b/skills/maple-agent-tracing-smolagents/SKILL.md @@ -141,7 +141,7 @@ def handle_message(conversation_id: str, text: str) -> str: - smolagents prompts the model to retry after every tool error until `max_steps`, then asks for an answer without tools (hallucination risk). If a tool can fail permanently, suggest a `step_callbacks` hook that tells the model to stop after the first failure; don't add it unasked. - Managed agents appear as `.run` spans under the manager's `Step N` span; the processor names their lanes. Parallel tool/agent calls (`max_tool_threads`) keep context; nothing to do. - `CodeAgent` with `executor_type` other than local (`e2b`, `docker`, `modal`, ...) runs tools outside the process: no tool spans. Tell the user; nothing to fix in tracing. -- Known, not fixable here: no `gen_ai.tool.call.id` on tool spans; a failed tool also marks its `Step N` span ERROR. +- Known, not fixable here: no `gen_ai.tool.call.id` on tool spans; a failed tool also marks its `Step N` span ERROR (Maple: toolErrorCount 1, plus an "Other errors" warning for the Step span). ## Step 6: Flush @@ -155,10 +155,10 @@ Run one real conversation (2-3 messages, same conversation id, at least one tool - Exactly one session per conversation id (two here), not one per message. Framework shows **smolagents**. - The first conversation has one turn per `agent.run()`; each turn's root span is `.run`. -- Transcript is non-empty (user text appears after `New task:`; tool results as `tool-response` messages). +- Transcript is non-empty (user text appears after `New task:`; tool results as `tool-response` messages). Turn labels and the session title read `New task:` (Maple uses the user message's first line). Expected; tell the user. - Model calls `OpenAIModel.generate` (or `.generate[_stream]`) have a model and non-zero input/output tokens, including streamed calls. - Session token totals equal the sum of the model calls (not 2x, not growing faster each turn). -- Tool calls are `execute_tool ` with JSON arguments (the actual call's, not a schema) and results. `execute_tool final_answer` at the end of each turn is expected. +- Tool calls are `execute_tool ` with JSON arguments (the actual call's, not a schema) and results. `execute_tool final_answer` at the end of each turn is expected, and Maple's tool-call count includes one per agent run (managed agents too). - A tool that raised is counted as failed; tools that returned are not. - Managed agents (if any) each have their own lane named after the agent. - Cost shows as unpriced (smolagents records no cost). Expected. diff --git a/skills/maple-agent-tracing-strands/SKILL.md b/skills/maple-agent-tracing-strands/SKILL.md index 3ca84360fb..f7d5f37f93 100644 --- a/skills/maple-agent-tracing-strands/SKILL.md +++ b/skills/maple-agent-tracing-strands/SKILL.md @@ -160,7 +160,8 @@ Run one short conversation (2-3 messages, one tool call; plus a failing tool if - One turn per agent call, with the user's message as the turn label. - Transcript shows user messages, replies, tool calls with args and results. Empty transcript → `gen_ai_span_attributes_only` missing or set after the first `Agent(`. - Spans: `invoke_agent ` → `execute_event_loop_cycle` → `chat` / `execute_tool `; model id on `chat` spans. -- Input/output tokens on every `chat` span, including streamed turns. Session total ≈ sum of `chat` spans, not several times more. +- Input/output tokens on every `chat` span, including streamed turns. Session total on the session detail page ≈ sum of `chat` spans, not several times more. The Agent Sessions LIST currently shows ~2x for Strands (Python and TS) even when setup is correct (Maple nets roll-ups only one level deep; `execute_event_loop_cycle` sits between `invoke_agent` and `chat`). Tell the user; don't try to fix it in their code. +- "Reply length" check shows skipped (finish reason only inside output messages). Expected. - Failed tool counted as failed; successful tools not. - Sub-agents in separate lanes with their own names. - Cost shows "unpriced" (Strands emits no cost). Expected. diff --git a/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md index 877670800a..5668fc76f8 100644 --- a/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md +++ b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md @@ -13,7 +13,7 @@ One conversation = one Maple Agent Session, one turn per `generate()`/`stream()` How it works: AI SDK 7 emits GenAI-semconv spans through `@ai-sdk/otel` (`invoke_agent ` → `step ` → `chat ` + `execute_tool `) on tracer `gen_ai`, once `registerTelemetry(new OpenTelemetry())` has run. Content is on by default. Maple detects the AI SDK by `ai.*` attributes on the `gen_ai`/`ai` scope and groups sessions by `gen_ai.conversation.id`, which the AI SDK never sets. You add it with `enrichSpan` from `runtimeContext`. -Known gaps (tell the user, don't try to fix): cost shows as "unpriced" (AI SDK emits no cost; Maple never prices tokens); on AI SDK 5/6 the final assistant reply is missing from transcripts. +Known gaps (tell the user, don't try to fix): cost shows as "unpriced" (AI SDK emits no cost; Maple never prices tokens); the session token total (list and detail page) is 2x the real usage, because Maple doesn't net the `invoke_agent` total against its `chat` spans two levels down (per-`chat` counts and the per-model breakdown are correct); on AI SDK 5/6 the final assistant reply is missing from transcripts. ## Step 0: Detect @@ -189,7 +189,7 @@ Run one real conversation: 2+ turns with the same id, one streamed, one tool cal - Framework shows **Vercel AI SDK**, not Unidentified. - Turns = number of `generate`/`stream` calls (an approval pause + resume = 2 turns); each has `invoke_agent `, `step `, `chat `, `execute_tool ` spans in one trace. - Transcript shows user messages, assistant replies and tool calls; turn labels are the user's messages. -- Every `chat` span has input and output tokens, including the streamed turn; the streamed `chat` span has TTFT. +- Every `chat` span has input and output tokens, including the streamed turn; the streamed `chat` span has TTFT. The session token total is 2x the sum of the `chat` spans (known Maple gap, not a setup error; don't try to fix it); the per-model breakdown matches the `chat` spans. - Tool calls have name, arguments, result; a throwing tool is counted as failed with its message; successful tools are not failed. - Sub-agents show as their own lanes named after their `functionId`. - No attribute contains the provider API key or `Bearer `. From 91834eee9e950ad5be156293d51096622ce75bd7 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 14:13:20 +0200 Subject: [PATCH 05/39] docs(agent-tracing): cut the human guides to a 5-minute setup, move detail into the skills --- .../content/docs/agent-sessions/overview.md | 87 +-- .../src/content/docs/agent-tracing.mdx | 31 +- .../src/content/docs/agent-tracing/agno.md | 206 +---- .../docs/agent-tracing/claude-agent-sdk.md | 213 +---- .../src/content/docs/agent-tracing/crewai.md | 173 +---- .../src/content/docs/agent-tracing/dspy.md | 160 +--- .../content/docs/agent-tracing/google-adk.md | 147 +--- .../content/docs/agent-tracing/haystack.md | 162 +--- .../content/docs/agent-tracing/langchain.md | 181 +---- .../src/content/docs/agent-tracing/litellm.md | 244 +----- .../content/docs/agent-tracing/llamaindex.md | 213 +---- .../src/content/docs/agent-tracing/mastra.md | 248 +----- .../microsoft-agent-framework.md | 200 +---- .../docs/agent-tracing/openai-agents.md | 199 +---- .../content/docs/agent-tracing/openrouter.md | 169 +--- .../docs/agent-tracing/opentelemetry.md | 735 ++---------------- .../docs/agent-tracing/provider-sdks.md | 211 ++--- .../content/docs/agent-tracing/pydantic-ai.md | 183 +---- .../content/docs/agent-tracing/smolagents.md | 122 +-- .../content/docs/agent-tracing/spring-ai.md | 251 +----- .../src/content/docs/agent-tracing/strands.md | 193 +---- .../docs/agent-tracing/vercel-ai-sdk.md | 271 +------ skills/maple-agent-tracing-agno/SKILL.md | 14 + .../SKILL.md | 15 + skills/maple-agent-tracing-crewai/SKILL.md | 9 + skills/maple-agent-tracing-dspy/SKILL.md | 15 + .../maple-agent-tracing-google-adk/SKILL.md | 22 + skills/maple-agent-tracing-haystack/SKILL.md | 32 + skills/maple-agent-tracing-langchain/SKILL.md | 12 + skills/maple-agent-tracing-litellm/SKILL.md | 17 +- .../maple-agent-tracing-llamaindex/SKILL.md | 12 + skills/maple-agent-tracing-mastra/SKILL.md | 18 +- .../SKILL.md | 27 + .../SKILL.md | 48 +- .../maple-agent-tracing-openrouter/SKILL.md | 18 + .../SKILL.md | 31 +- .../SKILL.md | 14 + .../maple-agent-tracing-pydantic-ai/SKILL.md | 16 + .../maple-agent-tracing-smolagents/SKILL.md | 11 + skills/maple-agent-tracing-spring-ai/SKILL.md | 47 +- skills/maple-agent-tracing-strands/SKILL.md | 9 + .../SKILL.md | 9 +- 42 files changed, 1062 insertions(+), 3933 deletions(-) diff --git a/apps/landing/src/content/docs/agent-sessions/overview.md b/apps/landing/src/content/docs/agent-sessions/overview.md index 9856f3f868..082f5320a1 100644 --- a/apps/landing/src/content/docs/agent-sessions/overview.md +++ b/apps/landing/src/content/docs/agent-sessions/overview.md @@ -1,18 +1,14 @@ --- title: "Agent Sessions" -description: "See an AI agent conversation as one session: every turn, model call and tool call, with its tokens, cost, timing and failures, built from the OpenTelemetry traces your agent already sends." +description: "Agent Sessions groups the OpenTelemetry traces of one AI agent conversation into a single view of its turns, model calls, tool calls, tokens, cost and failures." group: "Agent Sessions" order: 1 navLabel: "Overview" --- -A trace shows you one request. An agent conversation is rarely one request: a chat backend handles each user message separately, so a ten-message conversation is ten traces, and the model calls, tool calls and handoffs that matter are scattered across them. +A chat backend handles each user message as its own request, so a ten-message conversation is ten traces. **Agent Sessions** groups those traces back into one conversation and shows it turn by turn. To send your agent's traces, pick your framework in [Trace your AI agent](/docs/agent-tracing). -**Agent Sessions** puts them back together. Maple groups the traces of one conversation into a session and shows it the way you think about it: turn by turn, with the transcript, every model and tool call, what it cost, where the time went and what failed. Sessions are built from OpenTelemetry traces, so there is no Maple SDK to add. To get your agent's traces in, pick your framework in [Trace your AI agent](/docs/agent-tracing). - -## What is an agent session? - -A session is one conversation between a user (or a job) and your agent. Maple uses four levels: +## Sessions, turns and calls | Level | What it is | Where it comes from | | --- | --- | --- | @@ -21,56 +17,47 @@ A session is one conversation between a user (or a job) and your agent. Maple us | **Model call** | One request to an LLM: the prompt, the reply, tokens, finish reason. | A `chat` (or `generate_content`, `text_completion`) span. | | **Tool call** | One function the model asked to run, with its arguments and result. | An `execute_tool` span. | -Sub-agents sit inside a turn. When an orchestrator hands work to a `researcher` agent, the researcher's model and tool calls show up as their own lane, labeled with its `gen_ai.agent.name`. +Sub-agents show up inside a turn as their own lane, labeled with their `gen_ai.agent.name`. A background job with no user is also a session, usually one trace long. -A background job that never talks to a user is still a session: one run, usually one trace. +## Read a session -## What a session shows you - -This is one conversation with a support agent that uses the OpenTelemetry GenAI conventions. The customer asks to change a delivery address, gives one in Paris, and ends up canceling the order. +The example below is a support agent where the customer asks to change a delivery address and ends up canceling the order.
A session's overview page: a time breakdown bar, a findings list with a failed tool call, a tools table with calls, failures and a timeline, and a right column with cost by model and token buckets.
The overview. The failed tool call leads the page: update_shipping_address returned unsupported_destination in turn 2.
-The **overview** splits the wall clock into model time, tool time and idle time, and rolls cost and tokens up per model. Here the agent was busy for 14 seconds of a 2 minute 25 second session. The rest was the customer typing. - -Under it, Maple runs a set of checks over the session and gives it a verdict: it completed cleanly, completed with warnings, or failed, and if it failed, which span ended it. The checks that need attention come first, each with what to do about it: +The **overview** splits the session's wall clock into model time, tool time and idle time, and totals cost and tokens per model. Here the agent was busy for 14 seconds of a 2 minute 25 second session; the rest was the customer typing. -- **Completion**, **Context window**, **Reply length** and **Refusals**: did the model finish its answers, or did it run out of context, hit `max_tokens` or refuse? -- **Rate limits** and **Provider errors**: did the LLM provider fail calls, and did the session survive them? -- **Tool errors**, **Tool arguments**, **Tool timeouts** and **Tool availability**: which tools failed, and was it the tool or the arguments the model sent? -- **Repeated calls** and **Stalls**: is the agent looping, or did it stop making progress? -- **Prompt cache**: is the prompt prefix being reused across calls, or are you paying full price every turn? -- **Structured output**: did the model's JSON parse? - -When your instrumentation doesn't record something a check needs, like message content or tool arguments, the check says it was skipped and what to capture, instead of passing quietly. +Below that is the verdict (completed, completed with warnings, or failed, with the span that ended it) and the checks behind it, failing ones first. The checks cover completion, context window, reply length, refusals, rate limits, provider errors, tool errors, tool arguments, tool timeouts, tool availability, repeated calls, stalls, prompt cache and structured output. A check that needs data your instrumentation doesn't record says it was skipped and what to capture.
The transcript view of a session: system instructions, user and assistant messages in sequence, each model call annotated with its model, tokens, cost and finish reason, and a tool call row with its latency and payload sizes.
The transcript. Each model call carries its model, tokens, cost and finish reason; tool calls sit where the model made them.
-The **transcript** is the conversation as the model saw it: system instructions, user and assistant messages, and tool calls with their arguments and results, in order. It needs message content on your spans, which most instrumentations leave off by default. Each framework guide shows the switch. +The **transcript** is the conversation as the model saw it: system instructions, user and assistant messages, and tool calls with arguments and results. It needs message content on your spans, which most instrumentations leave off by default. Each framework guide shows the switch.
The trace view of a session: three turns, each with an invoke_agent span, chat spans labeled with their model and token counts, and execute_tool spans, on a time axis with the idle gaps between turns removed. One tool span is marked with its error type.
The trace. Spans grouped by turn, with 2 minutes 10 seconds of idle time cut from the axis and the failed tool span flagged.
-The **trace** view is every span of every turn on one time axis, with the idle gaps between turns removed so a slow tool doesn't hide next to a user who went for coffee. +The **trace** view puts every span of every turn on one time axis, with the idle time between turns removed.
The Agent Sessions list in Maple, one row per session with services, model, duration, LLM call and tool call counts, tokens, cost, errors and start time, and a filter sidebar on the left.
The list. One row per conversation, filterable by framework, service, environment, model, agent and tool.
-The **list** has one row per session with its model, duration, call counts, tokens, cost and errors. Sort by cost to find the expensive conversations, or filter to sessions that called a given tool. +The **list** has one row per session with its model, duration, call counts, tokens, cost and errors. Sort by cost to find expensive conversations, or filter to sessions that called a given tool. + +A session view loads up to 2,000 spans. Longer sessions show the first 2,000 and say so. -## Debug the tools your agent calls +## Find the tools that fail -When an agent misbehaves, the cause is usually a tool: it failed, it was slow, or the model called it with arguments it couldn't use. The **Tools** tab looks at every tool call across all sessions. +The **Tools** tab covers every tool call across all sessions.
The Tools tab: a metric strip with tool calls, sessions, error rate and duration, a chart of calls over time, and a table ranking each tool by calls, p50, p90, p95, error rate, errors, sessions and last call. @@ -87,46 +74,18 @@ When an agent misbehaves, the cause is usually a tool: it failed, it was slow, o
What exactly happened? An error group opens on the failed calls themselves: arguments on the left, the result on the right, and a link to the trace.
-Failures are grouped by what went wrong: Maple fingerprints the failed result (or the span's status message) with ids and numbers masked, so a thousand `order 12345 not found` errors are one group, not a thousand. A tool call counts as failed when its span has an `ERROR` status or an `error.type` attribute. Many frameworks catch a tool's exception and hand the message back to the model without marking the span, which makes a broken tool look healthy; the framework guides cover how to avoid that. - -## How Maple builds a session from your traces - -You don't need this to use Agent Sessions, but it explains what the framework guides ask you to configure. - -1. **Maple recognizes AI spans at ingest.** A span counts if it carries `gen_ai.operation.name` (the OpenTelemetry GenAI conventions) or matches the fingerprint of a framework Maple knows: Vercel AI SDK, OpenAI Agents SDK, LangChain and LangGraph, Mastra, Pydantic AI, CrewAI, Google ADK, Strands, Claude Code, Spring AI and others. Traces with no AI span don't appear. -2. **It reads the session id that framework uses.** For most frameworks that's `gen_ai.conversation.id`; several, including CrewAI, DSPy and Strands, use `session.id`. One span per trace is enough. A trace without one becomes a session of its own, which is why "every message is its own session" is the most common setup problem. -3. **It splits the session into turns**, normally one per trace, and decodes each model call's model, tokens and content, and each tool call's name, arguments and result. -4. **It counts tokens once.** Providers disagree on whether cached and reasoning tokens are included in the input and output counts. Maple resolves that per provider, and when a framework records usage on both an agent span and the model calls inside it, Maple keeps the model calls' numbers. - -Two limits are worth knowing up front: - -- **Maple reads span attributes.** Prompts and replies that a framework writes only to span events or OpenTelemetry logs are still stored (on the span, or under [Logs](/docs/explore/logs)), but they don't show up in the transcript. The framework guides say where each framework puts its content and how to move it onto spans when that's possible. -- **Maple shows cost your instrumentation reports; it doesn't price tokens itself.** Cost appears when spans carry `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or OpenInference's `llm.cost.total`. OpenRouter and some instrumentations send it. Otherwise the session shows tokens and reads as unpriced. - -A session view loads up to 2,000 spans. Longer sessions show the first 2,000 and say they were cut. - -## Get your agent into Agent Sessions - -Maple ingests OpenTelemetry over OTLP/HTTP, so the work is turning on your framework's tracing and pointing it at `https://ingest.maple.dev` (`https://ingest.eu.maple.dev` for EU organizations) with an ingest key from **Settings → Ingestion**. [Trace your AI agent](/docs/agent-tracing) has a guide for each framework, and a prompt that has a coding agent do the setup for you. - -The guides all aim for the same result: - -- one session per conversation, not one per message; -- the prompt, reply and tool arguments and results in the transcript; -- tokens on every model call, including streamed ones; -- failed tools marked as failed; -- sub-agents named, so they get their own lanes. +A tool call counts as failed when its span has an `ERROR` status or an `error.type` attribute. Failures with the same message are grouped, with ids and numbers masked, so a thousand `order 12345 not found` errors form one group. ## Query sessions from your coding agent -The same data is on the [MCP server](/docs/reference/mcp): `list_agent_sessions` finds sessions by cost, model, tool or failure, `get_agent_session` returns a session's verdict, checks and turns, and `get_agent_tools_overview` and `get_agent_tool_error` return the tool rankings and failure groups. Ask your coding agent "why did the last expensive session fail?" and it can answer from the traces. +The [MCP server](/docs/reference/mcp) exposes the same data. `list_agent_sessions` finds sessions by cost, model, tool or failure, `get_agent_session` returns a session's verdict, checks and turns, and `get_agent_tools_overview` and `get_agent_tool_error` return the tool rankings and failure groups. ## When a session doesn't look right -- **Every message is its own session.** No span in the trace carried a session id Maple reads for that framework. The framework's guide shows where to set it. -- **The transcript is empty.** Message content isn't on the spans: either capture is off (the default for most instrumentations), or the framework writes content to span events or logs only. Content that is on the spans must be a JSON string, such as a `[{role, parts}]` array; a plain string is ignored. -- **The framework shows as "Unidentified".** The spans follow the GenAI conventions but match no framework fingerprint. Sessions, transcripts and tools all work; only the framework facet is missing. Tell us which framework it is. -- **Token totals look doubled.** Two instrumentations recorded the same model call, typically the framework's and a provider SDK instrumentor like OpenAI's. Turn one off. -- **Nothing appears at all.** Check that ordinary traces from the service show up under **Explore → Traces** first. If they don't, the exporter isn't reaching Maple; a short-lived script that exits before flushing is the usual cause. If they do, none of the spans is recognized as AI; check the framework's tracing is actually on. +- **Every message is its own session.** No span carried a session id Maple reads (`gen_ai.conversation.id` for most frameworks, `session.id` for some). The framework's guide shows where to set it. +- **The transcript is empty.** Content capture is off, or the framework writes content only to span events or logs. Content on span attributes must be a JSON string such as a `[{role, parts}]` array. +- **Cost shows as unpriced.** Maple shows cost only when spans carry `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or `llm.cost.total`; it doesn't price tokens itself. +- **Token totals look doubled.** Two instrumentations recorded the same model call, usually the framework's and a provider SDK instrumentor. Turn one off. +- **Nothing appears at all.** Check **Explore → Traces** for the service first. No traces there means the exporter isn't reaching Maple, often a short-lived script that exits before flushing. Traces there but no session means the framework's tracing isn't on. -Using a framework we don't cover? Send us the framework and a sample trace at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP). Adding one is a detection rule on our side, not a new SDK. Until then, [the OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry) works for any agent in any language. +A framework shown as **Unidentified** still gets sessions, transcripts and tools. If yours isn't covered, send the framework and a sample trace to [support@maple.dev](mailto:support@maple.dev) or [Discord](https://discord.gg/BnXjKuwJqP). Until then, [the OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry) works for any agent in any language. diff --git a/apps/landing/src/content/docs/agent-tracing.mdx b/apps/landing/src/content/docs/agent-tracing.mdx index 0556c16cfa..d82b628b57 100644 --- a/apps/landing/src/content/docs/agent-tracing.mdx +++ b/apps/landing/src/content/docs/agent-tracing.mdx @@ -1,6 +1,6 @@ --- title: "Trace your AI agent" -description: "Setup guides for every agent framework and LLM gateway Maple supports, so each conversation shows up as one Agent Session with its transcript, tool calls, tokens and cost." +description: "Setup guides for sending each supported agent framework and LLM gateway to Maple as one Agent Session per conversation." group: "AI Agents" order: 0 navLabel: "Overview" @@ -9,13 +9,11 @@ navLabel: "Overview" import GuideGrid from "../../components/docs/GuideGrid.astro" import { AGENT_GUIDE_SECTIONS } from "../../lib/agent-tracing-guides" -Most agent frameworks can export OpenTelemetry traces, and almost none of them export good ones by default. The usual result of turning tracing on is a list of disconnected traces, one per user message, with no prompts in them, and a tool that failed forty times marked green. - -Each guide below gets a framework from there to a proper [Agent Session](/docs/agent-sessions/overview): one session per conversation, the transcript, every model and tool call with its tokens, failures marked as failures, and sub-agents in their own lanes. Maple ingests OTLP directly, so there is no Maple SDK to add. You turn on the framework's own instrumentation and point it at Maple. +Each guide below sets up one framework so every conversation shows up in Maple as one [Agent Session](/docs/agent-sessions/overview), with its transcript, model and tool calls, tokens and failures. Maple ingests OTLP directly, so you turn on the framework's own OpenTelemetry instrumentation and point it at Maple. There is no Maple SDK. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) skill, which detects your framework and installs the skill for it, so the agent only reads the steps for your stack. +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) skill, which detects your framework and installs the matching skill. ```text Set up Maple agent tracing in this project. @@ -25,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. Each guide below has the same prompt for its own framework. +Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. ## Choose your framework @@ -36,22 +34,11 @@ Use your key from **Settings → Ingestion**. Without one, the agent uses a plac ))} -If you use a gateway like OpenRouter or LiteLLM **and** a framework, pick one of the two for model calls. Instrumenting both records every call twice. Maple can merge the two copies only when both carry the same response id, and many frameworks don't record one, so treat the framework guide as the source of truth: it's the one that gives you turns, tools and sub-agents. The [OpenRouter guide](/docs/agent-tracing/openrouter) covers the one setup where both work together. - -## What every guide sets up - -The frameworks differ, the goal doesn't. A finished setup has: - -- **One session per conversation.** Chat backends handle each message as its own request, so each message is its own trace. A session id on the spans (`gen_ai.conversation.id` for most frameworks) ties them together. Almost no framework sets one by default, and without it every message is a separate session. -- **The transcript.** Prompts, replies and tool arguments and results, recorded on the spans. Most instrumentations leave content off by default for privacy, and some write it only to logs, which Maple doesn't read for the transcript. -- **Tokens on every model call**, streamed ones included. Streaming responses often report no usage unless you ask for it. -- **Failed tools marked as failed.** Frameworks usually catch a tool's exception and hand it back to the model, so the span looks successful unless the instrumentation marks it. -- **Named agents.** `gen_ai.agent.name` on each agent, so a handoff to a sub-agent shows up as its own lane. -- **Spans that actually arrive.** Exporters batch spans in the background. Scripts, CLIs, notebooks and serverless functions have to flush before they exit, or the last turn never reaches Maple. +If you use a gateway like OpenRouter or LiteLLM together with a framework, instrument only the framework, or every model call is recorded twice. The [OpenRouter guide](/docs/agent-tracing/openrouter) covers the one setup where both work together. ## Connection details -Every guide exports OTLP over HTTP to your organization's region with an ingest key from **Settings → Ingestion**: +Every guide exports OTLP over HTTP with an ingest key from **Settings → Ingestion**: ```bash export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" # https://ingest.eu.maple.dev for EU organizations @@ -60,10 +47,10 @@ export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" export OTEL_SERVICE_NAME="support-agent" ``` -The exporter appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. If you set the traces-only variable `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` or pass an `endpoint=` argument in code, most SDKs use it as-is, so include `/v1/traces` yourself. Maple's endpoint is OTLP over HTTP only, so an exporter left on its gRPC default fails; use `http/protobuf` or `http/json`. +Maple accepts OTLP over HTTP only, so set the protocol to `http/protobuf` if your exporter defaults to gRPC. If you pass a traces endpoint in code or through `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`, include `/v1/traces` yourself. -If your service samples traces, a sampled-out turn is a gap in the session, so keep agent traffic at 100%. See [Sampling and throughput](/docs/concepts/sampling-throughput). +Keep agent traffic at 100% sampling, or sampled-out turns leave gaps in the session. See [Sampling and throughput](/docs/concepts/sampling-throughput). ## Not listed? -Anything that emits the OpenTelemetry GenAI conventions works, in any language: [Any language](/docs/agent-tracing/opentelemetry) lists exactly what Maple reads. Frameworks instrumented with OpenInference or OpenLLMetry also land as sessions without a dedicated guide, with the framework shown as unidentified. Tell us what you run at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP) and we'll add a guide. +Anything that emits the OpenTelemetry GenAI conventions works, in any language. [Any language](/docs/agent-tracing/opentelemetry) lists what Maple reads. OpenInference and OpenLLMetry instrumentations also work, with the framework shown as **Unidentified**. Tell us what you run at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP) and we'll add a guide. diff --git a/apps/landing/src/content/docs/agent-tracing/agno.md b/apps/landing/src/content/docs/agent-tracing/agno.md index 4000ddb2ac..42279d4c52 100644 --- a/apps/landing/src/content/docs/agent-tracing/agno.md +++ b/apps/landing/src/content/docs/agent-tracing/agno.md @@ -1,17 +1,17 @@ --- title: "Trace Agno agents and teams with OpenTelemetry" -description: "Send Agno agent, team and tool spans to Maple with the OpenInference instrumentor, grouped into one Agent Session per conversation with transcripts, tokens and failed tools." +description: "Send Agno agent, team and tool spans to Maple as one Agent Session per conversation, with transcript, tokens and failed tools." group: "AI Agents" order: 26 navLabel: "Agno" icon: "agno" --- -Agno's tracing is built on OpenInference. The `openinference-instrumentation-agno` package wraps every agent and team run, every model call and every tool call, and Agno's own `setup_tracing()` uses the same instrumentor. The catch is where `setup_tracing()` sends the spans: into your AgentOS database, not to an OpenTelemetry endpoint. To get them into Maple you install the instrumentor yourself with an OTLP exporter. +`openinference-instrumentation-agno` traces every Agno agent and team run, model call and tool call. Agno's own `setup_tracing()` uses the same instrumentor but writes spans to your AgentOS database, so to reach Maple you install it yourself with an OTLP exporter. -The thing that goes wrong by default is the session id. Agno always stamps `session.id` on the run span, but when you don't pass `session_id=`, it generates one and keeps it on the `Agent` instance. A chat server with one module-level agent then puts every user's conversation into the same Maple session. +The one thing to get right is `session_id`. Without it, Agno generates an id once per `Agent` object, and a server with one shared agent puts every user's conversation into the same Maple session. -This guide covers Agno 3.0 (tested with 3.0.11) and `openinference-instrumentation-agno` 1.0.10 on Python 3.10 to 3.14. +Tested with Agno 3.0.11 and `openinference-instrumentation-agno` 1.0.10 on Python 3.10 to 3.14. ## Quick setup with a coding agent @@ -25,19 +25,15 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is under **Settings → Ingestion**. -## Install the instrumentor and export to Maple +## Install and point the exporter at Maple ```bash pip install -U "agno>=3.0" "openinference-instrumentation-agno>=1.0.10" \ opentelemetry-sdk opentelemetry-exporter-otlp-proto-http ``` -Version 1.0.8 of the instrumentor is the first that traces resumed human-in-the-loop runs, and 1.0.10 adds cost. Older versions work, with the gaps listed under [Troubleshooting](#troubleshooting). - -Configure the exporter with the standard OpenTelemetry variables: - ```bash export OTEL_SERVICE_NAME=support-agent export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production @@ -47,9 +43,11 @@ export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf export AGNO_TELEMETRY=false ``` -EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` to the endpoint itself. `AGNO_TELEMETRY=false` turns off Agno's anonymous usage pings to agno.com, which have nothing to do with your traces. +EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. `AGNO_TELEMETRY=false` only turns off Agno's anonymous usage pings. -Then create one tracer provider at startup: +## Initialize tracing + +Create one tracer provider in `tracing.py` and import it at the top of your entry point, before you build agents or an `AgentOS`: ```py # tracing.py @@ -60,8 +58,6 @@ from opentelemetry.exporter.otlp.proto.http.trace_exporter import OTLPSpanExport from opentelemetry.sdk.trace import TracerProvider from opentelemetry.sdk.trace.export import BatchSpanProcessor -# Resource comes from OTEL_SERVICE_NAME / OTEL_RESOURCE_ATTRIBUTES, -# endpoint and key from OTEL_EXPORTER_OTLP_*. provider = TracerProvider() provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) @@ -73,15 +69,11 @@ AgnoInstrumentor().instrument( ) ``` -Import `tracing` at the top of your entry point, before you build agents or an `AgentOS`. The instrumentor patches Agno's run functions and every model class in `agno.models`, so agents created later are traced without further changes. - -`enable_genai_semconv=True` is not optional for Maple. Without it the spans only carry OpenInference attributes (`llm.input_messages.0.message.content`, `llm.token_count.prompt`). The Agent Sessions list still shows tokens and models from those, but the session detail page shows no transcript. With it, each span also gets `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.agent.name`. The environment variable `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same thing if you can't change the `instrument()` call. +Keep `enable_genai_semconv=True`. Without it the sessions list still shows tokens, but the session page has no transcript. If you can't change the `instrument()` call, set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` instead. -### If you use AgentOS tracing +### Keep the AgentOS traces view -`AgentOS(tracing=True)` and `agno.tracing.setup_tracing(db=...)` install the same instrumentor with a `DatabaseSpanExporter`, which is what feeds the AgentOS traces view. Both skip their setup when a real `TracerProvider` is already registered. So if `tracing.py` runs first, AgentOS stops writing traces to its database. - -To keep both, add Agno's database exporter to your provider next to the OTLP one, with the same `db` you give AgentOS: +`AgentOS(tracing=True)` and `setup_tracing(db=...)` skip their setup once `tracing.py` has registered a provider, so the AgentOS traces view stops filling. To keep it, add Agno's database exporter to your provider, with the same `db` you give AgentOS: ```py from agno.tracing.exporter import DatabaseSpanExporter @@ -89,11 +81,9 @@ from agno.tracing.exporter import DatabaseSpanExporter provider.add_span_processor(BatchSpanProcessor(DatabaseSpanExporter(db=db))) ``` -Don't call `AgnoInstrumentor().instrument()` twice. The second call logs "Attempting to instrument while already instrumented" and does nothing, so its `config` is ignored. - -## Group each conversation into one session +## Pass session_id on every run -Every `agent.run()`, `team.run()` and their async versions start a new trace. Maple joins those traces into a session using the `session.id` attribute on the run span (`support_agent.run`). Agno sets it from the `session_id` argument, so pass your conversation id on every call: +Each `run()` or `arun()` starts a new trace. Maple joins them into a session by the `session.id` Agno puts on the run span, taken from the `session_id` argument: ```py from agno.agent import Agent @@ -123,140 +113,19 @@ async def chat_stream(conversation_id: str, user_id: str, message: str): yield event.content ``` -If you leave it out, Agno generates a UUID on the first run and assigns it to `agent.session_id`, and every later run of that `Agent` object reuses it. That's right for a script that creates one agent per conversation. It's wrong for a server that builds the agent once at import time, where every user's messages land in a single Maple session that grows forever. - -`session_id` is also how Agno finds the conversation's history in the agent's `db`, so the id you pass for tracing is the same one you need for memory. Members of a `Team` inherit the team's session id, and `continue_run()` takes `session_id=` too. - -Maple reads `session.id` for Agno spans. With `enable_genai_semconv=True` the root span also carries `gen_ai.conversation.id` with the same value, which Maple ignores for Agno. You don't need `using_session()` from `openinference.instrumentation`: Agno already sets the id on the run span, and one span per trace is enough. - -## Record prompts, responses and tool calls - -The OpenInference instrumentor records content by default. Each model span carries the full message list sent to the model (system prompt, history, tool results) and the model's reply, including tool calls. Each tool span carries the arguments and the result. With `enable_genai_semconv=True` they are written as `gen_ai.input.messages` and `gen_ai.output.messages` in the `[{role, parts}]` format Maple renders as a transcript. - -Maple reads span attributes only. The instrumentor emits no span events or OTLP logs, so nothing else needs enabling. - -To keep content out of Maple, set OpenInference's masking variables before the instrumentor starts: - -```bash -export OPENINFERENCE_HIDE_INPUT_MESSAGES=true # prompts sent to the model -export OPENINFERENCE_HIDE_OUTPUT_MESSAGES=true # model replies -export OPENINFERENCE_HIDE_INPUTS=true # run and tool inputs -export OPENINFERENCE_HIDE_OUTPUTS=true # run and tool outputs -``` - -Masked values are replaced with `__REDACTED__` before the span is exported, so they never leave your process. Sessions still show models, tokens, tools and failures, with an empty transcript. One exception: tool arguments (`tool.parameters` and `gen_ai.tool.call.arguments`) are not covered by any of these variables and are still exported. To drop them, or for finer control such as redacting emails but keeping the rest, run an OpenTelemetry Collector with a `redaction` or `transform` processor between your app and Maple. - -Tool results are recorded as `str(result)`, which is also what Agno sends back to the model. A tool that returns a `dict` shows up as a Python repr (`{'city': 'Berlin'}`), which isn't valid JSON, so Maple shows it as plain text. Return a JSON string from tools that produce structured data: - -```py -import json - -from agno.tools import tool - - -@tool -def get_weather(city: str) -> str: - """Get the current weather for a city.""" - return json.dumps({"city": city, "temperature_c": 21, "condition": "partly cloudy"}) -``` - -## Tools, errors and team members - -Each tool call is its own span, named after the function (`get_weather`), with `gen_ai.operation.name=execute_tool` and `gen_ai.tool.name`. When a tool raises, Agno catches the exception and hands the message back to the model, but the instrumentor still marks the tool span as failed with status `ERROR` and the exception text as its message. Maple counts it as a failed tool call and groups repeated failures by that message. The run span stays OK because the run itself continued. - -A tool that returns an error string instead of raising looks like a success. If you want a failure to count, raise. - -A `Team` delegates to its members through a `delegate_task_to_member` tool: - -```py -from agno.team import Team, TeamMode - -weather_worker = Agent(name="weather_worker", model=primary, tools=[get_weather]) -budget_worker = Agent(name="budget_worker", model=secondary, tools=[calculate]) -transport_worker = Agent(name="transport_worker", model=primary, tools=[fetch_transport_data]) - -team = Team( - name="travel_team", - mode=TeamMode.coordinate, - model=primary, - members=[weather_worker, budget_worker, transport_worker], -) - -await team.arun("Produce a mini briefing about Amsterdam.", session_id=conversation_id) -``` - -The whole team run is one trace. The members' runs sit directly under the leader's run, next to the `delegate_task_to_member` tool spans rather than inside them: +Use the conversation id your app already stores. Agno uses the same id to load history from the agent's `db`, so tracing and memory stay in sync. Pass it to `team.run()`, workflows and `continue_run()` as well. Team members inherit the team's id, and a resumed human-in-the-loop run shows up as a second turn in the same session. -```text -travel_team.arun invoke_agent (team leader) -├─ OpenRouter.ainvoke chat -├─ delegate_task_to_member execute_tool -├─ delegate_task_to_member execute_tool -├─ weather_worker.arun invoke_agent -│ ├─ OpenRouter.ainvoke chat -│ ├─ get_weather execute_tool -│ └─ OpenRouter.ainvoke chat -├─ transport_worker.arun invoke_agent -│ ├─ OpenRouter.ainvoke chat -│ ├─ fetch_transport_data execute_tool (ERROR) -│ └─ OpenRouter.ainvoke chat -└─ OpenRouter.ainvoke chat (leader's final answer) -``` +## Teams, failed tools and content -Maple opens a lane for each member because each member span has its own `gen_ai.agent.name`. The `delegate_task_to_member` calls show up as ordinary tool calls on the leader, with the member id and task as arguments. Give every `Agent` and `Team` a `name=`. Unnamed ones produce spans called `Agent.run` or `Team.run` with no agent name, and Maple can't give them a lane. With `team.arun()` in `coordinate` mode, members the leader calls in one step run concurrently, and their spans overlap in time. The sync `team.run()` runs them one after another. +Give every `Agent` and `Team` a `name=`. Maple opens a lane for each named team member, and unnamed ones show up as `Agent.run` or `Team.run` with no lane. -Instrumentor 1.0.10 leaves the team's span attached to the current context after `team.run()` or `team.arun()` returns. Any agent run that follows in the same thread or asyncio task becomes a child of the finished team run, in its trace. Web frameworks that handle each request in its own task or copied context, such as FastAPI, are not affected. In scripts, workers and loops, give each team run its own context: +A tool that raises is marked failed, even though Agno hands the error back to the model and the run continues. A tool that returns an error string counts as a success. -```py -import asyncio -import contextvars - -# async -result = await asyncio.create_task(team.arun(prompt, session_id=conversation_id)) - -# sync -result = contextvars.copy_context().run(team.run, prompt, session_id=conversation_id) -``` - -### Human-in-the-loop approvals - -Agno's confirmation flow pauses the run, and `continue_run()` resumes it. Pass the same `session_id` to both: - -```py -@tool(requires_confirmation=True) -def delete_file(path: str) -> str: - """Delete a file from disk. Destructive: requires user approval.""" - ... - - -response = agent.run(message, session_id=conversation_id) -if response.is_paused: - for requirement in response.active_requirements: - requirement.confirm() # or requirement.reject(note="...") - response = agent.continue_run( - run_id=response.run_id, - requirements=response.requirements, - session_id=conversation_id, - ) -``` - -The paused run and the resumed run are two traces, `support_agent.run` and `support_agent.continue_run`, both carrying the conversation's `session.id`. Maple shows them as two turns of the same session, and the approved tool call appears once, in the second. `continue_run()` needs a `db` on the agent. - -Workflows (`agno.workflow`) are instrumented too and accept `session_id=` the same way. - -## Tokens and cost - -Every model span carries input and output tokens, plus cache read and cache write tokens when the provider reports them. Streaming runs (`stream=True`) record usage as well, because Agno reads it from the final chunk. The run span has no token counts of its own, so nothing is counted twice. - -Reasoning tokens are not broken out into their own attribute, so Maple can't show them separately. Model spans also carry no `gen_ai.response.id` and no time-to-first-token, and tool spans carry no `gen_ai.tool.call.id` (the id is only inside the message parts). Maple still links tool results to calls through the conversation history, but it can't show streaming latency for Agno. - -Cost appears when the model provider returns a price with the response. OpenRouter does, and the instrumentor records it as `llm.cost.total` in USD. Maple never prices tokens itself, so calls to providers that don't return a cost, such as OpenAI or Anthropic called directly, show as unpriced. - -For now, only the **Agent Sessions** list shows that cost. The session's own page doesn't read `llm.cost.total` yet and shows the cost as not reported. Tokens, models and call counts match on both. +To keep content out of Maple, set `OPENINFERENCE_HIDE_INPUT_MESSAGES`, `OPENINFERENCE_HIDE_OUTPUT_MESSAGES`, `OPENINFERENCE_HIDE_INPUTS` and `OPENINFERENCE_HIDE_OUTPUTS` to `true` before the instrumentor starts. Tool arguments are still exported; removing them needs a `redaction` processor in an OpenTelemetry Collector. ## Flush before short-lived processes exit -`BatchSpanProcessor` exports every 5 seconds. A long-running server needs nothing extra: the SDK flushes on normal shutdown. Scripts, CLIs, notebooks, queue workers and serverless handlers need an explicit flush, or the last spans are lost: +`BatchSpanProcessor` exports every 5 seconds. Long-running servers flush on shutdown, but scripts, CLIs, notebooks, workers and serverless handlers need an explicit flush: ```py from tracing import provider @@ -268,37 +137,22 @@ finally: provider.shutdown() # scripts: flush and stop at the end of the process ``` -In a Jupyter notebook, call `provider.force_flush()` after the cell that runs the agent. On AWS Lambda and similar platforms, call `force_flush()` before returning from the handler and never `shutdown()`, since the next invocation reuses the process. +On AWS Lambda and similar platforms, call only `force_flush()`, since the next invocation reuses the process. ## Check that it works -Run one conversation of two or three turns with the same `session_id`, including one tool call. Within a minute, open **Agent Sessions** in Maple and filter by your service name. You should see: - -- **One session per conversation**, with your `session_id` as its id and **Agno** as the framework. A session called `trace:...` means the run span had no `session.id`. -- **One turn per `run()` call**, labelled with the user's message. -- **A transcript** with the system prompt, user messages, assistant replies and tool calls. An empty transcript with non-zero tokens means `enable_genai_semconv` is off. -- **Model calls** named after the Agno model class (`OpenRouter.invoke`, `OpenAIChat.ainvoke`, `Claude.invoke_stream`), with the model id (`openai/gpt-4o-mini`) as the model. -- **Tool calls** named after your functions, with arguments and results, and failed ones marked. -- **Agents** named after your `Agent(name=...)` and `Team(name=...)`, with a lane per team member. -- **Tokens** on every model call, and a cost in the sessions list if your provider returns one. The session page shows cost as not reported for Agno. +Run a conversation of two or three turns with one `session_id`, including a tool call, and open **Agent Sessions** filtered by your service name. You should see one session with your `session_id` as its id, framework **Agno**, one turn per `run()`, a transcript, and model calls such as `OpenRouter.invoke` with tokens. A session named `trace:...` means the run span had no `session.id`. -A human-in-the-loop approval shows up as two turns in the same session: the run that paused (`support_agent.run`) and the resumed run (`support_agent.continue_run`), each in its own trace. +Cost appears in the sessions list when your provider returns one (OpenRouter does). The session page shows it as not reported for Agno, and providers without prices show as **unpriced**. ## Troubleshooting -- **Nothing arrives in Maple.** The spans went to AgentOS's database instead. You called `setup_tracing()` or created `AgentOS(tracing=True)` before `tracing.py` ran, so the global provider is Agno's, and `set_tracer_provider` in `tracing.py` logged "Overriding of current TracerProvider is not allowed". Run `tracing.py` first. If only the last few runs are missing, see [Flush before short-lived processes exit](#flush-before-short-lived-processes-exit). -- **Every user's conversation is in one giant session.** The `Agent` is shared and no `session_id=` is passed, so Agno reuses its auto-generated id. Pass `session_id=` on every `run()`, `arun()` and `continue_run()`. -- **Every turn is its own session.** You pass a new id per request, often a fresh `uuid4()`. Use the conversation or thread id your app already stores. -- **Sessions list shows tokens, but the detail page has no transcript or model.** `enable_genai_semconv` is off, so the spans only have OpenInference attributes, which Maple's detail page doesn't decode for Agno yet. Pass `TraceConfig(enable_genai_semconv=True)` or set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true`, and check that `instrument()` isn't being called a second time by other code. -- **Resumed runs show up as loose model and tool calls with no session.** Your instrumentor is older than 1.0.8, which didn't wrap `continue_run()`. Upgrade to 1.0.10. -- **Team members appear as `Agent.run` with no lane.** The member has no `name=`. -- **Every model call appears twice.** A second instrumentor wraps the same client, usually `openinference-instrumentation-openai`, OpenLIT, or Phoenix's `register(auto_instrument=True)` picking up every installed OpenInference package. Keep only the Agno instrumentor. -- **An agent run lands inside the previous team run's trace.** The instrumentor leaves a finished team run's span attached to the thread or task ([agno#5573](https://github.com/agno-agi/agno/issues/5573)). Run each team run in its own context with `asyncio.create_task(...)` or `contextvars.copy_context().run(...)`, as shown in [Tools, errors and team members](#tools-errors-and-team-members). Plain agent runs, streamed or not, don't leak. -- **Tool results show as text instead of JSON.** The tool returns a `dict` or `list`, which Agno records as a Python repr. Return `json.dumps(...)`. -- **"Failed to detach context" in the logs after a streamed team run.** A known instrumentor issue with async generators ([agno#5208](https://github.com/agno-agi/agno/issues/5208)). The spans are still exported; it's log noise. -- **`SqliteDb` fails with "requires that the Python 'greenlet' library is installed".** Agno 3's SQLite store imports SQLAlchemy's asyncio support. SQLAlchemy doesn't pull in `greenlet` on every platform (Apple silicon, for one), so install `agno[sqlite]` and `greenlet`. This matters for tracing because `continue_run()` needs a `db`. -- **A failed tool shows as successful.** The tool returned an error message instead of raising. Raise an exception so the span gets status `ERROR`. -- **No cost on any session.** Your provider doesn't return prices. Maple shows the tokens and marks the session unpriced. +- **Nothing arrives in Maple.** `setup_tracing()` or `AgentOS(tracing=True)` ran before `tracing.py` and the spans went to the AgentOS database. Import `tracing` first. +- **Every user is in one giant session.** The shared `Agent` runs without `session_id=`. Pass it on every `run()`, `arun()` and `continue_run()`. +- **Every turn is its own session.** You pass a new id per request, often a fresh `uuid4()`. Use the stored conversation id. +- **Tokens in the list, no transcript on the session page.** `enable_genai_semconv` is off, or other code called `instrument()` first (the second call is ignored). +- **An agent run lands inside the previous team run's trace.** Instrumentor 1.0.10 leaves the team span attached to the thread or task. In scripts and workers, run each team run in its own context with `asyncio.create_task(...)` or `contextvars.copy_context().run(...)`. +- **Every model call appears twice.** Remove the second instrumentor, usually `openinference-instrumentation-openai`, OpenLIT or Phoenix's `register(auto_instrument=True)`. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md index 615ff64bfd..fe846738a0 100644 --- a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md @@ -1,17 +1,17 @@ --- title: "Trace Claude Agent SDK agents and Claude Code sessions with OpenTelemetry" -description: "Turn on the tracing Claude Code has built in, so each Agent SDK conversation or Claude Code session shows up in Maple as one Agent Session with its prompts, model calls, tool calls and tokens." +description: "Turn on Claude Code's built-in OpenTelemetry so each Agent SDK conversation or Claude Code session shows up in Maple as one Agent Session." group: "AI Agents" order: 14 navLabel: "Claude Agent SDK & Claude Code" icon: "claude" --- -The Claude Agent SDK has no telemetry of its own. Every `query()` starts the Claude Code CLI as a child process, and the CLI has OpenTelemetry built in: a span per user turn (`claude_code.interaction`), per model request (`claude_code.llm_request`) and per tool call (`claude_code.tool`), plus log events and metrics. Maple recognizes these spans and turns them into Agent Sessions, so there is nothing to install besides the SDK. You configure it with environment variables, the same ones whether you run the TypeScript SDK, the Python SDK or `claude` in your terminal. +Every Agent SDK `query()` starts the Claude Code CLI as a child process, and the CLI exports OpenTelemetry spans for each turn, model request and tool call. You configure it with environment variables, the same ones for the TypeScript SDK, the Python SDK and `claude` in your terminal. There is nothing else to install. -Tracing is the part that goes wrong. Spans are a beta behind `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`, and without it the CLI exports metrics and logs and not a single span, so Agent Sessions stays empty while everything looks configured. The second surprise comes later: Claude's replies and the per-request cost are only on log events, which Maple's session views don't read, so sessions show the prompts and tool calls but no assistant text and no cost. +Two things need care. Spans only exist with `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`, and in the SDK every `query()` starts a new session unless you resume the conversation's session id. -This guide covers `@anthropic-ai/claude-agent-sdk` 0.3.283 (TypeScript), `claude-agent-sdk` 0.2.160 (Python) and Claude Code 2.1.283, which both SDKs bundle. +Tested with `@anthropic-ai/claude-agent-sdk` 0.3.283 (TypeScript), `claude-agent-sdk` 0.2.160 (Python) and Claude Code 2.1.283. ## Quick setup with a coding agent @@ -25,19 +25,15 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is under **Settings → Ingestion**. EU organizations should say EU region. -## Export Claude Code telemetry to Maple +## Pass the telemetry variables to the CLI -Every setup below sets the same variables. What each one does: +Without `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` the CLI exports metrics and logs but no spans, and Agent Sessions stays empty. `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf` is also required, because Claude Code has no default protocol. EU organizations use `https://ingest.eu.maple.dev`. -- `CLAUDE_CODE_ENABLE_TELEMETRY=1` turns telemetry on. `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` turns on spans, which are what Agent Sessions are built from. -- `OTEL_TRACES_EXPORTER=otlp` sends the spans. The logs and metrics exporters are optional: they carry cost and Claude's replies, which you can search under **Logs** but which don't appear in the session views. -- `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf` is required. Claude Code has no default protocol, and Maple ingest speaks OTLP over HTTP. -- `OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev` with `OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer YOUR_INGEST_KEY`. The CLI appends `/v1/traces`, `/v1/logs` and `/v1/metrics` itself. EU organizations use `https://ingest.eu.maple.dev`. -- `OTEL_SERVICE_NAME` names the service. Without it every agent reports as `claude-code`. +The snippets below also drop any inherited `TRACEPARENT`. Claude Code's Bash tool and most CI systems set one, and without this your agent's turns nest inside that outer trace. -Never set an exporter to `console` in an SDK app. The SDK reads the CLI's standard output as its message stream, and console telemetry corrupts it. +Never set an exporter to `console` in an SDK app. The SDK reads the CLI's standard output as its message stream. ### TypeScript Agent SDK @@ -45,7 +41,7 @@ Never set an exporter to `console` in an SDK app. The SDK reads the CLI's standa npm install @anthropic-ai/claude-agent-sdk zod ``` -In TypeScript, `options.env` **replaces** the child's environment instead of adding to it. Spread `process.env` so the CLI keeps `PATH` and `ANTHROPIC_API_KEY`, and drop any inherited trace context (see [Keep each turn in its own trace](#keep-each-turn-in-its-own-trace)): +In TypeScript, `options.env` replaces the child's environment, so spread `process.env` to keep `PATH` and `ANTHROPIC_API_KEY`: ```ts // maple-env.ts @@ -67,31 +63,20 @@ export const mapleEnv: Record = { OTEL_RESOURCE_ATTRIBUTES: "deployment.environment.name=production", OTEL_TRACES_EXPORT_INTERVAL: "1000", OTEL_LOGS_EXPORT_INTERVAL: "1000", - // Content, off by default. See "Record prompts and tool calls" below. + // Content, off by default. See "Choose what content to record" below. OTEL_LOG_USER_PROMPTS: "1", OTEL_LOG_TOOL_DETAILS: "1", OTEL_LOG_TOOL_CONTENT: "1", } ``` -Pass it on every call: - -```ts -import { query } from "@anthropic-ai/claude-agent-sdk" -import { mapleEnv } from "./maple-env" - -for await (const message of query({ prompt: "What changed in the last commit?", options: { env: mapleEnv } })) { - if (message.type === "result") console.log(message.subtype === "success" ? message.result : message.subtype) -} -``` - ### Python Agent SDK ```bash pip install claude-agent-sdk ``` -In Python, `ClaudeAgentOptions.env` is merged on top of the inherited environment, so pass only the telemetry variables. Because it merges, removing an inherited `TRACEPARENT` has to happen on `os.environ`: +In Python, `ClaudeAgentOptions.env` is merged over the inherited environment, so pass only the telemetry variables and remove `TRACEPARENT` from `os.environ`: ```py # maple_env.py @@ -113,38 +98,20 @@ MAPLE_ENV = { "OTEL_RESOURCE_ATTRIBUTES": "deployment.environment.name=production", "OTEL_TRACES_EXPORT_INTERVAL": "1000", "OTEL_LOGS_EXPORT_INTERVAL": "1000", - # Content, off by default. See "Record prompts and tool calls" below. + # Content, off by default. See "Choose what content to record" below. "OTEL_LOG_USER_PROMPTS": "1", "OTEL_LOG_TOOL_DETAILS": "1", "OTEL_LOG_TOOL_CONTENT": "1", } ``` -```py -import asyncio -from claude_agent_sdk import ClaudeAgentOptions, ResultMessage, query -from maple_env import MAPLE_ENV +Pass the env on every `query()` call, as in the next section. You can also set the same variables in your Dockerfile or deployment manifest and skip `env`, as long as no `TRACEPARENT` is set there. +An `env` block in `~/.claude/settings.json` or the project's `.claude/settings.json` overrides `options.env`. A server app that doesn't need settings files can pass `settingSources: []` (Python `setting_sources=[]`). -async def main(): - async for message in query( - prompt="What changed in the last commit?", - options=ClaudeAgentOptions(env=MAPLE_ENV), - ): - if isinstance(message, ResultMessage): - print(message.result) - - -asyncio.run(main()) -``` +### Claude Code in your terminal, IDE or desktop app -Because the CLI inherits the process environment in both SDKs, you can also set these variables in your Dockerfile or deployment manifest and skip `env` entirely. That is what Anthropic recommends for production. You still need to make sure no `TRACEPARENT` is set there. - -Settings files win over `env`. Unless you pass `settingSources` (Python `setting_sources`), the CLI loads `~/.claude/settings.json` and the project's `.claude/settings.json`, and an `env` block in either overrides the same variable in `options.env`. In our test, `OTEL_SERVICE_NAME` from user settings replaced the one the app passed. That matters on a developer machine that also sends its own Claude Code sessions to Maple. A server app that doesn't need file settings can pass `settingSources: []`. - -### Claude Code in your terminal, IDE or the desktop app - -To see your own Claude Code sessions in Maple, put the variables under `env` in `~/.claude/settings.json`: +Put the variables under `env` in `~/.claude/settings.json`, then start a new `claude` session: ```json { @@ -164,19 +131,13 @@ To see your own Claude Code sessions in Maple, put the variables under `env` in } ``` -Start a new `claude` session to pick it up. Terminal sessions report `service.name=claude-code`, and sessions from the desktop app's Code tab report `claude-code-desktop`. - -It has to be your user settings, your shell, or managed settings. Since 2.1.282, Claude Code ignores the variables that turn export on, set the endpoint or capture content (`CLAUDE_CODE_ENABLE_TELEMETRY`, `OTEL_LOG_*` and the like) in a repository's `.claude/settings.json` and `.claude/settings.local.json`, so a repo can't turn export on or pick where it goes. `/status` lists any it ignored. To roll this out to a team, put the same `env` block in [managed settings](https://code.claude.com/docs/en/managed-settings). +Use your user settings, your shell or [managed settings](https://code.claude.com/docs/en/managed-settings). Since 2.1.282, Claude Code ignores these variables in a repository's `.claude/settings.json`. Terminal sessions report the service `claude-code`, and each `claude` session is one Maple session. -## Group turns into one session +## Resume the session on every turn -Maple groups Claude Code spans by their `session.id` attribute, which the CLI puts on every span. One Claude Code session becomes one Maple session, with one turn per `claude_code.interaction` span. Each turn is its own trace. +Maple groups Claude Code spans by `session.id`, which is the Claude session id. A chat backend that calls `query()` once per message without resuming gets one Maple session per message, and the agent forgets the previous message. -In the terminal that needs no setup. `/clear` starts a new session, `claude --resume` and `--continue` keep the old one, and `--fork-session` starts a new one. - -In the SDK, **every `query()` call starts a new session** unless you resume one. A chat backend that calls `query()` once per user message and doesn't resume shows every message in Maple as its own one-turn session, and the agent has no memory of the previous message either. - -Store a session UUID with each conversation. Pass it as `sessionId` on the first turn, then as `resume` on every turn after that: +Store a UUID with each conversation. Pass it as `sessionId` on the first turn and as `resume` on every turn after: ```ts import { randomUUID } from "node:crypto" @@ -216,136 +177,32 @@ async def reply(conversation: dict, text: str) -> str | None: return None ``` -The session UUID is the Maple session id, so if your conversation ids are UUIDs you can use them directly and find a conversation in Maple by its own id. - -A few things break this: - -- `resume` reads the session's transcript from `~/.claude/projects/` on the machine that ran the earlier turns. If the next message can land on another host, mirror transcripts with the SDK's [`sessionStore`](https://code.claude.com/docs/en/agent-sdk/session-storage) option. -- `forkSession: true` (Python `fork_session=True`) gives the resumed conversation a new session id, so it becomes a new Maple session. -- `OTEL_METRICS_INCLUDE_SESSION_ID=false` removes `session.id` from spans too, and every turn becomes its own session. - -A long-lived process that keeps one session open needs none of this: a TypeScript `query()` fed an async iterable of messages, or a Python `ClaudeSDKClient`, sends every turn in the same session. - -Adding `gen_ai.conversation.id` or `maple_ai.session.id` doesn't help here. You can't add attributes to the CLI's spans, and Maple reads `session.id` for Claude Code. - -### Keep each turn in its own trace - -When your application has an active OpenTelemetry span, both SDKs pass its context to the CLI as `TRACEPARENT`, and each turn's `claude_code.interaction` span becomes a child of your span. That is useful: the agent turn shows up inside the HTTP request that triggered it. - -An **inherited** `TRACEPARENT` is the problem. Claude Code sets one on every command its Bash tool runs, CI systems set them, and the SDK passes the environment through. If you ask Claude Code to run your agent script, every turn of your agent nests inside that Claude Code session's trace. The snippets above drop `TRACEPARENT` and `TRACESTATE` from the inherited environment, and the SDK still injects your own active span's context on top. - -Interactive `claude` sessions ignore an inbound `TRACEPARENT`. Only SDK and `claude -p` runs read it. - -## Record prompts and tool calls - -Claude Code redacts content by default. Each flag adds one kind: - -| Variable | What it adds to the spans | What you see in Maple | -| --- | --- | --- | -| `OTEL_LOG_USER_PROMPTS=1` | `user_prompt` on `claude_code.interaction` (otherwise ``) | Turns titled with the prompt, and the user messages in the transcript | -| `OTEL_LOG_TOOL_DETAILS=1` | `full_command` for Bash, `file_path` for Read, Edit and Write, `subagent_type` for the Agent tool, and the full error message of a failed tool | The Bash command or file path as the tool call's arguments, and the real error on failed calls | -| `OTEL_LOG_TOOL_CONTENT=1` | A `tool.output` span event with what the tool returned | The tool call's result, for Read, Bash, Edit and Write (Edit and Write also need `OTEL_LOG_TOOL_DETAILS`), and from 2.1.283 for MCP tools, including your SDK tools, WebFetch and WebSearch | - -Some content never reaches the session views, whatever you set: - -- **Claude's replies.** They are only on the `assistant_response` log event (`OTEL_LOG_ASSISTANT_RESPONSES`, which follows `OTEL_LOG_USER_PROMPTS` when unset). With the logs exporter on, you can read them under **Logs**. The transcript shows the prompts and tool calls, not the answers. -- **The system prompt.** -- **Arguments of other tools.** Maple shows the Bash command and the file path. The arguments of MCP tools, including the ones you define with the SDK, are only in the `tool_result` log event. - -Anthropic's detailed beta tracing (`ENABLE_BETA_TRACING_DETAILED` with `BETA_TRACING_ENDPOINT`) adds more content attributes to spans, but it also redirects your logs and traces to that endpoint, and Maple doesn't read the attributes it adds. Leave it off. - -### Privacy - -The flags are per category, so `OTEL_LOG_TOOL_CONTENT=1` sends whatever files Claude reads and whatever commands print, secrets included. Turn on only what your Maple organization is allowed to store. Content is truncated at 60 KB per attribute (`CLAUDE_CODE_OTEL_CONTENT_MAX_LENGTH`), with a `[TRUNCATED ...]` marker. - -Terminal sessions signed in with a Claude account also carry `user.email` on every span and event. To drop or mask attributes before they leave your network, send the data through an OpenTelemetry Collector with a `redaction` or `attributes` processor and point the Collector at Maple. - -The ingest key can only write telemetry, so it is safe in an environment variable or settings file. Keep private `maple_sk_` keys out of both. - -## Tools, errors and sub-agents - -Every `claude_code.tool` span is one tool call, named by its `tool_name`: `Bash`, `Read` and `Edit` for built-ins, `mcp____` for MCP tools and the ones you define with `createSdkMcpServer`, and `Agent` for a sub-agent. The two phases inside a tool call, `claude_code.tool.blocked_on_user` (the permission wait) and `claude_code.tool.execution` (the run), are shown in the trace but not counted as tool calls of their own. - -A tool fails when its execution does. When a tool handler throws, or returns `isError: true`, the CLI marks `claude_code.tool.execution` with `success=false`, and Maple marks the tool call failed with `error.type` set to the CLI's `error_class` (`McpToolCallError` for SDK tools) and the error as its result. Without `OTEL_LOG_TOOL_DETAILS=1` the error is only that class name; with it, it is the full message your tool threw, such as `transport data service unavailable (503)`. - -A call the user or `canUseTool` rejects has a `blocked_on_user` span with `decision=reject` and no execution. Maple shows it as a call without a result, not as a failure. - -A failed model request (`success=false` on `claude_code.llm_request`, with `status_code` and `error`) counts as a failed LLM call. - -Sub-agents, defined with the `agents` option or in `.claude/agents/`, run through the `Agent` tool. Their model and tool calls nest under that `Agent` tool call in the same trace, so a whole delegation is one turn, and parallel sub-agents show up as overlapping calls. - -Claude Code can also run sub-agents in the background. In our tests, 2.1.283 did that for SDK `agents` even though the model never asked for it. The parent turn then ends right away, and each sub-agent that finishes starts a new turn in the same session, with a `` block as its prompt. Your `query()` loop receives one `result` per turn instead of one in total, and SDK tool calls made by background sub-agents failed with "The tool call was interrupted before a result was received", which Maple counts as failed tool calls. If your app expects one answer per `query()`, keep sub-agents in the foreground: - -```ts -for await (const message of query({ - prompt, - options: { agents, env: { ...mapleEnv, CLAUDE_CODE_DISABLE_BACKGROUND_TASKS: "1" } }, -})) { - if (message.type === "result") console.log(message.subtype === "success" ? message.result : message.subtype) -} -``` - -With it, the whole delegation is one turn and one trace, and parallel sub-agents still run side by side. +`resume` reads the transcript from `~/.claude/projects/` on the machine that ran the earlier turns. If the next message can land on another host, use the SDK's [`sessionStore`](https://code.claude.com/docs/en/agent-sdk/session-storage) option. A Python `ClaudeSDKClient`, or a TypeScript `query()` fed an async iterable, keeps one session for all its turns and needs none of this. -What you don't get is a lane per sub-agent. Maple opens a lane for each distinct `gen_ai.agent.name`, and Claude Code puts no agent name on its spans, only an opaque `agent_id` and, with `OTEL_LOG_TOOL_DETAILS=1`, the `subagent_type` on the `Agent` call. The agent filter in Agent Sessions stays empty for Claude Code sessions for the same reason. +## Choose what content to record -## Tokens and cost +Claude Code redacts content by default. `OTEL_LOG_USER_PROMPTS=1` records prompts, which title each turn. `OTEL_LOG_TOOL_DETAILS=1` records Bash commands, file paths and full tool error messages. `OTEL_LOG_TOOL_CONTENT=1` records what each tool returned. -Each `claude_code.llm_request` span carries `input_tokens`, `output_tokens`, `cache_read_tokens` and `cache_creation_tokens`, plus `ttft_ms` and the stop reason. Maple maps them to its token buckets and time to first token. Anthropic's `input_tokens` excludes both cache buckets, and Maple counts it that way, so total input is the sum of the three. The CLI always streams and still records usage, so there is no streaming gap to work around. +`OTEL_LOG_TOOL_CONTENT=1` sends whatever files Claude reads and whatever commands print, secrets included. Turn on only what your Maple organization is allowed to store. -Claude Code reports the cache buckets on every call, even when they're zero. If your prompts are shorter than the model's minimum cacheable length (a short custom `systemPrompt` often is), nothing is cached and the session's **Prompt cache** check warns with a 0% hit rate. That warning is accurate, not a tracing problem. +## Let each turn finish exporting -Maple doesn't show cost for these sessions. Maple never prices tokens itself, and Claude Code puts cost only on the `api_request` log event (`cost_usd`) and the `claude_code.cost.usage` metric, never on a span. Sessions read as **unpriced**. - -To track spend anyway, keep the logs and metrics exporters on. The `claude_code.api_request` records are searchable under **Logs** with their `cost_usd` and `session.id`, and `claude_code.cost.usage` can go on a dashboard. Both are Claude Code's client-side estimate at list price, unless your organization sets `modelPricing` in managed settings. In an SDK app, the result message's `total_cost_usd` is the same estimate for one `query()`. - -## Flush short-lived processes - -The CLI exports spans every 5 seconds and flushes when it exits cleanly, but that last flush has a short timeout. In the SDK, a `query()` with a string prompt starts a CLI process that exits when the turn ends, so every turn ends with that flush. - -- Set `OTEL_TRACES_EXPORT_INTERVAL` and `OTEL_LOGS_EXPORT_INTERVAL` to `1000`, as in the snippets above, so most spans leave during the turn. -- Let the message loop run to the `result` message. Returning from the loop at `result`, as `reply()` does, is fine. Calling `close()` on a query, aborting it, or breaking out of the loop before that kills the CLI before it flushes. -- In a script, keep the process alive for about 5 seconds after the last `query()` finishes, so the CLI's final export completes. -- On serverless platforms, finish the loop before you return the response. A frozen instance can't flush. -- For `claude -p` in CI, set the same two intervals. +The CLI exits at the end of each turn, and its final flush has a short timeout. Let every `query()` loop reach its `result` message; breaking out early, calling `close()` or aborting kills the CLI before it flushes. In a script, keep the process alive about 5 seconds after the last `query()`. On serverless platforms, finish the loop before returning the response. ## Check that it works -Run one conversation with two turns and at least one tool call, for example with the `reply()` function above. Then open **Agent Sessions** in Maple. You should see: - -- **One session** for the conversation, with the framework shown as **Claude Agent SDK** (Claude Code sessions from the terminal show the same name) and your `OTEL_SERVICE_NAME` as the service. -- **One turn per message**, titled with the user's prompt. -- The model ID as Claude Code sent it (for example `anthropic/claude-haiku-4.5` when calling through OpenRouter), the LLM call count, and the four token buckets. -- The tool calls by name, with arguments and results as described in [Record prompts and tool calls](#record-prompts-and-tool-calls), and failed tools marked failed. -- Cost shown as **unpriced**, and no assistant text in the transcript. Both are expected, as explained above. - -How the spans map: - -| Claude Code span | In Agent Sessions | -| --- | --- | -| `claude_code.interaction` | A turn, titled with the prompt | -| `claude_code.llm_request` | A model call: model, tokens, time to first token, finish reason, failure | -| `claude_code.tool` | A tool call: name, arguments, result, failure | -| `claude_code.tool.blocked_on_user`, `claude_code.tool.execution` | Part of their tool call in the trace view, not calls of their own | +Run a two-turn conversation through `reply()` with at least one tool call, then open **Agent Sessions**. You should see one session with the framework **Claude Agent SDK**, one turn per message titled with the prompt, model calls with their tokens, and the tool calls by name. -With the logs exporter on, **Logs** has one `claude_code.user_prompt`, one `claude_code.api_request` per model call and one `claude_code.tool_result` per tool run for the same `session.id`. +The transcript has no assistant replies and cost shows as unpriced. Claude Code puts both only on log events, which you can search under **Logs** when the logs exporter is on. ## Troubleshooting -- **Metrics and logs arrive, but no traces and no sessions.** `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` or `OTEL_TRACES_EXPORTER=otlp` is missing. Both are required for spans. -- **Nothing arrives at all.** `OTEL_EXPORTER_OTLP_PROTOCOL` is unset (there is no default) or set to `grpc`. Set `http/protobuf`. To see export errors, set `CLAUDE_CODE_OTEL_DIAG_STDERR=1` and read the SDK's `stderr` callback, or run `claude --debug-file /tmp/claude.log` and look for `[3P telemetry]` lines. -- **The CLI ignores your settings.** They are in the repository's `.claude/settings.json`, which Claude Code ignores for telemetry since 2.1.282. Move them to `~/.claude/settings.json` or your shell. `/status` lists what it ignored. -- **401 errors, or data going to another backend.** Managed settings, or an org-distributed `~/.claude/remote-settings.json`, set `OTEL_EXPORTER_OTLP_ENDPOINT` or `OTEL_EXPORTER_OTLP_HEADERS`, and managed values win over yours. Ask whoever manages Claude Code in your organization, or add Maple to the managed configuration. -- **Every message is its own one-turn session.** Each `query()` starts a new session. Resume the conversation's session id as in [Group turns into one session](#group-turns-into-one-session), and check that `forkSession` is off and `OTEL_METRICS_INCLUDE_SESSION_ID` isn't `false`. -- **Your agent's turns appear inside another trace.** The process inherited a `TRACEPARENT`, typically because Claude Code or CI started it. Drop `TRACEPARENT` and `TRACESTATE` from the environment you pass to the SDK. -- **TypeScript: the CLI fails to start or can't authenticate after adding `env`.** `options.env` replaced the whole environment. Spread `process.env` into it. -- **SDK output is garbled or `query()` throws a JSON parse error.** An exporter is set to `console`. Use `otlp` or `none`. -- **Turns are untitled and the transcript has no user messages.** `OTEL_LOG_USER_PROMPTS=1` is missing, so the prompt is ``. -- **Your SDK tools have no results.** You need Claude Code 2.1.283 or later (Agent SDK 0.3.283 for TypeScript, 0.2.160 for Python) and `OTEL_LOG_TOOL_CONTENT=1`. -- **The last turn of a script is missing, or ends at a model call.** The process exited before the CLI's final export. See [Flush short-lived processes](#flush-short-lived-processes). -- **Extra turns titled ``, and sub-agent tool calls failed with "interrupted before a result was received".** Claude Code ran the sub-agents in the background. Set `CLAUDE_CODE_DISABLE_BACKGROUND_TASKS=1` as in [Tools, errors and sub-agents](#tools-errors-and-sub-agents). -- **The service name, endpoint or content flags aren't the ones you passed in `env`.** An `env` block in `~/.claude/settings.json` or `.claude/settings.json` overrides `options.env`. Pass `settingSources: []`, or remove the keys from the settings file. -- **Every model call appears twice.** You also installed a hook-based instrumentor for the SDK, such as OpenInference's. Remove it; the CLI's own spans are the ones Maple groups by session. +- **Metrics and logs arrive, but no sessions.** Set `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` and `OTEL_TRACES_EXPORTER=otlp`. +- **Nothing arrives at all.** `OTEL_EXPORTER_OTLP_PROTOCOL` is unset or `grpc`. Set `http/protobuf`. `CLAUDE_CODE_OTEL_DIAG_STDERR=1` prints export errors to the SDK's `stderr` callback. +- **Every message is its own session.** Resume the conversation's session id as shown above, and don't set `forkSession`. +- **Your agent's turns appear inside another trace.** The process inherited a `TRACEPARENT`. Drop it and `TRACESTATE` from the environment. +- **The CLI ignores your values.** They are in a repository's `.claude/settings.json`, or a settings file overrides `options.env`. Use `~/.claude/settings.json`, or pass `settingSources: []`. +- **Extra turns titled ``.** Claude Code ran sub-agents in the background. Add `CLAUDE_CODE_DISABLE_BACKGROUND_TASKS: "1"` to the env. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/crewai.md b/apps/landing/src/content/docs/agent-tracing/crewai.md index ad7792d0e4..20f8e9023a 100644 --- a/apps/landing/src/content/docs/agent-tracing/crewai.md +++ b/apps/landing/src/content/docs/agent-tracing/crewai.md @@ -1,17 +1,15 @@ --- title: "Trace CrewAI crews and flows with OpenTelemetry" -description: "Send CrewAI crews and flows to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools and an agent lane per crew member." +description: "Send CrewAI crews and flows to Maple with OpenInference, one Agent Session per conversation." group: "AI Agents" order: 21 navLabel: "CrewAI" icon: "crewai" --- -CrewAI sends nothing to your OpenTelemetry backend on its own. Its built-in telemetry is anonymous usage analytics that goes to CrewAI on a private tracer provider, and its OpenTelemetry export is a CrewAI AMP feature. The traces come from OpenInference: `openinference-instrumentation-crewai` records crews, flows, tasks and tools, and a second instrumentor for the SDK CrewAI calls records the model calls, prompts and tokens. +CrewAI doesn't export traces to your backend on its own. OpenInference's `openinference-instrumentation-crewai` records crews, agents and tools, and a second instrumentor for the SDK CrewAI calls records the model calls, prompts and tokens. Without that second instrumentor you get no transcript, and without a session id every `kickoff()` is its own session. -Most broken CrewAI traces are missing that second instrumentor. With only the CrewAI one, you get agent and tool spans with no model, no tokens and no transcript. The other gap is the conversation: CrewAI has no chat thread, so every `kickoff()` is its own trace with nothing linking it to the previous message. - -This guide covers CrewAI 1.15 with `openinference-instrumentation-crewai` 1.1.18 and `openinference-instrumentation-openai` 0.1.61, on Python 3.10 to 3.13. +Tested with CrewAI 1.15, `openinference-instrumentation-crewai` 1.1.18 and `openinference-instrumentation-openai` 0.1.61 on Python 3.10 to 3.13. ## Quick setup with a coding agent @@ -25,9 +23,9 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. -## Install the instrumentors and export to Maple +## Install the instrumentors ```bash pip install "crewai>=1.15" "openinference-instrumentation-crewai>=1.1.18" \ @@ -35,7 +33,7 @@ pip install "crewai>=1.15" "openinference-instrumentation-crewai>=1.1.18" \ "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -The OpenAI instrumentor is right for most apps because CrewAI 1.x calls most providers through the `openai` SDK. Which instrumentor you need depends on the model string you pass to `LLM(...)`: +Pick the model instrumentor by the model string you pass to `LLM(...)`: | Model string | SDK CrewAI calls | Instrumentor | | --- | --- | --- | @@ -45,9 +43,9 @@ The OpenAI instrumentor is right for most apps because CrewAI 1.x calls most pro | `bedrock/…` | `boto3` | `openinference-instrumentation-bedrock` | | Anything else (needs `crewai[litellm]`) | `litellm` | `openinference-instrumentation-litellm` | -Instrument only the SDKs your crews actually call. The Arize Phoenix CrewAI page still recommends the LiteLLM instrumentor, which records nothing when CrewAI uses its native providers. +Install only the ones your crews use. The LiteLLM instrumentor records nothing for the native providers in the first four rows. -Point the exporter at Maple with the standard OpenTelemetry variables: +## Point the exporter at Maple ```bash export OTEL_SERVICE_NAME=support-crew @@ -58,11 +56,13 @@ export CREWAI_DISABLE_TELEMETRY=true export CREWAI_TRACING_ENABLED=false ``` -EU organizations use `https://ingest.eu.maple.dev`. Set the base URL: `OTLPSpanExporter()` with no arguments appends `/v1/traces`. An `endpoint=` passed in code is used as is, so it must end in `/v1/traces`. +EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, it has to end in `/v1/traces`. + +The last two variables turn off CrewAI's anonymous analytics and its own trace uploader, whose first-run prompt waits for input at the end of a run. Don't use `OTEL_SDK_DISABLED=true` for this, because it disables your Maple traces too. -The last two variables turn off CrewAI's own pipelines. `CREWAI_DISABLE_TELEMETRY` stops the anonymous analytics export to `telemetry.crewai.com`. `CREWAI_TRACING_ENABLED=false` stops the CrewAI AMP trace uploader and its first-run "view your traces" prompt, which waits for input at the end of a run. Don't use `OTEL_SDK_DISABLED=true`, which CrewAI's telemetry docs also mention: it disables the OpenTelemetry SDK for the whole process, and your Maple traces with it. +## Initialize tracing -Then add a `tracing.py` and import it at the top of your entry point, before you build any crew: +Add a `tracing.py` and import it at the top of your entry point, before the first `kickoff()`: ```py # tracing.py @@ -98,19 +98,13 @@ CrewAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_de OpenAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) ``` -What each part does: +Pass `config` with `enable_genai_semconv=True` to every instrumentor, or the session page has no transcript and no tool details. `CrewAIAgentNames` gives each agent role its own lane. `skip_dep_check=True` stops an instrumentor from silently skipping itself when its version check disagrees with your installed packages. -- **`enable_genai_semconv=True`** makes both instrumentors write the OpenTelemetry GenAI attributes (`gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*`, `gen_ai.conversation.id`) next to their OpenInference ones when each span ends. Maple's session page reads the GenAI names for CrewAI's agent and tool spans, so without it those spans have no operation or tool details, and model calls have no message transcript. The environment variable `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same if it's set before `instrument()` runs. -- **`CrewAIAgentNames`** fills a gap in the CrewAI instrumentor, which puts the agent's role in the span name and in `graph.node.id` but never in an agent-name attribute. Without it, every agent in a crew shares one lane in Maple. -- **`skip_dep_check=True`** stops an instrumentor from skipping itself, with only an error log, when its version check disagrees with your installed CrewAI or `openai` package. - -Both instrumentors patch classes in place, so `instrument()` only has to run before the first `kickoff()`. If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second one. Add `CrewAIAgentNames()` and the OTLP exporter to the existing provider and pass that provider to `instrument()`. +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `CrewAIAgentNames()` and the exporter to that provider and pass it to `instrument()`. ## Group a conversation into one session -A `Crew` runs its tasks once and returns. There's no thread or session id, and the instrumentors set none: `crew_id` changes with every `Crew` object and `crew_key` is the same for every user of the same crew, so neither works as a conversation id. Maple groups traces into a session by `session.id`, which the instrumentors set only inside OpenInference's `using_session` context manager. - -In a chat backend, build the crew per message, put the user's message first in the task description, and run it inside `using_session` with the conversation id your app already stores: +CrewAI has no conversation id, so wrap every `kickoff()` in OpenInference's `using_session` with the id your app stores the chat under: ```py import tracing # noqa: F401 (first import) @@ -143,40 +137,15 @@ def handle_message(conversation_id: str, text: str, history: str) -> str: return build_crew(text, history).kickoff().raw ``` -Each `kickoff()` is one trace and one turn in the session. CrewAI doesn't remember earlier messages, so pass the history yourself, as above. Maple labels each turn with the first line of the prompt CrewAI builds, `Current Task: `, which is why the user's message goes first. - -Give the crew a `name`. An unnamed crew's root span is `Crew_.kickoff`, a different name on every request. - -If you skip `using_session`, every message shows up in **Agent Sessions** as its own one-turn session named after its trace id. Setting `gen_ai.conversation.id` on your own spans doesn't help, because Maple reads `session.id` for CrewAI. - -`using_session` also works for conversational flows, where CrewAI's own session id is the flow's `state.id`. Use the same value for both, and give the flow a `name`: - -```py -from crewai.flow import ConversationConfig, ConversationState, Flow - - -@ConversationConfig(llm=llm, system_prompt="You are a concise assistant.") -class SupportFlow(Flow[ConversationState]): - name = "support_flow" - - -support_flow = SupportFlow() - - -def handle_turn(conversation_id: str, text: str) -> str: - with using_session(conversation_id): - return support_flow.handle_turn(text, session_id=conversation_id) -``` - -Each `handle_turn` runs one `kickoff()`, so each message is a trace under a `support_flow.kickoff` span, with `support_flow.route_conversation` and `support_flow.converse_turn` below it. Without `name`, the root is `Flow_.kickoff`, and the name changes to `Flow_` after the first turn. +Put the user's message first in the task description, because Maple labels each turn with the first line of the prompt. Name the crew, or its root span gets a new `Crew_` name on every request. `crew_id` and `crew_key` don't work as conversation ids. -### Async kickoffs +For conversational flows, wrap `flow.handle_turn(text, session_id=conversation_id)` in the same `using_session` and set `name = "support_flow"` on the flow class. -Use `kickoff()` or `await crew.kickoff_async()`. Don't use `await crew.akickoff()`: it runs a separate native-async code path that the CrewAI instrumentor doesn't patch, so there's no crew or agent span and every model call and tool call becomes its own trace. `kickoff_async()` runs the instrumented `kickoff()` in a thread and keeps the session and trace. +Use `kickoff()` or `await crew.kickoff_async()`. The instrumentor doesn't patch `akickoff()`, so with it every model and tool call becomes its own trace. -### Streaming +## Streaming crews -`Crew(stream=True)` returns a `CrewStreamingOutput` right away and calls `kickoff()` again in a thread once you iterate it. The instrumentor records both calls, so each streamed message produces two traces: an empty `support.kickoff` and the real one. Wrap the turn in one span of your own so both land in one trace and one turn: +`Crew(stream=True)` calls `kickoff()` twice, which gives each message an extra empty turn. Wrap the streamed turn in one span of your own: ```py from opentelemetry import trace @@ -196,105 +165,25 @@ def stream_message(conversation_id: str, text: str, history: str, send) -> None: send(chunk.content) ``` -`gen_ai.operation.name` makes Maple treat the wrapper as the turn's agent span, so the two crew spans under it don't count as two turns. `LLM(stream=True)` without `Crew(stream=True)` streams inside a single `kickoff()` and needs none of this. - -## Record prompts, responses and tool calls - -Content capture is on by default. Each model call carries the messages CrewAI sent (the agent's role, goal and backstory as the system message, then `Current Task: …` with the context from earlier tasks) and the model's reply, including tool calls. Agent spans carry the task and its output, and tool spans carry the tool's arguments and result. With the GenAI dual-write on, Maple renders the model calls as the session transcript. - -To keep prompts and outputs out of your traces, add the hide switches to the config both instrumentors share: - -```py -config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) -``` - -`hide_inputs` drops the input messages and replaces `input.value` with `__REDACTED__`, and `hide_outputs` does the same for outputs. The session keeps its turns, model and tool calls, tokens and failures, with an empty transcript. `hide_input_text` and `hide_output_text` keep the message structure but redact the text. Each switch also has an `OPENINFERENCE_HIDE_*` environment variable. - -The switches don't cover everything. The crew's `.kickoff` span always records `crew_tasks` (every task description, which is where the user's message goes in the example above), `crew_inputs` (the `kickoff(inputs=...)` values) and `crew_agents` (roles, goals and backstories). A flow's kickoff span records `flow_inputs` the same way. If prompts must never leave your infrastructure, delete those attributes in an OpenTelemetry Collector with an `attributes` processor before the data reaches Maple. +`LLM(stream=True)` without `Crew(stream=True)` doesn't need this. -Agent roles, task names and tool names are always recorded, because they're span names. CrewAI's analytics, if you leave them on, also collect roles and tool names, so keep personal data out of both. +## Flush in short-lived processes -## Tools, errors and agents - -Each tool call is a `.run` span with `gen_ai.operation.name` `execute_tool` and the tool's name in `gen_ai.tool.name`. A tool that raises is marked failed without extra code: the span ends with status `ERROR` and the exception message, for example `transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. CrewAI then sends `Error executing tool: …` back to the model, and the agent and crew spans stay `OK` because the crew itself carried on. - -Tool spans come from CrewAI's native function calling, which every provider in the table above uses. A model that CrewAI drives with text-based ReAct prompts instead (some custom `BaseLLM` subclasses and LiteLLM models without function calling) runs tools through a path the instrumentor doesn't patch, so those calls have no tool span. - -Each task is an agent span named `.._execute_core`, with `gen_ai.operation.name` `invoke_agent`. `CrewAIAgentNames` gives it the role as `gen_ai.agent.name`, and Maple opens a lane for each agent in the crew. Tasks with `async_execution=True` run in threads; the instrumentor copies the trace context into them, so parallel tasks stay in the same trace as siblings under the crew span. - -In a hierarchical crew (`process=Process.hierarchical`), CrewAI adds a manager agent called `Crew Manager` that delegates with the `Delegate work to coworker` and `Ask question to coworker` tools. The delegated work runs through `Agent.execute_task`, which the instrumentor doesn't patch, so the coworker's model calls appear inside the delegation's tool span with no agent span of their own and no lane. The manager's own task is an agent span like any other. - -Tool approvals with CrewAI's `@before_tool_call` hook need no tracing changes. The hook runs inside the `kickoff()`, before the tool span starts, so the approval stays in the turn's trace and an approved call is one tool span: - -```py -from crewai.hooks import before_tool_call - - -@before_tool_call(tools=["delete_file"]) -def approve_delete(context): - answer = context.request_human_input(prompt=f"Allow delete_file({context.tool_input})?") - return None if answer.strip().lower() == "approve" else False -``` - -A blocked call has no tool span at all: the model gets `Tool execution blocked by hook` as the result, and the refusal shows only in the next model call's input. - -In a flow, `.kickoff` is the root and each `@start`, `@listen` and `@router` method is a `.` span, with any crew or `Agent.kickoff()` it runs nested inside. A flow paused by `@human_feedback` and continued with `flow.resume(...)` continues in a new request, and `resume()` isn't patched, so wrap it in `using_session` with the same id and a span of your own, like the streaming wrapper above. - -## Tokens and cost - -Every model call carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals, along with cached and reasoning tokens when the provider reports them. Crew and agent spans carry no usage of their own, so nothing is counted twice. The model is the one the provider returned, such as `openai/gpt-4o-mini` or `anthropic/claude-haiku-4.5` behind OpenRouter. The provider is the SDK the call went through, so every model behind OpenRouter shows as `openai`. - -Streamed calls keep their token counts: CrewAI's OpenAI provider requests `stream_options={"include_usage": True}` whenever it streams. - -Maple shows cost only when a span carries one. The OpenAI, Anthropic and Gemini instrumentors never record cost, so those sessions show as **unpriced**, with token counts. LiteLLM computes a price for the models it knows, and the LiteLLM instrumentor records it as `llm.cost.total`, which Maple reads. - -`memory=True` and `planning=True` make model calls of their own (memory analysis, embeddings, a planning agent). They appear in the session because they're real, billed calls. - -## Flush spans before the process exits - -`BatchSpanProcessor` exports every 5 seconds, and the `TracerProvider` flushes from an `atexit` handler on a normal interpreter exit. That covers most scripts and CLIs, but not a killed process, `os._exit`, a frozen serverless instance or a notebook. Flush yourself in those cases: - -```py -from tracing import provider - -try: - handle_message("conv-42", "What's the weather in Berlin?", history="") -finally: - provider.force_flush() # serverless: before returning; notebooks: after each run -``` - -Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. +The SDK flushes on a normal interpreter exit, which covers servers and `crewai run`. In serverless handlers and notebooks, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. ## Check that it works -Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: - -- **One session** for the conversation, with one turn per `kickoff()`. Each turn's trace starts at `support.kickoff` (your crew's name), or `support_flow.kickoff` for a flow named `support_flow`. -- **Framework: CrewAI** in the session list. -- **The transcript**: the system message built from the agent's role, goal and backstory, `Current Task: …` with your message, and the model's replies. Turns are labeled `Current Task: `. -- **Model calls** named `ChatCompletion` (from the OpenAI instrumentor), each with a model and input and output tokens. -- **Tool calls** named `get_weather.run` and `calculate.run`, with results. -- **Agents**: one lane per role, from `assistant.reply._execute_core` and its siblings. -- **Cost**: unpriced, unless your models go through LiteLLM. +Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session labeled **CrewAI**, with one turn per `kickoff()` starting at `support.kickoff`, `ChatCompletion` model calls with tokens, tool calls like `get_weather.run`, and one lane per agent role. -A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. +Cost shows as **unpriced** unless your models go through LiteLLM. That's expected. ## Troubleshooting -- **Agent and tool spans, but no model calls or tokens.** The instrumentor for your provider's SDK isn't installed. Match it to the model string with the table above; `openai/…` and `openrouter/…` models need `openinference-instrumentation-openai`, not the LiteLLM one. -- **No spans at all.** `instrument()` never ran, ran after the first kickoff, or skipped itself on its version check. Import `tracing` first, keep `skip_dep_check=True`, and look for an exporter error in the logs. Check that `OTEL_SDK_DISABLED` isn't set. -- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. -- **One session per message.** The kickoff isn't inside `using_session(...)`, or the id changes per request. Wrap every `kickoff()` and use the stored conversation id. -- **Every model call and tool call is its own trace.** The crew runs with `akickoff()`. Use `kickoff()` or `kickoff_async()`. -- **An empty extra turn for each streamed message.** `Crew(stream=True)` calls `kickoff()` twice. Wrap the turn in one span, as in [Streaming](#streaming). -- **Agent and tool spans have no details on the session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)` to both instrumentors. -- **All agents in one lane.** `CrewAIAgentNames` isn't on the provider, or it was added to a different provider than the one passed to `instrument()`. -- **Tool arguments show the tool's input schema.** The dual-write copies `tool.parameters`, which is the schema, into `gen_ai.tool.call.arguments`. The call's actual arguments are in the tool span's `input.value`. This is an OpenInference mapping bug with no workaround in the span processor, because the arguments are set after the span starts. -- **Flow turns are named `Flow_.kickoff`.** The flow class has no `name`. Set `name = "support_flow"` on the class. -- **Prompts still visible with `hide_inputs=True`.** They're in the crew span's `crew_tasks` and `crew_inputs` attributes, which the switches don't cover. Delete them in a Collector. -- **A delegated coworker has no lane.** Hierarchical delegation runs the coworker through `Agent.execute_task`, which isn't instrumented. Its model calls are inside the `Delegate work to coworker` tool span. -- **Every model call appears twice.** Two model-layer instrumentors cover the same call, for example LiteLLM's and OpenAI's with a LiteLLM model that calls the `openai` SDK, or `litellm.callbacks=["otel"]` next to an OpenInference instrumentor. Keep one. -- **The process hangs at exit asking about traces.** CrewAI's first-run trace prompt. Set `CREWAI_TRACING_ENABLED=false`. +- **Agent and tool spans, but no model calls or tokens.** The instrumentor for your model's SDK is missing. Match it to the model string with the table above. +- **No spans at all.** Import `tracing` before the first kickoff, keep `skip_dep_check=True`, make sure `OTEL_SDK_DISABLED` isn't set, and check the logs for exporter errors. +- **One session per message.** The kickoff isn't inside `using_session(...)`, or the id changes per request. +- **Every model and tool call is its own trace.** Replace `akickoff()` with `kickoff()` or `kickoff_async()`. +- **The process hangs at exit asking about traces.** Set `CREWAI_TRACING_ENABLED=false`. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/dspy.md b/apps/landing/src/content/docs/agent-tracing/dspy.md index 41782a0985..6258869b74 100644 --- a/apps/landing/src/content/docs/agent-tracing/dspy.md +++ b/apps/landing/src/content/docs/agent-tracing/dspy.md @@ -1,17 +1,15 @@ --- title: "Trace DSPy programs and ReAct agents with OpenTelemetry" -description: "Send DSPy programs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, cost and failed tools, including modules that run in dspy.Parallel." +description: "Send DSPy programs to Maple as one Agent Session per conversation, with transcript, model and tool calls, tokens and cost." group: "AI Agents" order: 27 navLabel: "DSPy" icon: "python" --- -DSPy has no OpenTelemetry code of its own. Spans come from OpenInference's `openinference-instrumentation-dspy`, which patches `Module.__call__`, `Predict.forward`, the adapters, `LM.__call__` and `Tool.__call__`. That gives you the shape of a program (which module called which predictor, which tool ran, what the model was sent), but no token counts, no tool names and no conversation id. +DSPy has no tracing of its own. OpenInference's `openinference-instrumentation-dspy` records a span for every module, predictor, model call and tool call, but no token counts, no tool names and no conversation id. You add the first two with a small DSPy callback, and the conversation id by wrapping each call in `using_session`. -The usual fix for the missing tokens, adding `openinference-instrumentation-litellm`, stopped working in DSPy 3.4. Most models now run on DSPy's own `lm15` engine instead of LiteLLM, so the LiteLLM instrumentor records nothing: in our test, a call to `openrouter/openai/gpt-4o-mini` produced zero LiteLLM spans. This guide fills the gaps with a DSPy callback instead, which works on both engines. - -This guide covers DSPy 3.4 with `openinference-instrumentation-dspy` 0.1.45 on Python 3.10 or later, for `dspy.ReAct` agents and your own `dspy.Module` programs. +Tested with DSPy 3.4 and `openinference-instrumentation-dspy` 0.1.45 on Python 3.10 or later. ## Quick setup with a coding agent @@ -25,9 +23,9 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is under **Settings → Ingestion**. -## Install the instrumentor and export to Maple +## Install and point the exporter at Maple ```bash pip install "dspy>=3.4" "openinference-instrumentation-dspy>=0.1.45" "openinference-instrumentation>=0.1.66" \ @@ -35,8 +33,6 @@ pip install "dspy>=3.4" "openinference-instrumentation-dspy>=0.1.45" "openinfere "opentelemetry-instrumentation-threading>=0.66b0" ``` -Point the exporter at Maple with the standard OpenTelemetry variables: - ```bash export OTEL_SERVICE_NAME=support-agent export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production @@ -45,7 +41,9 @@ export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf ``` -EU organizations use `https://ingest.eu.maple.dev`. Set the base URL: `OTLPSpanExporter()` with no arguments appends `/v1/traces`. An `endpoint=` passed in code is used as is and has to end in `/v1/traces`. +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. + +## Initialize tracing Add a `tracing.py` and import it at the top of your entry point, before your DSPy modules are defined: @@ -67,14 +65,13 @@ DSPyInstrumentor().instrument(tracer_provider=provider, config=TraceConfig(enabl ThreadingInstrumentor().instrument() ``` -- **`enable_genai_semconv=True`** makes the instrumentor also write the OpenTelemetry GenAI attributes (`gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.provider.name`, `gen_ai.request.model`, `gen_ai.tool.call.result`) when each span ends. Maple's session page reads those for DSPy, not the OpenInference originals, so without it the session page has no transcript and no model. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same if it's set before `instrument()` runs. -- **`ThreadingInstrumentor`** carries the trace context into threads. `dspy.Parallel`, `Evaluate` and a plain `ThreadPoolExecutor` start every worker without it, so each worker becomes its own trace with no parent and no session (see [Parallel modules](#parallel-modules-stay-in-one-trace)). +`enable_genai_semconv=True` writes the `gen_ai.*` attributes Maple's session page reads. Without it the session has no transcript and no model. `ThreadingInstrumentor` keeps `dspy.Parallel`, `Evaluate` and thread pool workers in the caller's trace and session. -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), don't create a second one. Add the OTLP exporter to the existing provider and pass that provider to `instrument()`. +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add the OTLP exporter to that one and pass it to `instrument()` instead of creating a second. -### Add the Maple callback +## Add the Maple callback -The instrumentor leaves out four things Maple needs: tokens and cost on model spans, tool names and arguments on tool spans, an agent span per program, and a way to tell DSPy's adapter spans apart from model calls. A DSPy callback runs inside each of those spans, so it can add them. Save this as `maple_dspy.py`: +The callback adds tokens and cost to model spans, names and arguments to tool spans, and an agent span for each module class you wrote. Save it as `maple_dspy.py`: ```py # maple_dspy.py @@ -166,17 +163,11 @@ from maple_dspy import MapleCallback dspy.configure(lm=dspy.LM("openai/gpt-4o-mini", temperature=0), callbacks=[MapleCallback()]) ``` -The callback works because the instrumentor patches DSPy's classes from the outside, so DSPy's own callback hooks run while the instrumentor's span is the current one. Any `dspy.Module` subclass you write becomes an agent in Maple, named after its class. DSPy's built-in modules (`Predict`, `ChainOfThought`, `ReAct`) stay plain steps inside it. The GenAI dual-write never overwrites a key that's already set, so the callback's values win. - -### Why not MLflow or the OpenTelemetry DSPy package - -DSPy's documented tracing path is `mlflow.dspy.autolog()`. MLflow can export OTLP and translate to GenAI attributes (`MLFLOW_ENABLE_OTEL_GENAI_SEMCONV`, MLflow 3.11 and later), but it pulls in the whole MLflow package, its GenAI mapping is documented for provider SDK spans, and it has no conversation id Maple can read. - -The OpenTelemetry project published `opentelemetry-instrumentation-genai-dspy` on 24 September 2026 as a 1.2 beta. It emits GenAI attributes directly, but only for tools, ReAct loops and retrieval. It records no model calls, and relies on a provider instrumentor for those, which DSPy 3.4's native engine bypasses. Maple also identifies DSPy by the OpenInference instrumentor, so its spans would show as an unidentified framework. We'll revisit it once it covers model calls. +Every `dspy.Module` subclass you write becomes an agent in Maple, named after its class. Built-in modules like `Predict` and `ReAct` stay steps inside it. Give worker modules descriptive class names (`WeatherWorker`) so each gets its own lane. ## Group a conversation into one session -DSPy has no session, thread or conversation id. A chat program takes the conversation as a `dspy.History` input field, and every call to your module is a new root span and a new trace. Maple groups a DSPy program's traces into one session by the `session.id` attribute, which the instrumentor only sets inside OpenInference's `using_session` context manager: +Each call to your module starts a new trace. Maple joins them into one session by `session.id`, which the instrumentor sets inside OpenInference's `using_session` block: ```py import dspy @@ -205,96 +196,15 @@ def handle_message(conversation_id: str, question: str, turns: list[dict]) -> st return answer ``` -Use your app's own conversation id, the one it stores the chat under. A new UUID per request gives you one session per message, and a constant gives every user one shared session. `using_session` puts the id in the OpenTelemetry context, and the instrumentor copies it onto every span it creates inside the block, including spans in worker threads once `ThreadingInstrumentor` is on. - -If you skip this, every call to your module shows up in **Agent Sessions** as its own one-turn session named after its trace id. The GenAI dual-write also copies `session.id` into `gen_ai.conversation.id`, but that key alone doesn't group anything: Maple reads `session.id` for DSPy. - -Don't store the conversation on the module as `self.history`. DSPy appends every model call to a list attribute named `history` on each calling module, so a `dspy.History` there crashes the first model call with `TypeError: object of type 'History' has no len()`. Pass it as an input field, or name the attribute something else. - -## Stream replies with dspy.streamify - -A streamed turn is traced like any other: same session, same spans, tokens and cost on every model call. Two details decide whether that holds. Here is a streaming handler you can return from an SSE or WebSocket endpoint: - -```py -stream_assistant = dspy.streamify(assistant, stream_listeners=[dspy.streaming.StreamListener("answer")]) - - -async def stream_message(conversation_id: str, question: str, turns: list[dict]): - with using_session(conversation_id): - async for chunk in stream_assistant(question=question, conversation=dspy.History(messages=turns)): - if isinstance(chunk, dspy.streaming.StreamResponse): - yield chunk.chunk - elif isinstance(chunk, dspy.Prediction): - turns.append({"question": question, "answer": chunk.answer}) -``` - -- **Call `dspy.streamify` after `dspy.configure(callbacks=[MapleCallback()])`.** `streamify` copies the callback list at the moment you call it. A streamed program built at import time, before `configure` runs, streams without the callback: no tokens, no tool names, and every `ChatAdapter.__call__` counted as a model call. -- **Put `using_session` around the loop that reads the stream.** The program doesn't start until the first chunk is requested. If the `with` block only wraps the `stream_assistant(...)` call, for example in an endpoint that returns `StreamingResponse(stream_assistant(...))`, it has already exited when the model runs, and the turn lands in a session of its own. - -DSPy asks the provider for usage on streamed calls, so the streamed `LM.__call__` span has input and output tokens and cost. In our test, the streamed answer to "What is the capital of France?" recorded 341 input and 47 output tokens. Two things are missing on streamed calls: `gen_ai.response.id`, because DSPy's native engine doesn't keep the id from a stream, and a time-to-first-token attribute, which nothing in this setup records. - -## Record prompts, responses and tool calls - -Content capture is on by default. Every `LM.__call__` span carries the messages DSPy's adapter sent to the model (the signature's instructions as the system message, then the formatted fields) and the raw reply, and every tool span carries the tool's arguments and result. The callback adds a short user and assistant message on each agent span, built from your module's string inputs and outputs, which Maple uses as the turn's title. - -The model messages look like DSPy's prompt format, not like a chat: the user message starts with `[[ ## question ## ]]` and the reply with `[[ ## next_thought ## ]]` or `[[ ## answer ## ]]`. That's what the model actually saw. A `dspy.History` input is expanded into earlier user and assistant messages, so each call repeats the whole conversation so far. - -To keep prompts and outputs out of your traces, set the instrumentor's switches, which the callback also reads: - -```bash -export OPENINFERENCE_HIDE_INPUTS=true -export OPENINFERENCE_HIDE_OUTPUTS=true -``` - -The session still shows its turns, model and tool calls, tokens and failures, with an empty transcript and no tool arguments or results. Set them before `tracing.py` runs. Narrower switches (`OPENINFERENCE_HIDE_INPUT_TEXT`, `OPENINFERENCE_HIDE_OUTPUT_TEXT`, `OPENINFERENCE_HIDE_LLM_INVOCATION_PARAMETERS`) redact parts of each message; the callback's agent messages follow only the two above. - -## Tools, errors and sub-agents - -Each tool call is a span named `.__call__`, of OpenInference kind `TOOL` and `gen_ai.operation.name` `execute_tool`. The callback adds `gen_ai.tool.name`, the tool's docstring as `gen_ai.tool.description`, and its arguments. `dspy.ReAct` also calls a built-in `finish` tool when it's done, so every ReAct run ends with a `finish.__call__` span, and `finish` shows up in Maple's tool list. - -A tool that raises is marked failed with no extra code. The instrumentor ends the tool span with status `ERROR` and the exception as the status message, for example `RuntimeError: transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. `ReAct` catches the exception and hands `Execution error in fetch_transport_data: ...` back to the model, so the `ReAct.forward` span and your module's span stay `OK`, and the program's return value doesn't tell you the tool failed. - -Tool spans have no `gen_ai.tool.call.id`. `ReAct` asks the model for the next tool as text fields (`next_tool_name`, `next_tool_args`), not through the provider's tool-calling API, so there's no id to link. - -DSPy's idiom for sub-agents is module composition: an orchestrator module that calls worker modules. With the callback, every module class you wrote is an `invoke_agent` span with its class name as `gen_ai.agent.name`, and Maple opens a lane for each agent whose name differs from its caller's. Name the classes after what they do (`WeatherWorker`, `BudgetWorker`); two instances of one class share a name and a lane. - -### Parallel modules stay in one trace - -`dspy.Parallel` runs modules in a `ThreadPoolExecutor` and copies only DSPy's own settings into each worker thread, not the OpenTelemetry context. Without `ThreadingInstrumentor`, each worker's spans start a new trace with no parent and no `session.id`: in our test, a three-worker fan-out became four traces, with 61 of 73 spans outside the trace that started them, including the only failed tool. With it, the same program is one trace and one session: - -```py -class Briefing(dspy.Module): - def __init__(self): - super().__init__() - self.weather = WeatherWorker() - self.transport = TransportWorker() - - def forward(self, city): - weather, transport = dspy.Parallel(num_threads=2)( - [(self.weather, {"city": city}), (self.transport, {"city": city})] - ) - return dspy.Prediction(briefing=f"{weather.findings}\n{transport.findings}") -``` - -The same applies to your own `ThreadPoolExecutor` and `threading.Thread`. `ThreadingInstrumentor` doesn't reach `multiprocessing`; a module that runs in another process starts its own trace. - -## Tokens and cost +Use the id your app stores the chat under. A new UUID per request gives one session per message, and a constant puts every user in one session. -The instrumentor records no token counts. The callback reads them from the provider response DSPy keeps in `lm.history` and writes them on the `LM.__call__` span: `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, cached input tokens and reasoning tokens when the provider reports them, plus `gen_ai.response.id` and the model the provider answered with. +If you stream with `dspy.streamify`, call it after `dspy.configure(callbacks=[MapleCallback()])`, since it copies the callback list when called. Put `using_session` around the loop that reads the stream, because the program only starts on the first chunk. -Three cases have no tokens on the span: +To keep prompts and outputs out of your traces, set `OPENINFERENCE_HIDE_INPUTS=true` and `OPENINFERENCE_HIDE_OUTPUTS=true` before `tracing.py` runs. The callback follows both. -- **Cache hits.** `dspy.LM` caches responses by default (`cache=True`). A repeated prompt is answered from the cache with no provider call, so the callback records nothing for it. Pass `cache=False` while you're checking the numbers. -- **History turned off.** `dspy.configure(disable_history=True)` or `max_history_size=0` stops DSPy from keeping the response the callback reads. -- **A custom engine.** An `LM` with your own `engine=` reports whatever usage your engine puts on its response. +## Flush before short-lived processes exit -Cost is DSPy's own estimate, the `cost` field of each history entry, written as `gen_ai.usage.cost`. On the `lm15` engine DSPy prices tokens from its bundled model metadata; on the LiteLLM engine it's LiteLLM's `response_cost`. Maple never prices tokens itself, so when DSPy has no price for a model, that call shows as **unpriced**. Treat the figure as an estimate, not your bill. - -Don't add `openinference-instrumentation-litellm` or `-openai` next to this setup. On DSPy 3.4's native engine they record nothing. On the LiteLLM engine (`dspy.LM(..., engine="litellm")`, and Anthropic models by default) they add a second model span under every `LM.__call__`, and Maple counts each call twice. - -## Flush spans before the process exits - -`BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` flushes on a normal interpreter exit, which covers most scripts and CLIs. That doesn't happen when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: +`BatchSpanProcessor` exports every 5 seconds and flushes on a normal exit. A killed process, a serverless handler or a notebook needs an explicit flush: ```py from tracing import provider @@ -305,38 +215,20 @@ finally: provider.force_flush() # serverless: before returning; notebooks: after each run ``` -Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. DSPy optimizers (`MIPROv2`, `GEPA`, `BootstrapFewShot`) and `dspy.Evaluate` make hundreds of calls; trace them in a separate service name or not at all, so they don't bury your production sessions. - ## Check that it works -Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, with `cache=False` on the LM. Then open **Agent Sessions** in Maple. You should see: - -- **One session** for the conversation, framework **DSPy**, with one turn per call to your module. Each turn's trace starts at `ChatAssistant.forward` (your module's class name), titled with the question you asked. -- **Model calls** named `LM.__call__`, each with a model, input and output tokens. Every model call sits under `Predict.forward`, `Predict(StringSignature).forward` and `ChatAdapter.__call__` spans; those are DSPy's steps, not extra calls. -- **The transcript**: the messages DSPy sent, in its `[[ ## field ## ]]` format, and the replies. -- **Tool calls** named `get_weather.__call__` and `finish.__call__`, with arguments and results. `finish` is ReAct's end-of-loop tool and counts as a tool call, so a turn that used one tool shows two. -- **Agents**: one per module class you wrote, with a lane for each worker module. -- **Cost** per call where DSPy has a price for the model. -- **Checks**: a tool that raised fails the session's **Tool availability** check, with the exception as its headline. - -A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. +Run a conversation of two or three messages through `handle_message` with one conversation id, including one tool call, with `cache=False` on the LM so every call reaches the provider. In **Agent Sessions** you should see one session with framework **DSPy** and one turn per call, each starting at `ChatAssistant.forward`. -Expect a **Prompt cache** warning on a short test conversation. OpenRouter reports cached input tokens even when they're zero, so Maple judges the cache hit rate, and providers only cache long prompts (OpenAI from 1,024 tokens; Anthropic models only with explicit cache markers). In our test the prompts peaked at 870 tokens and none were cached. +Model calls are `LM.__call__` spans with a model and tokens. The transcript uses DSPy's `[[ ## field ## ]]` prompt format. Tool calls include `finish.__call__`, which is `dspy.ReAct`'s built-in end-of-loop tool. Cost is DSPy's own estimate, and models DSPy has no price for show as **unpriced**. ## Troubleshooting -- **No spans at all.** `instrument()` never ran, or the exporter can't reach Maple. Import `tracing` first in the entry point and look for an `OTLPSpanExporter` error in the logs. -- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. -- **Tokens in the list, empty session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)` to `instrument()`. -- **No tokens anywhere.** `MapleCallback` isn't registered, a later `dspy.configure(callbacks=[...])` or `dspy.context(callbacks=[...])` replaced it, or the calls were cache hits. -- **One session per message.** The call isn't inside `using_session(...)`, or the id changes per request. -- **A streamed turn has no tokens, or is a session of its own.** `dspy.streamify` ran before `dspy.configure(callbacks=[MapleCallback()])`, or `using_session` exits before the stream is read. See [Stream replies](#stream-replies-with-dspystreamify). -- **`dspy.Parallel` workers are separate traces with no session.** `ThreadingInstrumentor().instrument()` didn't run. It has to run before the worker threads start. +- **No spans, or exports fail with 404.** Import `tracing` first. An `endpoint=` passed in code must end in `/v1/traces`; the env variable takes the base URL. +- **Tokens in the list, empty session page.** Pass `TraceConfig(enable_genai_semconv=True)` to `instrument()`. +- **No tokens anywhere.** `MapleCallback` isn't registered, a later `dspy.configure(callbacks=[...])` replaced it, or the calls were cache hits. +- **One session per message.** The call runs outside `using_session(...)`, or the id changes per request. +- **Every model call counted twice.** Remove `openinference-instrumentation-litellm` or `-openai`; the callback already records model calls. - **`TypeError: object of type 'History' has no len()`.** A module attribute is named `history`. Rename it, or pass the `dspy.History` as an input field. -- **Every model call counted twice.** A LiteLLM or OpenAI instrumentor is also installed, or the callback isn't registered and `ChatAdapter.__call__` is read as a model call by its name. Remove the extra instrumentor and register the callback. -- **A `finish` tool in every run.** That's `dspy.ReAct`'s built-in tool for ending the loop, not one of yours. -- **Your module shows `OK` after a tool failed.** `ReAct` turns tool exceptions into text for the model. The failed tool span is still marked `ERROR` and counted. -- **Two spans for one step after a parse error.** When `ChatAdapter` can't parse a reply, DSPy retries with `JSONAdapter`, so one `Predict` span holds two adapter spans, each with its own `LM.__call__`. Both calls happened and both are billed. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/google-adk.md b/apps/landing/src/content/docs/agent-tracing/google-adk.md index fc3a4b5558..7ae397a500 100644 --- a/apps/landing/src/content/docs/agent-tracing/google-adk.md +++ b/apps/landing/src/content/docs/agent-tracing/google-adk.md @@ -1,17 +1,17 @@ --- title: "Trace Google ADK agents with OpenTelemetry" -description: "Send Google Agent Development Kit (ADK) traces to Maple with the full transcript, tool arguments and results, tokens, and one session per ADK session." +description: "Send Google Agent Development Kit (ADK) traces to Maple with the transcript, tool calls and tokens, one session per ADK session." group: "AI Agents" order: 22 navLabel: "Google ADK" icon: "googleadk" --- -Google's Agent Development Kit (ADK) creates OpenTelemetry spans itself, with no instrumentation package: one span per run, per agent, per model call and per tool call, under the instrumentation scope `gcp.vertex.agent`. Every span carries the ADK session id as `gen_ai.conversation.id`, so a multi-turn chat groups into one Maple session without any extra code. +Google's Agent Development Kit (ADK) emits OpenTelemetry spans for every run, agent, model call and tool call, and stamps the ADK session id on them, so a multi-turn chat groups into one Maple session. You need no instrumentation package. -What goes wrong by default is the transcript. ADK writes prompts, replies and tool payloads into its own `gcp.vertex.agent.llm_request`, `llm_response`, `tool_call_args` and `tool_response` attributes, which Maple doesn't read, so the session shows models and tokens next to an empty conversation. Two environment variables switch ADK to the OpenTelemetry GenAI message format that Maple renders. The other trap: with a plain `Runner`, nothing exports at all until you register a tracer provider yourself. +Two things need setup. By default ADK writes prompts and replies into its own attributes, which Maple doesn't read, so you switch it to the GenAI format with two environment variables. And with a plain `Runner`, nothing exports until you register a tracer provider. -This guide covers ADK for Python 2.10 and later. ADK for Go and Kotlin emit the same span names, but their setup isn't covered here. +Tested with ADK for Python 2.10. ADK for Go and Kotlin emit the same span names, but their setup isn't covered here. ## Quick setup with a coding agent @@ -25,7 +25,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is under **Settings → Ingestion**. EU organizations should say EU region. ## Install ADK and the OTLP exporter @@ -33,9 +33,9 @@ Use your key from **Settings → Ingestion**. Without one, the agent uses a plac pip install "google-adk>=2.10" litellm opentelemetry-exporter-otlp-proto-http ``` -`litellm` is only needed for non-Gemini models through ADK's `LiteLlm` wrapper. ADK 2.10 pins `opentelemetry-sdk` to 1.42.1 or lower, so let pip pick the exporter version that matches instead of pinning a newer one. +`litellm` is only needed for non-Gemini models through ADK's `LiteLlm` wrapper. Don't pin a newer OpenTelemetry version: ADK 2.10 caps `opentelemetry-sdk` at 1.42.1. -### Configure the export with environment variables +## Configure the export and the transcript format ```bash OTEL_SERVICE_NAME=support-agent @@ -48,16 +48,13 @@ OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false ``` -- The `%20` is an encoded space. OpenTelemetry header values are URL-encoded, and the Python SDK decodes it back to `Bearer YOUR_INGEST_KEY`. -- The exporter appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. If you set `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` instead, give the full `https://ingest.maple.dev/v1/traces`. -- The `proto-http` exporter always sends OTLP over HTTP with protobuf, which is what Maple ingests. Don't install the gRPC exporter. -- For an EU organization, use `https://ingest.eu.maple.dev`. +The `%20` is an encoded space; the Python SDK decodes it. EU organizations use `https://ingest.eu.maple.dev`. Use `SPAN_ONLY` exactly: `true` sends content to log records, which Maple doesn't read for the transcript. -The content variables are explained [below](#record-prompts-responses-and-tool-calls). +These settings store every prompt and tool result in Maple. To keep structure and tokens without content, leave `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` unset and delete the two `gen_ai.tool.call.*` lines from the plugin below. -### Register a tracer provider when you run ADK with a Runner +## Register a tracer provider -`adk web` and `adk api_server` build a tracer provider from the `OTEL_EXPORTER_OTLP_*` variables on startup. A `Runner` in your own FastAPI app, worker or script does not: ADK's spans go to OpenTelemetry's no-op default and nothing is exported, with no warning. Register one yourself: +`adk web` and `adk api_server` build a tracer provider from the environment. A `Runner` in your own app, worker or script doesn't, and its spans go nowhere without a warning. Add a `telemetry.py`: ```py # telemetry.py @@ -104,7 +101,9 @@ provider.add_span_processor(SkipDuplicateToolSpans(OTLPSpanExporter())) trace.set_tracer_provider(provider) ``` -Import it as the first line of your entry point, before your agents and before the first `runner.run_async()`: +ADK doesn't record tool arguments or results, so the `ToolCallAttributes` plugin adds them. `SkipDuplicateToolSpans` drops two tool spans that would otherwise count one call twice. + +Import `telemetry` as the first line of your entry point and register the plugin on the runner: ```py # main.py @@ -138,13 +137,13 @@ runner = Runner( ) ``` -If your app already sets up OpenTelemetry (Logfire, Sentry, a platform agent), don't create a second provider. Add the `SkipDuplicateToolSpans(OTLPSpanExporter())` processor to the existing one, since only the first provider set globally wins. +If your app already sets up OpenTelemetry (Logfire, Sentry, a platform agent), add the `SkipDuplicateToolSpans(OTLPSpanExporter())` processor to that provider instead of creating a second one. -Under `adk web` or `adk api_server`, skip `telemetry.py`'s provider and rely on the environment variables. Register the plugin on your `App`, which is where the CLI looks for plugins: `App(name="support", root_agent=agent, plugins=[ToolCallAttributes()])`. The `SkipDuplicateToolSpans` processor can't be added there, since ADK owns that provider, so parallel calls and approval pauses add extra tool spans. +Under `adk web` or `adk api_server`, skip the provider and register the plugin on your `App`: `App(name="support", root_agent=agent, plugins=[ToolCallAttributes()])`. -## Group turns into one session with the ADK session id +## Use one ADK session per conversation -ADK stamps `session.id` as `gen_ai.conversation.id` on every `invoke_agent` and `generate_content` span, and Maple groups traces by it. Each `runner.run_async()` call is one trace and one turn, so a conversation is one Maple session as long as every turn uses the same ADK session: +Each `runner.run_async()` call is one turn. Pass your own conversation id as `session_id` on every turn, and the whole conversation is one Maple session: ```py async def chat(conversation_id: str, user_id: str, text: str) -> str: @@ -159,76 +158,13 @@ async def chat(conversation_id: str, user_id: str, text: str) -> str: return reply ``` -Use your own thread or chat id as the ADK session id. With `auto_create_session=True`, the runner creates the session on the first turn under that id. A persistent session service (`DatabaseSessionService`, `VertexAiSessionService`) keeps it across restarts. - -The common mistake is calling `create_session()` on every request without a `session_id`. ADK then mints a fresh UUID per turn, the model forgets the conversation, and Maple shows one single-turn session per message. Use `run_async()` rather than the synchronous `runner.run()` in servers. - -## Record prompts, responses and tool calls - -ADK has two content paths, and only one of them reaches Maple: - -| Setting | What it does | Default | -| --- | --- | --- | -| `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental` | Switches `generate_content` spans to the current GenAI conventions | off | -| `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` | Puts `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions` and `gen_ai.tool.definitions` on the `generate_content` span | no content | -| `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS` | ADK's own `gcp.vertex.agent.*` JSON attributes on `call_llm` and `execute_tool` spans | on | - -With both OpenTelemetry settings, every `generate_content` span carries the full request history and the reply as `[{role, parts}]` JSON: user text, assistant text, `tool_call` parts with arguments and `tool_call_response` parts with results. That's the shape Maple's transcript renders. - -- Without the stability opt-in, content goes to OpenTelemetry log records as ``, and Maple reads neither. -- A value of `true` for the capture variable means `EVENT_ONLY` for backward compatibility: log records, not spans. Use `SPAN_ONLY`. -- `ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false` only removes ADK's legacy copies, which Maple ignores. Without it, each model call sends its full history twice. - -Tool spans don't get arguments or results from ADK in any mode yet. The `ToolCallAttributes` plugin from the setup adds them as `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, so the tool pages in Maple show what each call received and returned. ADK runs tool callbacks inside the tool's span, which is why `trace.get_current_span()` is the right span there. - -Both settings are read per run, so you can also set them per request with `RunConfig(telemetry=TelemetryConfig(genai_semconv_stability_opt_in="experimental", capture_message_content=ContentCapturingMode.SPAN_ONLY))` from `google.adk.telemetry.context`. - -### Privacy - -Everything a user types and every tool result is stored in Maple once content capture is on. To keep the structure, tokens and tool names but no content, leave `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` unset and delete the two `gen_ai.tool.call.*` lines from the `ToolCallAttributes` plugin. Keep the plugin itself, since it also marks approval pauses. To redact instead, scrub values in a tool callback or in an OpenTelemetry Collector with the `redaction` or `transform` processor before the data leaves your network. - -## Tool errors, sub-agents and agents as tools - -Tool spans are `execute_tool {tool name}` with `gen_ai.tool.name`, `gen_ai.tool.description` and `gen_ai.tool.call.id` matching the model's tool call. ADK 2.7 and later mark a failed tool call with span status ERROR and `error.type` in two cases: - -- **The tool raises.** The exception is recorded on the span, and the whole run fails unless something handles it. -- **The tool returns a dict with a non-empty `"error"` key.** ADK sets `error.type=TOOL_ERROR`. This is the way to report an expected failure to the model and still see it in Maple. - -ADK's tutorials often return `{"status": "error", "error_message": "..."}`. ADK doesn't recognize that shape, so those calls show as successful. Use `{"error": "..."}`. - -If a tool raising should not end the run, let the model see the error instead with an `on_tool_error_callback`. The span keeps its ERROR status because the returned dict has an `error` key: - -```py -class ToolErrorsAsResults(BasePlugin): - def __init__(self): - super().__init__(name="tool_errors_as_results") - - async def on_tool_error_callback(self, *, tool, tool_args, tool_context, error): - return {"error": str(error)} -``` - -A tool that asks for confirmation (`FunctionTool(func, require_confirmation=True)`) gets two `execute_tool` spans with the same call id: one when ADK pauses the call, and one when the approved call runs in the next `run_async()`. The pause span isn't marked failed, but Maple would count it as a second call. The plugin marks it and `SkipDuplicateToolSpans` drops it, so each approved call counts once, in the trace where it ran. Send the approval with the same `session_id` and it stays in the same session. If the user rejects the call, the span is marked failed with ADK's "This tool call is rejected" result. - -### Sub-agents - -Every agent run is an `invoke_agent {agent name}` span with `gen_ai.agent.name`, and each model call's `generate_content` span repeats the agent name, so Maple opens one lane per agent. `SequentialAgent`, `ParallelAgent` and `LoopAgent` nest their children's `invoke_agent` spans, and `ParallelAgent` children run concurrently as sibling spans. +With `auto_create_session=True`, the runner creates the session under that id on the first turn. Don't call `create_session()` without a `session_id` on each request; ADK then mints a new id per turn and every message becomes its own session. -To call an agent like a tool, add it to `sub_agents` with `mode="single_turn"`. ADK runs it inside the parent's session. Each delegation produces an `execute_tool {agent name}` span, with the sub-agent's reply as its result, and a sibling `invoke_agent {agent name}` span. Both sit directly under the parent agent, so Maple opens a lane for the sub-agent and counts the `execute_tool` span as one tool call. When the model delegates to several agents in one response, they run in parallel. +To call an agent like a tool, add it to `sub_agents` with `mode="single_turn"`. `AgentTool` runs the sub-agent under a second session id, which can move the turn into a separate session. -Avoid wrapping agents in `AgentTool`. It runs the sub-agent in a new in-memory session with a new id, so its spans carry a second `gen_ai.conversation.id` inside your trace. Maple keeps one id per trace and picks the larger of the two, which can move that turn into a session of its own. ADK's API docs also discourage `AgentTool` in favor of `mode="single_turn"`. +## Flush before a short-lived process exits -## Tokens and cost - -`generate_content` spans carry `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.reasoning.output_tokens` when the model reports them. ADK counts cached tokens inside the input and thinking tokens inside the output, which is how Maple adds them up. - -- Streamed turns (`RunConfig(streaming_mode=StreamingMode.SSE)`) report usage too. `LiteLlm` requests it with `stream_options.include_usage`. -- The `call_llm` span above each `generate_content` repeats the same usage. Maple nets a parent's usage against its children, so tokens and the LLM call count cover each call once. -- The session checks don't net yet. Headlines such as "All 16 model calls were answered first time" count the `call_llm` and `generate_content` spans of each call separately, so they show twice the LLM call count. -- ADK doesn't record cost, and Maple doesn't price tokens, so ADK sessions show as **unpriced**. LiteLLM computes a cost, but it never reaches ADK's spans. - -## Flush spans before a short-lived process exits - -`BatchSpanProcessor` sends spans every 5 seconds. A script, CLI, notebook cell or job that exits sooner loses the last turn. Flush and shut down the provider when the work ends: +`BatchSpanProcessor` sends spans every 5 seconds, so a script, notebook cell or job that exits sooner loses the last turn. Flush when the work ends: ```py # script.py @@ -250,44 +186,21 @@ async def main(): asyncio.run(main()) ``` -In a long-running server, call `provider.shutdown()` from your shutdown hook (FastAPI `lifespan`, for example). On Cloud Run or another platform that freezes the CPU between requests, call `provider.force_flush()` before the response returns, or background export stalls until the next request. +In a server, call `provider.shutdown()` from your shutdown hook. On Cloud Run or another platform that freezes the CPU between requests, call `provider.force_flush()` before each response returns. ## Check that it works -Run one conversation of two or three turns, one of which calls a tool, then open **Agent Sessions** in Maple. Spans leave the process within 5 seconds and usually show up within a minute. You should see: +Run a conversation of two or three turns, one of them calling a tool, then open **Agent Sessions**. You should see one session with the framework **Google ADK**, one turn per `run_async()`, a transcript with the tool calls, their arguments and results, and tokens on each model call. -- **One session per ADK session id**, with the framework shown as **Google ADK** and one turn per `run_async()` call. The `run_async()` that sends a confirmation is a turn of its own, labeled with the original request. -- **A transcript** on the session page: the user messages, the assistant replies, and each tool call with its arguments and result. -- **LLM calls and tokens** for each `generate_content {model}` span, with the model from `gen_ai.request.model` (for LiteLLM, the full id such as `openrouter/openai/gpt-4o-mini`). -- **Tool calls** named after your functions, with failed ones counted as errors. -- **Cost** shown as unpriced. - -In the trace view, a turn looks like this: - -```text -invocation -└─ invoke_agent assistant - ├─ call_llm - │ └─ generate_content openrouter/openai/gpt-4o-mini - ├─ execute_tool get_weather - └─ call_llm - └─ generate_content openrouter/openai/gpt-4o-mini -``` +Cost shows as unpriced, because ADK doesn't record it. ## Troubleshooting -- **No spans at all.** You run ADK through a `Runner` and never registered a tracer provider. Only `adk web` and `adk api_server` build one from the environment. Import `telemetry.py` first. -- **Sessions show tokens but an empty transcript.** `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental` or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` is missing, or the capture variable is `true`, which means log records only. Both must be in the process environment before the run starts. -- **Every message is its own session.** Each request creates a new ADK session. Pass your conversation id as `session_id` on every turn. -- **One turn lands in a different session.** An `AgentTool` ran a sub-agent under its own session id. Use `sub_agents` with `mode="single_turn"`. -- **A tool named `(merged tools)`.** ADK's summary span for parallel tool calls. Add the `SkipDuplicateToolSpans` processor. -- **A confirmation-gated tool counts twice.** ADK records the paused call and the approved call as separate spans. Register `ToolCallAttributes` and `SkipDuplicateToolSpans` together: the plugin marks the pause and the processor drops it. -- **A failing tool shows as successful.** It returned `{"status": "error", ...}` or a string. Return a dict with an `"error"` key, or raise. -- **The streamed reply appears twice in the transcript, once in pieces.** With `StreamingMode.SSE`, ADK records every streamed chunk in `gen_ai.output.messages` and then the complete reply. The tokens are right; the transcript repeats the text. -- **Every model call appears twice, with doubled tokens.** Another instrumentor wraps the same calls: `litellm.callbacks = ["otel"]`, `openinference-instrumentation-google-adk`, or an OpenAI or LiteLLM instrumentor. ADK's own spans are enough; remove the others. -- **`gen_ai.system` says `gemini` for an OpenAI model.** ADK before 2.7 hardcoded it. Upgrade. With the settings in this guide, `generate_content` spans carry no provider attribute, which doesn't affect grouping, tokens or the transcript. -- **The model name has a provider prefix.** ADK records the requested model (`openrouter/openai/gpt-4o-mini`), never the model that served the response. It also records no `gen_ai.response.id`, so Maple can't merge a call that a second instrumentor reports again. That's one more reason to keep ADK's spans as the only ones. -- **401 from the exporter.** The header must read `Authorization=Bearer%20`, with the key from **Settings → Ingestion** for the right region. +- **No spans at all.** You run a `Runner` without a registered tracer provider. Import `telemetry.py` first. +- **Tokens but an empty transcript.** `OTEL_SEMCONV_STABILITY_OPT_IN` or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` is missing, or the capture variable is `true`. +- **Every message is its own session.** Pass the same conversation id as `session_id` on every turn. +- **A failing tool shows as successful.** It returned `{"status": "error", ...}`. Return a dict with an `"error"` key, or raise. +- **Every model call appears twice.** Another instrumentor wraps the same calls (`litellm.callbacks = ["otel"]`, `openinference-instrumentation-google-adk`). Remove it. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/haystack.md b/apps/landing/src/content/docs/agent-tracing/haystack.md index 02064d5bdc..17ec752d1a 100644 --- a/apps/landing/src/content/docs/agent-tracing/haystack.md +++ b/apps/landing/src/content/docs/agent-tracing/haystack.md @@ -1,15 +1,15 @@ --- title: "Trace Haystack agents with OpenTelemetry" -description: "Send Haystack 3 Agent and pipeline runs to Maple as one Agent Session per conversation, with the transcript, model, tokens, cost and failed tool calls." +description: "Send Haystack 3 Agent and pipeline runs to Maple as one Agent Session per conversation, with transcript, model, tokens, cost and failed tool calls." group: "AI Agents" order: 28 navLabel: "Haystack" icon: "haystack" --- -Haystack 3 traces its own pipelines, components, Agent steps and tool calls through the `opentelemetry-haystack` tracer. Those spans use a private `haystack.*` vocabulary: the model id and token counts exist only inside a JSON blob of the model's reply, a failed tool call ends with status `Unset`, and there is no conversation id anywhere. Sent to Maple as-is, every Haystack run shows up with the right shape but with no model calls, no tokens, no transcript, no tool failures, and one session per request. +Haystack 3 traces its pipelines, components, Agent steps and tool calls through `opentelemetry-haystack`, but it keeps the model, tokens and messages inside `haystack.*` JSON blobs Maple doesn't read, leaves failed tool calls unmarked and has no conversation id. This guide adds `maple_haystack.py`, a subclass of Haystack's tracer that writes the GenAI attributes Maple reads, and a `conversation()` block that groups runs into one session. -This guide keeps Haystack's own spans and adds a single-file tracer of about 120 lines, `maple_haystack.py`, that writes the OpenTelemetry GenAI attributes Maple reads onto them while they are still open. It covers Python 3.10+ with `haystack-ai` 3.2 and `opentelemetry-haystack` 1.0, for the `Agent` component, `AgentTool`/`PipelineTool` sub-agents and chat generators in plain pipelines. +Tested with `haystack-ai` 3.2 and `opentelemetry-haystack` 1.0 on Python 3.10 or later. ## Quick setup with a coding agent @@ -23,28 +23,16 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is under **Settings → Ingestion**. -## Why not the plain Haystack tracer or OpenInference - -There are three ready-made ways to get OpenTelemetry spans out of Haystack. None of them gives Maple a complete session on its own: - -| Option | What Maple gets | What is missing | -| --- | --- | --- | -| `OpenTelemetryTracer` from `opentelemetry-haystack` | Pipeline, component, Agent, step and tool spans, detected as **Haystack** | Model, tokens, transcript and tool failures (all inside `haystack.*` blobs Maple does not read), session id | -| `openinference-instrumentation-haystack` | Model and tokens on generator spans | Tool spans (tools run inside the Agent, which it records as one opaque chain), session id (`using_session` writes `session.id`, which Maple ignores for this dialect), framework label ("Unidentified") | -| OpenLLMetry `opentelemetry-instrumentation-haystack` | `Pipeline.run`, plus `OpenAIGenerator` and `OpenAIChatGenerator` calls | Agent steps, tool spans, tokens, every other generator (OpenRouter, Anthropic, ...), content in indexed `gen_ai.prompt.N.*` keys Maple does not read, session id | - -The tracer in this guide is a subclass of the first option. It keeps every span and tag Haystack emits, so nothing you already see in Maple's trace view changes, and adds `gen_ai.*` attributes next to them. Don't run it together with the OpenInference or OpenLLMetry instrumentor: each would record every model call a second time. - -## Install and export to Maple +## Install and add the Maple tracer ```bash pip install "haystack-ai>=3.2" "opentelemetry-haystack>=1.0" \ "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Save this file next to your app as `maple_haystack.py`: +Save this file next to your app as `maple_haystack.py`. It keeps every span and tag Haystack emits and adds `gen_ai.*` attributes to the Agent, model and tool spans: ```py """Haystack tracer that adds the OpenTelemetry GenAI attributes Maple reads.""" @@ -170,16 +158,9 @@ class MapleHaystackTracer(OpenTelemetryTracer): yield span ``` -It maps Haystack's spans as follows: - -| Haystack span | Maple reads | -| --- | --- | -| `haystack.agent.run` | `invoke_agent`, agent name from the pipeline component or `AgentTool` that runs it | -| `haystack.agent.step.llm`, and any `*ChatGenerator` component | `chat`: model, finish reason, input/output/cache/reasoning tokens, cost, messages | -| `haystack.agent.step.tool` | `execute_tool`: tool name, arguments, result, and `Error` status when the tool failed | -| every span | `gen_ai.conversation.id` inside a `conversation()` block | +## Export to Maple -Then configure OpenTelemetry once at startup and hand Haystack the tracer: +Configure OpenTelemetry once at startup and hand Haystack the tracer: ```py # telemetry.py @@ -207,20 +188,13 @@ export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" ``` -For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` to the endpoint itself. +EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. -Import `telemetry` before the first `pipeline.run()` or `agent.run()`. Unlike Haystack's content switch, the tracer has no import-order trap: Haystack looks up the active tracer on every span, so any run after `enable_tracing()` is traced. - -Two details matter here: - -- **Keep the tracer name `"haystack"`.** It becomes the instrumentation scope, which is how Maple labels the sessions as Haystack. Maple also recognizes Haystack's span names, but the scope covers spans added in newer Haystack releases too. -- **If the app already has a `TracerProvider`** (from FastAPI instrumentation or another library), skip the provider lines and pass `trace.get_tracer("haystack")` from the existing one. A second provider sends every span twice or not at all. +Import `telemetry` before the first `pipeline.run()` or `agent.run()`. Keep the tracer name `"haystack"`, because Maple uses it to label the sessions as Haystack. If the app already has a `TracerProvider`, skip the provider lines and pass `trace.get_tracer("haystack")` from the existing one. ## Group turns into one session -Haystack's `Agent` is stateless: your app keeps the message history and passes it into every run, so nothing on the wire says that two runs belong to the same chat. Each `pipeline.run()` is its own trace, and without a conversation id Maple shows each one as a separate one-turn session. - -Wrap every run in `conversation()` with your own chat id: +Haystack's `Agent` is stateless: your app passes the message history into every run, and each run is its own trace. Wrap every run in `conversation()` with your chat's id: ```py from haystack import Pipeline @@ -243,88 +217,31 @@ def handle_message(chat_id: str, history: list[ChatMessage], text: str) -> list[ return [m for m in result["assistant"]["messages"] if not m.is_from("system")] ``` -Use the id your app already stores for the chat, such as a database row id or a thread id from your frontend. Don't mint a new UUID per request, and don't use one id for the whole process: both break the grouping, in opposite directions. - -`conversation()` is a `ContextVar`, so it is scoped to the current request in async and threaded servers alike, and it follows the Agent into the worker threads that run tools in parallel. It writes `gen_ai.conversation.id` on every span of the run. Maple needs it on at least one span per trace to join that trace to the session. - -## Record prompts, responses and tool calls +Use the id your app already stores for the chat, such as a database row id or a frontend thread id. A new UUID per request gives one session per message, and one id for the whole process merges every user into one session. `conversation()` uses a `ContextVar`, so it stays scoped to the request in async and threaded servers and follows the Agent into its tool threads. -With the default `content=True`, the tracer writes: +## Agent names, content and tokens -- `gen_ai.input.messages` and `gen_ai.output.messages` on every model call, as `{role, parts}` arrays with text, tool calls and tool results; -- `gen_ai.system_instructions` with the Agent's system prompt; -- `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result` on every tool span; -- Haystack's own `haystack.*.input`/`.output` tags, as before. +Maple names each agent, and gives it a lane, from the pipeline component that runs it (`assistant` above) or from the `name=` of the `AgentTool` that wraps it. An Agent run directly with `agent.run()` is called `agent`. -Maple builds the transcript and the turn labels from the `gen_ai.*` messages. The `haystack.*` blobs only show up in the raw span attributes. - -Maple currently shows a tool span's result only when it is a JSON object or array. A tool that returns plain text, like `"Sunny, 21°C"`, still has its result in the transcript, as the tool result the next model call receives, but its tool call row shows no result. Return a dict from tools whose results you want on the tool pages. - -`HAYSTACK_CONTENT_TRACING_ENABLED` has no effect with this tracer: `MapleHaystackTracer` decides on its own. Without the tracer, that variable is read once, at the first `import haystack`, and setting it any later silently records nothing. - -To keep prompts and responses out of Maple, turn content off: +To keep prompts and responses out of Maple, pass `content=False`: ```py tracing.enable_tracing(MapleHaystackTracer(trace.get_tracer("haystack"), content=False)) ``` -Model, tokens, cost, finish reasons, tool names and tool failures are still recorded, because they come from the reply metadata rather than the text. The error message of a failed tool stays on the span status, and Haystack's message quotes the call's arguments (``Failed to invoke Tool `fetch_transport_data` with parameters {'city': 'Rome'}``). If arguments can carry personal data, redact them in `_tool()` before `set_status`. - -Haystack also puts the whole pipeline input on the root span as `haystack.pipeline.input_data`, a plain tag that its own content switch never gated. With `content=False` the tracer drops it, so the user's message doesn't leak through the back door. To redact rather than drop, filter the values in `_messages()` before they are written. - -## Tools, errors and sub-agents - -Each tool call is a `haystack.agent.step.tool` span with `gen_ai.tool.name` set to the tool's name. Tools of one step run in parallel threads, and their spans sit side by side under the step. - -When a tool raises, Haystack wraps the exception in `ToolInvocationError`, feeds the error text back to the model, and writes `{"error": "..."}` as the tool output. The span itself ends with status `Unset`, so without the tracer the failure is invisible. The tracer sets status `Error` with the message and `error.type=ToolInvocationError`, which is what Maple counts as a failed tool call. This happens with both values of `raise_on_tool_invocation_failure`. - -Maple opens a lane for every agent with a distinct `gen_ai.agent.name`. The tracer takes that name from what runs the Agent: - -- an Agent added to a pipeline gets its component name, `assistant` in the example above; -- an Agent wrapped in `AgentTool(agent=..., name="weather_worker", ...)` gets the tool name, and Maple shows the tool call as a delegation to that sub-agent; -- an Agent inside a `PipelineTool` gets its component name in the inner pipeline; -- an Agent run directly with `agent.run()` is called `agent`. +Model, tokens, cost, tool names and failures are still recorded. `HAYSTACK_CONTENT_TRACING_ENABLED` has no effect with this tracer. -A multi-agent setup with the orchestrator in a pipeline and workers as `AgentTool`s looks like this: - -```py -from haystack.tools import AgentTool - -weather_worker = AgentTool( - agent=Agent(chat_generator=chat_generator, tools=[get_weather], system_prompt="Report the weather."), - name="weather_worker", - description="Look up the current weather for a city.", -) -# budget_worker and transport_worker are built the same way -orchestrator = Agent(chat_generator=chat_generator, tools=[weather_worker, budget_worker, transport_worker]) -pipeline = Pipeline() -pipeline.add_component("orchestrator", orchestrator) - -with conversation(briefing_id): - pipeline.run({"orchestrator": {"messages": [ChatMessage.from_user("Brief me on Amsterdam.")]}}) -``` - -Approval gates (`ConfirmationHook` at `before_tool`) stay inside the same run and the same session. A call the user rejects never reaches the tool, so it has no tool span; the model sees the rejection as a tool result, which is visible in the transcript. - -## Tokens and cost - -Tokens come from the `usage` object Haystack's generators put in each reply's `meta`. The tracer reads the OpenAI field names (`prompt_tokens`, `completion_tokens`, `prompt_tokens_details.cached_tokens` and `.cache_write_tokens`, `completion_tokens_details.reasoning_tokens`), which is what `OpenAIChatGenerator`, `OpenRouterChatGenerator` and other OpenAI-compatible generators report. - -**Streaming needs one extra flag with OpenAI.** A streamed OpenAI response only carries usage when you ask for it, and Haystack doesn't: +A streamed `OpenAIChatGenerator` reply carries no token counts unless you ask for them (OpenRouter always sends usage): ```py OpenAIChatGenerator(model="gpt-4o-mini", generation_kwargs={"stream_options": {"include_usage": True}}) ``` -OpenRouter always sends usage in the last streamed chunk, so `OpenRouterChatGenerator` needs nothing extra. - -Maple never prices tokens itself. It shows cost only when a span carries one, and the only Haystack generator that reports a price is OpenRouter's (`usage.cost`, in USD), which the tracer copies to `gen_ai.usage.cost`. With any other provider, sessions show tokens and read as **unpriced**. +Cost only appears with `OpenRouterChatGenerator`, the one Haystack generator that reports a price. Other providers show tokens and read as **unpriced**. -The Agent itself reports no usage, so there is nothing to double count: each model call is counted once, on its `haystack.agent.step.llm` span. +## Flush before short-lived processes exit -## Short-lived processes - -`BatchSpanProcessor` exports in the background every 5 seconds. A script, CLI, notebook cell, cron job or serverless handler that exits sooner loses the last batch. Flush before the process ends: +`BatchSpanProcessor` exports every 5 seconds. A script, CLI, notebook, cron job or serverless handler that exits sooner loses the last batch: ```py try: @@ -338,41 +255,18 @@ Long-running servers only need `provider.shutdown()` in their shutdown hook. ## Check that it works -Run one conversation of at least two turns, one of them with a tool call. Within a minute, open **Agent Sessions** in Maple and filter by your service name. You should see: - -- **one session per conversation id**, labelled Haystack, with one turn per `pipeline.run()`; -- a transcript with the user's messages, the assistant's replies and the tool calls, each turn labelled with its user message; -- an LLM call per `haystack.agent.step.llm` span, with the model (for example `openai/gpt-4o-mini`) and input and output tokens, including the streamed turn; -- a tool call per `haystack.agent.step.tool` span, named after the tool, with a failed tool counted as an error; -- the agent name from your pipeline component or `AgentTool`, and a lane per sub-agent; -- cost on OpenRouter, or "unpriced" with other providers. - -Each turn's trace looks like this: - -```text -haystack.pipeline.run - haystack.component.run assistant - haystack.agent.run invoke_agent, agent "assistant" - haystack.agent.step - haystack.agent.step.llm chat, openai/gpt-4o-mini - haystack.agent.step.tool execute_tool, get_weather - haystack.agent.step - haystack.agent.step.llm chat -``` +Run a conversation of at least two turns, one with a tool call, and open **Agent Sessions** filtered by your service name. You should see one session per conversation id, labelled Haystack, with one turn per `pipeline.run()`. Each `haystack.agent.step.llm` span shows up as a model call with model and tokens, and each `haystack.agent.step.tool` span as a tool call, with failed tools marked. -An Agent with hooks, such as a `ConfirmationHook`, also gets a `haystack.agent.hook` span before its tool calls. Like `haystack.agent.step`, it carries no model or tool attributes, so Maple doesn't count it as a call. +A tool that returns plain text shows its result in the transcript but not on its tool call row. Return a dict to see the result there too. ## Troubleshooting -- **Every request is its own session.** The run happened outside a `conversation()` block, or each request passes a fresh id. Wrap the `pipeline.run()` call itself, and pass the chat's stored id. -- **Sessions show the right spans but no model calls, tokens or transcript.** Haystack is still using the plain `OpenTelemetryTracer`. Check that `enable_tracing()` gets a `MapleHaystackTracer`, and that no later call replaces it. -- **No Haystack spans at all.** Haystack 3 no longer turns tracing on when `opentelemetry-sdk` is installed. Call `tracing.enable_tracing(...)` before the first run. -- **Every model call appears twice.** The OpenInference or OpenLLMetry Haystack instrumentor is also active. Remove it; this tracer already records every model call. -- **The streamed turn has no tokens.** `OpenAIChatGenerator` without `stream_options.include_usage`. Add it to `generation_kwargs`. -- **Tokens are zero with a non-OpenAI generator.** Its `meta["usage"]` doesn't use the OpenAI field names. Print `result["replies"][0].meta["usage"]` once and add its keys to `_chat()`. -- **The agent is called `agent`.** It ran through `agent.run()`, not as a pipeline component or `AgentTool`. Add it to a `Pipeline` under the name you want to see. -- **A failed tool isn't counted.** The tool caught its own exception and returned a normal value. Let it raise, or return `{"error": ...}`, which Haystack uses for failures too. -- **Sessions are split or missing turns in a script.** The process exited before the batch was exported. Call `provider.force_flush()` and `provider.shutdown()` in `finally`. +- **Every request is its own session.** Wrap the `pipeline.run()` call itself in `conversation()`, and pass the chat's stored id. +- **Spans but no model calls, tokens or transcript.** `enable_tracing()` got the plain `OpenTelemetryTracer`, or a later call replaced `MapleHaystackTracer`. +- **No Haystack spans at all.** Haystack 3 doesn't trace until you call `tracing.enable_tracing(...)`. Call it before the first run. +- **Every model call appears twice.** Remove `openinference-instrumentation-haystack` or OpenLLMetry's `opentelemetry-instrumentation-haystack`. +- **Tokens are zero with a non-OpenAI generator.** Print `result["replies"][0].meta["usage"]` once and add its keys to `_chat()`. +- **A failed tool isn't counted.** The tool caught its own exception. Let it raise, or return `{"error": ...}`. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/langchain.md b/apps/landing/src/content/docs/agent-tracing/langchain.md index 8b40c5743b..539d3069cc 100644 --- a/apps/landing/src/content/docs/agent-tracing/langchain.md +++ b/apps/landing/src/content/docs/agent-tracing/langchain.md @@ -1,17 +1,15 @@ --- title: "Trace LangChain and LangGraph agents with OpenTelemetry" -description: "Send LangChain and LangGraph runs to Maple as Agent Sessions, one per thread, with a readable transcript, model and tool calls, tokens, failed tools and sub-agent lanes." +description: "Send LangChain and LangGraph runs to Maple with OpenInference, one Agent Session per thread." group: "AI Agents" order: 13 navLabel: "LangChain & LangGraph" icon: "langchain" --- -LangChain and LangGraph report every run through their callback system: each graph node, chat model call and tool call starts and ends a run. Two libraries turn those runs into OpenTelemetry spans. OpenInference's `openinference-instrumentation-langchain` adds its own callback handler, and LangSmith's SDK can export its runs over OTLP instead of to smith.langchain.com. Both can export to Maple. This guide uses OpenInference with its GenAI dual-write, because that's the one that gives Maple a transcript it can render. +LangChain and LangGraph report every run through callbacks, and OpenInference's `openinference-instrumentation-langchain` turns those runs into OpenTelemetry spans. Every `invoke()` starts a new trace, so you have to pass the conversation's `thread_id` for Maple to group a chat into one session. -The default that goes wrong is the conversation. Every `invoke()` of an agent or graph is a new trace, so a chat of ten messages arrives as ten traces, and nothing links them until you pass a `thread_id`. The same `thread_id` a LangGraph checkpointer already needs is the one Maple groups sessions by. - -This guide covers Python: LangChain 1.4 (`create_agent`) and LangGraph 1.2 (`StateGraph`) with `openinference-instrumentation-langchain` 0.1.76, on Python 3.10 or later. +Tested with LangChain 1.4 (`create_agent`), LangGraph 1.2 and `openinference-instrumentation-langchain` 0.1.76 on Python 3.10 or later. LangChain.js isn't covered yet. ## Quick setup with a coding agent @@ -25,9 +23,9 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. -## Install the instrumentor and export to Maple +## Install the instrumentor ```bash pip install "langchain>=1.4" "langgraph>=1.2" "langchain-openai>=1.6" \ @@ -35,9 +33,9 @@ pip install "langchain>=1.4" "langgraph>=1.2" "langchain-openai>=1.6" \ "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Pin `openinference-instrumentation` explicitly. The LangChain instrumentor accepts versions back to 0.1.61, and the GenAI dual-write this guide depends on isn't in the oldest of them. +Pin `openinference-instrumentation` explicitly. Older versions lack the GenAI output this setup depends on. -Point the exporter at Maple with the standard OpenTelemetry variables: +## Point the exporter at Maple ```bash export OTEL_SERVICE_NAME=support-agent @@ -46,9 +44,11 @@ export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" ``` -EU organizations use `https://ingest.eu.maple.dev`. `OTLPSpanExporter()` with no arguments appends `/v1/traces` to the base URL. If you pass `endpoint=` in code instead, it's used as is and has to end in `/v1/traces`. +EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, it has to end in `/v1/traces`. + +## Initialize tracing -Then add a `tracing.py` and import it at the top of your entry point: +Add a `tracing.py` and import it at the top of your entry point, before the first `invoke()`: ```py # tracing.py @@ -89,20 +89,15 @@ LangChainInstrumentor().instrument( ) ``` -What each part does: - -- **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.conversation.id` next to its OpenInference attributes when each span ends. Without it, Maple reads the spans as generic OpenInference: the transcript is a raw JSON blob and the `thread_id` is ignored, so every turn is its own session. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, as long as it's set before `TraceConfig` is built. -- **`AgentSpans`** fills two gaps in how the spans are labeled. The instrumentor names agent spans only by the word "agent": a `create_agent(name="support_agent")` span becomes an agent, `name="assistant"` stays a plain chain, and neither gets `gen_ai.agent.name`, which Maple needs to name agents and open a lane per sub-agent. And a span with no operation is classified by its name, so LangGraph's `tools` node would count as one more tool call and a `ChatPromptTemplate` as one more model call. Marking them `invoke_workflow` keeps both out of the counts. The processor runs at span start, and the dual-write never overwrites a key that's already set. - -The instrumentor hooks LangChain's callback manager, so import order doesn't matter, as long as `instrument()` runs before the first `invoke()`. It traces every LangChain runnable in the process: agents, graphs, chains, chat models, tools and retrievers. +`enable_genai_semconv=True` is required. Without it, Maple ignores the `thread_id` and shows the transcript as raw JSON. -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second one. Add `AgentSpans()` and the OTLP exporter to the existing provider and pass that provider to `instrument()`. +`AgentSpans` names your agents so Maple gives each one its own lane. Put every agent's `name=` in `AGENT_NAMES`, sub-agents included. `STEP_NAMES` keeps graph steps out of the tool and model counts; if your tool node has another name containing "tool", like `run_tools`, add it there. -LangSmith keeps working next to this. With `LANGSMITH_TRACING=true` and a LangSmith key, runs still go to smith.langchain.com over LangSmith's own API, and nothing is sent twice to Maple. Don't also set `LANGSMITH_OTEL_ENABLED`, which would add a second copy of every span to your provider (see [why not LangSmith's exporter](#why-not-langsmiths-opentelemetry-exporter) below). +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `AgentSpans()` and the exporter to that provider and pass it to `instrument()`. -## Group a conversation into one session with thread_id +## Group a conversation with thread_id -Maple groups traces into a session by `gen_ai.conversation.id`. The instrumentor sets it on every span of a run from the run's metadata, taking the first of `session_id`, `conversation_id` and `thread_id`. LangGraph copies `configurable` values into that metadata, so the `thread_id` you already pass for a checkpointer is enough: +Maple groups turns into a session by `gen_ai.conversation.id`, which the instrumentor fills from the run's `thread_id`. Pass your app's conversation id on every `invoke()`, `stream()` and `Command(resume=...)`: ```py import tracing # first, before the first invoke() @@ -127,95 +122,15 @@ def handle_message(conversation_id: str, text: str) -> str: return result["messages"][-1].content ``` -The checkpointer isn't what sets the session. A graph compiled without one still gets `gen_ai.conversation.id` from `configurable.thread_id`, so a stateless pipeline, like a multi-agent graph run once per request, can pass a `thread_id` only for tracing. A plain LangChain chain (a `prompt | model` pipeline, no graph) doesn't copy `configurable` into the instrumentor's metadata, so pass the id as metadata instead: `chain.invoke(inputs, {"metadata": {"thread_id": conversation_id}})`. - -Use your app's own conversation id, the one it stores the chat under. A new UUID per request gives you one session per message again, and a constant puts every user in one session. - -If you skip it, every `invoke()` shows up in **Agent Sessions** as its own one-turn session named after its trace id. A human-in-the-loop resume (`Command(resume=...)`) is a new `invoke()` and a new trace too, and the shared `thread_id` is the only thing that puts it back in the same session as the turn it interrupted. - -## Record prompts, responses and tool calls - -Content capture is on by default. Every chat model span carries the full message list sent to the model and the reply, as `gen_ai.input.messages` and `gen_ai.output.messages` in `{role, parts}` form, and Maple renders them as the session transcript. The system prompt is the first input message. Tool spans carry the tool's result in `gen_ai.tool.call.result`, as LangChain's serialized `ToolMessage`, so the result reads as a small JSON object with the output under `data.content`. - -Turn labels are the one part that goes wrong. Maple labels a turn with the user message on the first span of the turn that has messages, and in a `create_agent` graph that's the `model` node's span, where the instrumentor records only the thread's first message. With a checkpointer, every turn of a conversation is labeled with its opening message. The transcript inside each turn still starts with the right message. - -With a checkpointer, every model span repeats the thread's whole history, so a long conversation gets large. Maple has no per-attribute limit, and ingest accepts requests up to 20 MiB. - -To keep prompts and outputs out of your traces: - -```py -config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) -``` - -`hide_inputs` and `hide_outputs` drop the messages and replace `input.value` and `output.value` with `__REDACTED__`. The GenAI attributes are built from the masked values, so they're empty too. The session still shows its turns, model and tool calls, tokens and failures, with an empty transcript. Each switch has an `OPENINFERENCE_HIDE_*` environment variable, and narrower ones exist: `hide_input_text` and `hide_output_text` keep the message structure but redact the text. For pattern-based redaction, use the `redaction` processor in an OpenTelemetry Collector. - -## Tools, errors and sub-agents - -Each tool call is a span named after the tool, with `gen_ai.operation.name` `execute_tool`, `gen_ai.tool.name`, `gen_ai.tool.description` and the result. The arguments aren't on the tool span, and neither is `gen_ai.tool.call.id`, because the instrumentor doesn't record them there. The arguments are still in the transcript, in the model's tool call just before. - -A tool that raises is marked failed without extra code: its span ends with status `ERROR` and the exception plus its traceback as the status message, starting with `RuntimeError('transport data service unavailable (503)')`, and Maple counts it on the session and on the tool's page. - -What happens to the run next is up to LangChain. `create_agent` re-raises any exception from a tool by default, so one broken tool fails the whole `invoke()`. To hand the error to the model and keep going, add a `wrap_tool_call` middleware: - -```py -from langchain.agents.middleware import wrap_tool_call -from langchain_core.messages import ToolMessage - - -@wrap_tool_call -def tool_errors_to_model(request, handler): - try: - return handler(request) - except Exception as e: - return ToolMessage(content=f"Tool error: {e}", tool_call_id=request.tool_call["id"], status="error") - - -agent = create_agent(model, tools=tools, name="assistant", middleware=[tool_errors_to_model]) -``` - -The tool span is still marked failed, because the tool's own run ends with the exception before the middleware catches it. In a `StateGraph` with LangGraph's `ToolNode`, `ToolNode(tools, handle_tool_errors=True)` does the same. - -Human-in-the-loop interrupts aren't errors. `interrupt()` and `HumanInTheLoopMiddleware` pause the graph by raising `GraphInterrupt`, and the instrumentor ends that span with status `OK`. The pause ends the turn's trace, and the resume starts a new one, so an approved action shows as two turns in the same session: the request, then the resumed tool call and reply. - -Sub-agents need the `AgentSpans` processor from the setup. Add every agent's name to `AGENT_NAMES`, and Maple shows each one in its own lane with its model and tool calls. The common LangChain pattern, a worker agent called from a tool of the orchestrator, looks like this: - -```py -AGENT_NAMES = {"orchestrator", "weather_worker"} # in tracing.py - -weather_worker = create_agent(model, tools=[get_weather], name="weather_worker") - - -@tool -def ask_weather_worker(city: str) -> str: - """Ask the weather worker for the current weather in a city.""" - result = weather_worker.invoke({"messages": [{"role": "user", "content": f"Weather in {city}?"}]}) - return result["messages"][-1].content - - -orchestrator = create_agent(model, tools=[ask_weather_worker], name="orchestrator") -``` - -The worker's run nests under the tool span and inherits the orchestrator's `thread_id`, so it lands in the same trace and session without passing anything. Maple shows the `ask_weather_worker` call as a delegation to `weather_worker`, with the tool's argument and result as the lane's input and output. In a `StateGraph` where each worker is a node, list the node names instead. - -Don't put "agent" in a tool's name. The instrumentor names a span of any kind an agent when its name contains the word, so a tool called `ask_weather_agent` loses its tool kind and Maple doesn't count it as a tool call. - -LangGraph runs tools inside a node, named `tools` in `create_agent` and usually in a `StateGraph` with a `ToolNode` too. Without an operation, Maple would count that node's span as a tool call because its name contains "tool", which is why `STEP_NAMES` lists it. If your tool node has another name with "tool" in it, like `run_tools`, add that name. - -The missing `gen_ai.tool.call.id` means Maple matches tool spans to the model's calls by name, which works unless one reply calls the same tool twice. +This works with or without a checkpointer. A plain chain (`prompt | model`, no graph) doesn't read `configurable`, so pass `{"metadata": {"thread_id": conversation_id}}` instead. Agents called from inside a tool inherit the caller's id. -## Tokens and cost +Use the id your app stores the chat under. A new UUID per request gives you one session per message. -Every chat model span carries input and output tokens from LangChain's `usage_metadata`, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` next to the OpenInference `llm.token_count.*` originals. The model is the one you configured, in `gen_ai.request.model`, and the provider comes from the LangChain integration: `ChatOpenAI` is `openai`, even for an Anthropic model behind OpenRouter or another OpenAI-compatible gateway. +`stream_usage=True` makes `ChatOpenAI` report tokens on streamed replies when it talks to a server other than api.openai.com, such as a custom `base_url`, vLLM or a gateway. -Streaming is where tokens go missing. `ChatOpenAI` only asks for usage on streamed responses (`stream_options.include_usage`) when it talks to api.openai.com. With a `base_url`, or `OPENAI_BASE_URL` set, it doesn't, and a server that only reports streamed usage when asked, such as vLLM, sends none. OpenRouter includes usage either way. Set `stream_usage=True` on the model, as in the example above, so it doesn't depend on the server. +## Flush in short-lived processes -Maple shows cost only when a span carries one, and neither LangChain nor the instrumentor records cost. Sessions show as **unpriced**, with token counts. - -Don't add `openinference-instrumentation-openai`, `-anthropic` or OpenLLMetry's LangChain instrumentor next to this one. Each wraps the same model request again, and every call gets a second model span with its own tokens. - -## Flush spans before the process exits - -The instrumentor ends each span when the run's callback fires, and `BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` registers an `atexit` handler that flushes on a normal interpreter exit, which covers most scripts and CLIs. It doesn't run when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: +The SDK flushes on a normal interpreter exit, and long-running servers need nothing. In serverless handlers, notebooks and task workers, flush after each run: ```py from tracing import provider @@ -226,59 +141,21 @@ finally: provider.force_flush() # serverless: before returning; notebooks: after each run ``` -Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. - -Context survives LangGraph's parallel nodes, `ainvoke`, and agents called from inside tools on Python 3.11 and later. On Python 3.10, asyncio doesn't carry context into tasks, so pass the node's `config` to every nested `ainvoke()` or its runs start new traces. If you run LangChain code in your own thread pool, use `ContextThreadPoolExecutor` from `langchain_core.runnables.config` instead of the standard library's. - -## LangGraph Server deployments - -On LangGraph's Agent Server (`langgraph dev` or a self-hosted server), the server imports the module that defines your graph, the one `langgraph.json` points to. Import `tracing` at the top of that module and set the `OTEL_*` variables in the server's environment (the `env` file in `langgraph.json`, or the container). The server is long-running, so `BatchSpanProcessor` exports on its own schedule and no flush is needed. We checked this with `langgraph dev` (langgraph-api 0.10.3). - -Every run on a server thread already carries that thread's id as `configurable.thread_id`, so each LangGraph thread becomes one Maple session with no extra code. If `tracing.py` lives outside the graph's directory, list its directory in `dependencies` in `langgraph.json` so the server can import it. - -## Why not LangSmith's OpenTelemetry exporter - -LangSmith's SDK can write its runs as OTLP spans (`LANGSMITH_OTEL_ENABLED` with `LANGSMITH_OTEL_ONLY`, no LangSmith account needed). Maple recognizes those spans, labels them **LangChain**, and reads the session from `langsmith.metadata.thread_id`, the same `configurable.thread_id`. Sessions, token counts and `gen_ai.tool.call.id` all arrive. We ran the same chat through it with `langsmith` 0.14.1, and the session page is worse on every other count: - -- The prompt and completion are sent as byte attributes holding LangChain's serialized objects. Maple stores bytes as hex, so the transcript is unreadable and turns have no labels. -- An interrupt marks the interrupted node's span `ERROR`, with `GraphInterrupt(...)` as an `exception` event, so every human-in-the-loop pause reads as a failure. -- Middleware wrappers such as `HumanInTheLoopMiddleware.wrap_tool_call` get their own spans, and Maple counts them, and the `tools` node, as extra tool calls. A tool the model called once shows up to three times. -- There are no agent names. `create_agent`'s name is only in `langsmith.metadata.lc_agent_name`, which Maple doesn't read, and a start-time processor can't fix it, because LangSmith sets `gen_ai.operation.name` after the span starts. -- Flushing takes two steps: `wait_for_all_tracers()` from `langchain_core.tracers.langchain`, then `provider.force_flush()`. The provider alone exports nothing, because LangSmith converts runs to spans on a background thread. - -If `LANGSMITH_OTEL_ENABLED` or `LANGSMITH_TRACING_MODE=otel` is already set in your app, remove it when you add the OpenInference setup, or every run arrives twice. - -## LangChain.js and LangGraph.js - -This guide doesn't cover JavaScript yet. LangSmith's OpenTelemetry mode in JS is experimental and its `initializeOTEL()` setup is deprecated, and the JS OpenInference instrumentor has no GenAI dual-write, so Maple ignores its `session.id` and shows one session per trace. For a TypeScript agent today, see [Trace your AI agent](/docs/agent-tracing) for the frameworks that are covered. +On LangGraph Server (`langgraph dev` or self-hosted), import `tracing` at the top of the graph module that `langgraph.json` points to, and set the `OTEL_*` variables in the server's environment. Each server thread already carries its `thread_id`. ## Check that it works -Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. Spans usually show up within a minute. You should see: - -- **One session** for the conversation, with one turn per `invoke()`. A second conversation with a different id is a second session. -- **Framework: Unidentified.** Maple recognizes LangChain by LangSmith's exporter, and the OpenInference spans read as generic GenAI spans. Everything else on the session page works. -- **The transcript**: your messages, the model's replies, and its tool calls. Every turn's label repeats the conversation's first message (see [Record prompts, responses and tool calls](#record-prompts-responses-and-tool-calls)); the second conversation's single turn is labeled correctly. -- **Agents**: `assistant`, plus one lane per sub-agent in `AGENT_NAMES`. Each turn's trace starts at the agent span, with `model` and `tools` node spans below it. -- **Model calls** named `ChatOpenAI` (or `ChatAnthropic`, and so on), each with a model, input and output tokens, including streamed ones. -- **Tool calls** named after your tools, like `get_weather`, with results, and a failing tool marked failed with its message. -- **Cost**: unpriced. +Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session with one turn per `invoke()`, a readable transcript, `ChatOpenAI` model calls with tokens, and tool calls named after your tools. -If a turn is missing, check that the process flushed. +The framework shows as **Unidentified** and cost as **unpriced**. Both are expected. With a checkpointer, every turn's label repeats the conversation's first message, but the transcript inside each turn is correct. ## Troubleshooting -- **No spans at all.** `instrument()` never ran, or ran with a different provider than the one exporting. Import `tracing` first, pass `tracer_provider=provider`, and look for `OTLPSpanExporter` errors in the logs. -- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. -- **One session per message.** No `thread_id` reached the run, or it changes per request. Pass `{"configurable": {"thread_id": conversation_id}}` to every `invoke()`, `stream()` and resume, or `{"metadata": {"thread_id": ...}}` for plain chains. -- **One session per message and a raw JSON transcript, even with a `thread_id`.** The GenAI dual-write is off, so Maple only sees OpenInference's `session.id`, which it doesn't read for these spans. Pass `TraceConfig(enable_genai_semconv=True)`. -- **Streamed replies have no tokens.** `ChatOpenAI` with a custom `base_url` doesn't request streamed usage. Set `stream_usage=True`. -- **One failing tool ends the whole run.** `create_agent` re-raises tool exceptions. Add the `wrap_tool_call` middleware, or `handle_tool_errors=True` on your `ToolNode`. -- **A tool shows up as an agent, not a tool call.** Its name contains "agent". Rename it. -- **More tool calls than the agent made.** A graph node whose name contains "tool" is counted as a tool call. Add its name to `STEP_NAMES`. -- **Every model call appears twice.** A provider instrumentor or `LANGSMITH_OTEL_ENABLED` is also active. Keep one. -- **No lanes for sub-agents.** Their names aren't in `AGENT_NAMES`, or two agents share a name. -- **Turns split into several traces on Python 3.10.** Pass `config` to nested async calls, or upgrade to Python 3.11. +- **No spans at all.** Import `tracing` before the first `invoke()`, pass `tracer_provider=provider`, and check the logs for `OTLPSpanExporter` errors. +- **One session per message.** The `thread_id` is missing or changes per request (plain chains need it in `metadata`). If it's set, check for `enable_genai_semconv=True`. +- **Streamed replies have no tokens.** Set `stream_usage=True` on `ChatOpenAI`. +- **Every model call appears twice.** Remove `LANGSMITH_OTEL_ENABLED` and any provider instrumentor such as `openinference-instrumentation-openai`. Plain `LANGSMITH_TRACING=true` is fine. +- **Extra tool calls, or a tool shown as an agent.** Add tool nodes to `STEP_NAMES`, and don't put "agent" in tool names. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md index f64127a273..3a13a49baa 100644 --- a/apps/landing/src/content/docs/agent-tracing/litellm.md +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -1,17 +1,15 @@ --- title: "Trace LiteLLM agents and the LiteLLM Proxy with OpenTelemetry" -description: "Send LiteLLM's OpenTelemetry spans to Maple from the Python SDK or the self-hosted LiteLLM Proxy, add the agent and tool spans LiteLLM can't emit, and group each conversation into one Agent Session." +description: "Send LiteLLM's model-call spans to Maple from the Python SDK or the LiteLLM Proxy, and add the agent and tool spans that group each conversation into one Agent Session." group: "AI Agents" order: 40 navLabel: "LiteLLM" icon: "litellm" --- -LiteLLM traces model calls, and only model calls. Its OpenTelemetry logger writes one span per `acompletion()`, with the model, the provider, token counts and the prompt and reply. LiteLLM has no agent loop, no tool executor and no notion of a conversation, so the loop around those calls is your code, and the agent and tool spans have to come from your code too. This guide shows the Python that adds them. +LiteLLM's OpenTelemetry logger writes one `chat ` span per model call, with the model, tokens, prompt and reply. It has no agent loop or notion of a conversation, so your code adds the agent and tool spans, and you pass a session id on every call so Maple can group the turns. -Two defaults go wrong for Maple. The default (v1) logger has no conversation id at all, so every model call lands in its own one-call session, and when your code already has a span open, v1 writes its attributes onto that span after it has ended and they are dropped. LiteLLM's newer v2 logger fixes both and turns `litellm_session_id` into `gen_ai.conversation.id`, but it is off by default, only traces the async API, and in LiteLLM 1.103 it can't load on OpenTelemetry 1.44 or newer. - -This guide covers the LiteLLM Python SDK and the self-hosted LiteLLM Proxy. It was tested with `litellm` 1.103.0 and the OpenTelemetry Python SDK 1.43.0 on Python 3.12. +Tested with `litellm` 1.103.0 and the OpenTelemetry Python SDK 1.43.0 on Python 3.12. ## Quick setup with a coding agent @@ -25,28 +23,21 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. - -## Trace in your app or at the proxy, not both - -There are two places LiteLLM can trace a model call: +Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. -- **In your app**, when your code calls `litellm.acompletion()` directly. The next sections cover this. -- **At the LiteLLM Proxy**, when your app sends OpenAI-compatible requests to a proxy you run. See [Trace at the LiteLLM Proxy](#trace-at-the-litellm-proxy). +## Trace in your app or at the proxy -In both cases your app still emits the agent and tool spans, because only your app knows about them. Pick one place for the model-call spans. If the proxy traces the calls and your app also instruments its OpenAI client, every call shows up twice in the same trace. +If your code calls `litellm.acompletion()`, trace in your app with the next sections. If your app sends OpenAI-compatible requests to a LiteLLM Proxy you run, see [Trace at the LiteLLM Proxy](#trace-at-the-litellm-proxy). Either way your app emits the agent and tool spans. Trace model calls in one place only, or every call shows up twice. -## Export LiteLLM SDK spans to Maple - -Install LiteLLM with the OpenTelemetry SDK and the OTLP/HTTP exporter: +## Install LiteLLM and the exporter ```bash pip install "litellm==1.103.0" "opentelemetry-sdk==1.43.0" "opentelemetry-exporter-otlp-proto-http==1.43.0" ``` -Keep OpenTelemetry below 1.44 on LiteLLM 1.103, the latest stable release as of September 2026. OpenTelemetry 1.44 removed the Events API that LiteLLM's v2 logger imports ([BerriAI/litellm#41990](https://github.com/BerriAI/litellm/issues/41990)). Importing `OpenTelemetryV2` as below then fails with `ModuleNotFoundError: No module named 'opentelemetry._events'`. The `callbacks: ["otel"]` form the proxy uses is worse: LiteLLM catches the error, logs `Error initializing custom logger` and keeps serving requests without exporting anything. The fix is merged, and the 1.104.0rc1 release candidate traces correctly with OpenTelemetry 1.45. Drop the pin once 1.104 is out. +Keep OpenTelemetry below 1.44 on LiteLLM 1.103. Newer versions break LiteLLM's v2 logger ([BerriAI/litellm#41990](https://github.com/BerriAI/litellm/issues/41990)). The fix ships in LiteLLM 1.104, after which you can drop the pin. -Point the exporter at Maple with the standard OpenTelemetry variables: +Point the exporter at Maple: ```bash export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" @@ -54,9 +45,11 @@ export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" ``` -For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. +For an EU organization, use `https://ingest.eu.maple.dev`. + +## Register LiteLLM's v2 logger -Then set up tracing once, when your process starts. Create your own `TracerProvider` and hand it to LiteLLM's v2 logger, so LiteLLM's spans and yours go through one exporter: +Use the v2 logger (`OpenTelemetryV2`). The default v1 logger never records a conversation id, so every call becomes its own session. Hand v2 your own `TracerProvider` so LiteLLM's spans and yours share one exporter: ```py # tracing.py @@ -79,7 +72,6 @@ provider = TracerProvider( provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) -# LiteLLM's v2 OpenTelemetry logger, writing into the same provider as your own spans. litellm.callbacks = [ OpenTelemetryV2( config=OpenTelemetryV2Config(capture_message_content="span_only"), @@ -90,13 +82,13 @@ litellm.callbacks = [ tracer = trace.get_tracer("support-agent") ``` -Import `tracing` at the top of your entry point, before the first model call. Passing the logger instance means you don't need `LITELLM_OTEL_V2=true`, and LiteLLM builds no provider or exporter of its own. If your app already has a `TracerProvider`, pass that one as `tracer_provider=` instead of creating a second. +Import `tracing` at the top of your entry point, before the first model call. If your app already has a `TracerProvider`, pass that one as `tracer_provider=`. Don't also add `"otel"` to `litellm.callbacks`, which registers a second logger. -Don't also put `"otel"` in `litellm.callbacks` or `success_callback`. That adds a second logger, and every model call is exported twice. +`capture_message_content="span_only"` records prompts and replies on the span, which Maple needs for the transcript. Use `"no_content"` to keep them out of Maple. -### Trace the agent loop around LiteLLM +## Wrap the agent loop in agent and tool spans -Each `acompletion()` becomes a `chat ` span. To get turns, sub-agents and tool calls in Maple, wrap each agent run in an `invoke_agent` span and each tool call in an `execute_tool` span. LiteLLM's spans nest under whatever span is current, so they land inside the agent span in the same trace: +Wrap each agent run in an `invoke_agent` span and each tool call in an `execute_tool` span. LiteLLM's `chat` spans nest under the current span, so a run is one trace. The v2 logger only traces the async API, so call `acompletion()`, not `completion()`: ```py # agent.py @@ -163,17 +155,16 @@ async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: messages.append(message.model_dump(exclude_none=True)) if not message.tool_calls: return message.content or "" - # Parallel tool calls run concurrently; each keeps the agent span as its parent. results = await asyncio.gather(*(run_tool(agent, call) for call in message.tool_calls)) for call, output in zip(message.tool_calls, results): messages.append({"role": "tool", "tool_call_id": call.id, "content": output}) ``` -The v2 logger only traces the async API. `litellm.completion()` and the other sync calls produce no span at all, because the logger closes spans in its async success callback. Use `acompletion()`. If your code is sync throughout, see the troubleshooting entry on sync code. +For sub-agents, give each agent its own `name` and call `run_agent` for the worker from inside a tool function, with the same conversation id. Maple draws one lane per agent name. -## Group every turn of a conversation into one session +## Group every turn into one session -Maple groups LiteLLM traces into sessions by `gen_ai.conversation.id`. The v2 logger writes it on every `chat` span from `litellm_session_id=`, or from `metadata={"session_id": ...}` if you already pass metadata. Pass it on every call, from the chat or thread id your app already has: +The v2 logger turns `litellm_session_id=` into `gen_ai.conversation.id`, which Maple uses as the session key. Pass the chat or thread id your app already has, stable for the whole conversation: ```py from agent import Agent, run_agent @@ -188,153 +179,15 @@ async def handle_message(chat_id: str, text: str) -> str: return await run_agent(assistant, chat_id, messages) ``` -The id must be stable for the whole conversation and different between conversations. A per-process constant merges every user into one session. - -Without it, each turn is its own session named `trace:`, and with the v1 logger there is no way to set it on LiteLLM's spans at all. - -Leave `gen_ai.conversation.id` off your own `invoke_agent` span. Maple labels a session with the framework of its earliest span that carries the session id, so putting it on your span labels the session **Unidentified** instead of **LiteLLM**. The id on LiteLLM's `chat` spans is enough, because Maple groups whole traces: one span with the id pulls in every span of its trace. - -## Record prompts, responses and tool calls - -The v2 logger records no content by default. `capture_message_content="span_only"` in `tracing.py` turns it on, and each `chat` span then carries `gen_ai.input.messages` and `gen_ai.output.messages` as JSON in the OpenAI chat format: the system prompt, the history, tool calls and tool results. Maple builds the transcript from those attributes. The environment variable `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only` does the same when LiteLLM builds the config itself, as the proxy does. - -One gap in the transcript today: Maple doesn't read the OpenAI `tool_calls` field inside these messages. A model call whose reply is only a tool request shows no reply text, and the tool call itself appears once, from your `execute_tool` span, with its arguments and result. Without the `execute_tool` spans below, tool calls would be missing from the transcript entirely. - -Don't use `event_only`. It moves content to OpenTelemetry log events, which Maple doesn't read, and the transcript is empty. - -To keep prompts out of Maple, set `capture_message_content="no_content"` and remove the `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result` lines from `run_tool`. Sessions keep their turns, models, tool names, tokens and errors, and the transcript is empty. - -`litellm.turn_off_message_logging = True` is the in-between option. It applies to every LiteLLM logging callback, and the messages keep their roles and tool calls, but each text becomes `redacted-by-litellm`. For pattern-based redaction of message content, run an OpenTelemetry Collector between your app and Maple. - -## Tools, errors and sub-agents - -Each tool call is an `execute_tool ` span with the tool name, the model's call id, the arguments and the result. Maple matches it to the model reply that requested it by `gen_ai.tool.call.id`. - -`run_tool` catches the exception, gives the model an `{"error": ...}` result so it can recover, and marks the span failed with `error.type` and an ERROR status. Maple counts the call as failed. If your loop instead returns error strings from tools without touching the span, Maple counts them as successes. - -A failed model call needs nothing from you. LiteLLM marks its `chat` span ERROR and sets `error.type` to the exception class, for example `RateLimitError`. - -For several agents, give each its own `name`. It becomes `gen_ai.agent.name` on its `invoke_agent` span, and Maple draws one lane per agent name. The simplest orchestrator is an agent whose tools are the other agents: each tool function calls `run_agent` with the same conversation id, so the whole run is one trace in one session: - -```py -from agent import Agent, run_agent - -# weather_worker, budget_worker, transport_worker and summary are Agent(...) instances -WORKERS = (weather_worker, budget_worker, transport_worker, summary) - - -def delegate(worker: Agent, conversation_id: str): - async def call(task: str) -> str: - return await run_agent(worker, conversation_id, [{"role": "user", "content": task}]) - - return call - - -def task_tool(name: str) -> dict: - return { - "type": "function", - "function": { - "name": name, - "description": f"Delegate a task to the {name} agent.", - "parameters": {"type": "object", "properties": {"task": {"type": "string"}}, "required": ["task"]}, - }, - } - - -async def briefing(conversation_id: str, request: str) -> str: - orchestrator = Agent( - "orchestrator", - "openrouter/openai/gpt-4o-mini", - "Call weather_worker, budget_worker and transport_worker in parallel, " - "then call summary once with their results, then reply with the summary.", - {w.name: delegate(w, conversation_id) for w in WORKERS}, - [task_tool(w.name) for w in WORKERS], - ) - return await run_agent(orchestrator, conversation_id, [{"role": "user", "content": request}]) -``` - -Each delegation is an `execute_tool weather_worker` span whose only child is `invoke_agent weather_worker`, which Maple shows as a sub-agent lane with the tool's arguments and result as the lane's input and output. When the model requests several workers in one reply, `asyncio.gather` in `run_agent` runs them in parallel, and they overlap in time as siblings under `invoke_agent orchestrator`. - -If your code calls the workers directly instead of through the model, wrap those calls in `with agent_span("orchestrator"):` so they still share one trace. `asyncio.gather` keeps the OpenTelemetry context, so workers started with it nest correctly. Thread pools don't: wrap work sent to `run_in_executor` or a `ThreadPoolExecutor` with `contextvars.copy_context().run`. +If you already pass `metadata`, `metadata={"session_id": ...}` works too. Don't set `gen_ai.conversation.id` on your own `invoke_agent` span, or the session is labeled **Unidentified** instead of **LiteLLM**. -## Tokens and cost +For streaming, pass `stream_options={"include_usage": True}` and consume the stream inside the agent span, or the streamed call has no token counts. -Each `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. Your own spans carry no usage, so nothing is counted twice. - -For streaming, pass `stream_options={"include_usage": True}` and consume the stream inside the agent span. LiteLLM closes the `chat` span when the stream ends, with the token counts and `gen_ai.response.time_to_first_chunk`: - -```py -async def stream_agent(agent: Agent, conversation_id: str, messages: list): - with agent_span(agent.name): - stream = await litellm.acompletion( - model=agent.model, - messages=[{"role": "system", "content": agent.instructions}, *messages], - stream=True, - stream_options={"include_usage": True}, - litellm_session_id=conversation_id, - ) - text = "" - async for chunk in stream: - delta = chunk.choices[0].delta.content if chunk.choices else None - if delta: - text += delta - yield delta - messages.append({"role": "assistant", "content": text}) -``` - -Cost shows as unpriced by default. LiteLLM prices every call, but the v2 logger writes the price to `litellm.cost.total` (and v1 buries it in the `hidden_params` JSON), and Maple reads cost only from `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or `llm.cost.total`, and never prices tokens itself. - -To get cost into Maple, add up LiteLLM's price for every call of a turn, sub-agents included, and put the total on the outermost `invoke_agent` span. Only the outermost one: Maple subtracts a sub-agent's reported cost from the agent above it, on the assumption that the parent's figure already includes it, so a per-agent total on every level undercounts the orchestrator. Replace `agent_span` in `agent.py`: - -```py -from contextvars import ContextVar - -# LiteLLM's price for every model call of the current turn, sub-agents included. -turn_costs: ContextVar[list | None] = ContextVar("turn_costs", default=None) - - -@contextmanager -def agent_span(name: str): - with tracer.start_as_current_span(f"invoke_agent {name}") as span: - span.set_attribute("gen_ai.operation.name", "invoke_agent") - span.set_attribute("gen_ai.agent.name", name) - if turn_costs.get() is not None: # a sub-agent: the outermost agent reports the cost - yield span - return - costs: list[float] = [] - token = turn_costs.set(costs) - try: - yield span - finally: - turn_costs.reset(token) - span.set_attribute("gen_ai.usage.cost", sum(costs)) -``` - -Then record each call's price. In `run_agent`, right after `acompletion` returns: - -```py - turn_costs.get().append(response._hidden_params.get("response_cost") or 0.0) -``` - -A stream carries its price on the last chunk's `usage.cost`. In `stream_agent`, keep it while you consume the stream and record it once at the end: - -```py - text, cost = "", 0.0 - async for chunk in stream: - if getattr(chunk, "usage", None) is not None: # the last chunk carries usage and cost - cost = getattr(chunk.usage, "cost", None) or cost - delta = chunk.choices[0].delta.content if chunk.choices else None - if delta: - text += delta - yield delta - turn_costs.get().append(cost) -``` - -The session and turn totals are then exact. The cost sits on the agent rather than on each model call, so the per-model breakdown stays unpriced. +Cost shows as unpriced. LiteLLM writes its price to an attribute Maple doesn't read. The [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-litellm) has a recipe that reports each turn's total. ## Trace at the LiteLLM Proxy -When your apps call a LiteLLM Proxy you run, trace the model calls there. Every app behind the proxy gets model-call spans without code changes, and content and retention settings live in one place. Enable the v2 logger in the proxy's `config.yaml`: +To trace at a proxy you run, enable the logger in its `config.yaml`: ```yaml model_list: @@ -347,7 +200,7 @@ litellm_settings: callbacks: ["otel"] ``` -and in the proxy's environment: +Set these in the proxy's environment: ```bash LITELLM_OTEL_V2=true @@ -359,7 +212,7 @@ OTEL_ENVIRONMENT_NAME=production OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only ``` -The official `ghcr.io/berriai/litellm` image ships OpenTelemetry 1.28 and the FastAPI instrumentation, so the 1.44 problem above doesn't apply to it. A pip-installed proxy needs the same pin plus the FastAPI instrumentation, which is what continues your app's trace: +The `ghcr.io/berriai/litellm` image works as is. A pip-installed proxy needs the OpenTelemetry pin plus the FastAPI instrumentation, which continues your app's trace: ```bash pip install "litellm[proxy]==1.103.0" "opentelemetry-sdk==1.43.0" \ @@ -367,7 +220,7 @@ pip install "litellm[proxy]==1.103.0" "opentelemetry-sdk==1.43.0" \ litellm --config config.yaml ``` -In your app, keep `agent_span` and `run_tool` from above, drop the LiteLLM logger from `tracing.py`, and send two headers with every request: `traceparent`, so the proxy's spans join the app's trace under the agent span, and `x-litellm-session-id`, which the proxy turns into `gen_ai.conversation.id`: +In your app, keep `agent_span` and `run_tool`, drop the LiteLLM logger from `tracing.py`, and send `traceparent` and `x-litellm-session-id` with every request: ```py import os @@ -386,17 +239,11 @@ async def call_model(conversation_id: str, messages: list, tools: list | None): ) ``` -`metadata: {"session_id": ...}` in the request body works as well. A W3C `baggage` header with `session.id=...` is picked up only when neither is present. +Don't also instrument the OpenAI client in the app, or calls and tokens double. On this path Maple currently shows the LLM call count at 2x the real number in the sessions list and 3x on the session page, because it counts the proxy's auth and server spans. Tokens, cost and the transcript are correct. -Don't add an OpenAI client instrumentor (OpenInference, OpenLLMetry, `opentelemetry-instrumentation-openai-v2`) to the app on top of this. The app's span and the proxy's span describe the same call, and Maple can't always tell them apart, so LLM call counts and tokens double. If you can't configure the proxy, do the opposite: leave its OpenTelemetry off and instrument the client in the app. +## Flush before a short-lived process exits -Each request then shows up in your app's trace as `invoke_agent` → `POST /chat/completions` (the proxy's server span) → `chat gpt-4o-mini`, next to an `auth /chat/completions` span for the proxy's key check. Depending on its setup, the proxy adds more housekeeping spans (database, Redis, guardrails). - -Maple currently counts each `auth /chat/completions` span as an extra LLM call, because it comes from LiteLLM and its name contains "chat". On the proxy path the LLM call count in the sessions list is double the real number. The session page also counts the proxy's `POST /chat/completions` server span, which carries `gen_ai.request.model`, so it shows three times the real number. Tokens, cost, the transcript and the session grouping are not affected. - -## Short-lived processes - -LiteLLM creates its `chat` span after the call returns, from a background logging queue. A script that exits right after its last call loses that call's span: the queue is drained at interpreter exit, after the `TracerProvider` has already shut down. Drain the queue, then flush, before the event loop ends: +LiteLLM creates its span after the call returns, from a background queue. A script, Lambda or notebook cell that ends right after its last call loses that span unless you drain the queue and flush: ```py import asyncio @@ -423,36 +270,19 @@ asyncio.run(main()) provider.shutdown() ``` -Call `flush_tracing()` at the end of every Lambda or Cloud Run job invocation, after each notebook cell that calls a model, and in your web framework's shutdown hook. A long-running server needs it only at shutdown. +A long-running server needs this only in its shutdown hook. ## Check that it works -Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: - -- one session per conversation, with the id you passed as `litellm_session_id`, and framework **LiteLLM**; -- one turn per top-level `invoke_agent` span, labeled with the user's message, and a transcript with the prompts, replies and tool calls; -- `invoke_agent ` spans from your code, `chat ` spans from LiteLLM inside them, and `execute_tool ` spans for tool calls; -- a lane per agent name for multi-agent runs, all in the caller's session; -- input and output tokens on every `chat` span, including streamed ones; -- failed tool calls marked as failed, with the error as the result; -- cost shown as unpriced, or the turn totals if you added `gen_ai.usage.cost`. +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session with the id you passed and framework **LiteLLM**, one turn per message, and a transcript with the prompts, replies and tool calls. Each turn holds your `invoke_agent` span with `chat ` and `execute_tool ` spans inside it. ## Troubleshooting -- **`ModuleNotFoundError: No module named 'opentelemetry._events'` at startup, or no LiteLLM spans and a logged `Error initializing custom logger`.** LiteLLM 1.103 with OpenTelemetry 1.44 or newer. Pin `opentelemetry-sdk` and the exporter to 1.43.0, or upgrade to LiteLLM 1.104 once it's released. -- **Your spans arrive but no `chat` spans.** The code calls the sync `litellm.completion()`, which the v2 logger doesn't trace. Switch to `acompletion()`. -- **Spans named `litellm_request` and `raw_gen_ai_request`, and each call is its own session.** That is the v1 logger: `litellm.callbacks = ["otel"]` without the v2 instance. Use the `OpenTelemetryV2` setup above. -- **Model, tokens and prompts missing, and the SDK logs `Setting attribute on ended span`.** Also v1: with a span already open it writes onto that span instead of creating its own. If you must stay on v1 for sync code, set `USE_OTEL_LITELLM_REQUEST_SPAN=true` and `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental`, and put `gen_ai.conversation.id` on your `invoke_agent` span, since v1 can't carry it. The framework then shows as **Unidentified**. -- **Every call is its own session.** No `litellm_session_id=` (SDK) or `x-litellm-session-id` header (proxy) was sent. Pass it on every call. -- **The framework shows as Unidentified.** Your own span carries `gen_ai.conversation.id` or `maple_ai.session.id`. Remove it and let LiteLLM's spans carry the id. -- **Spans arrive with no prompts or replies.** Content capture is off, which is the v2 default. Set `capture_message_content="span_only"` or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only`. -- **The last call of a script or Lambda is missing.** The process ended before LiteLLM's logging queue ran. Await `flush_tracing()` before the loop ends. -- **Every model call appears twice.** Two loggers (the v2 instance plus `"otel"` in `litellm.callbacks`), or the proxy and an in-app OpenAI instrumentor both tracing the same call. Keep one. -- **Proxy spans land in their own traces, apart from the app's agent span.** The request carried no `traceparent`. Inject it with `propagate.inject(headers)` inside the agent span, and don't set `OTEL_IGNORE_CONTEXT_PROPAGATION` on the proxy. -- **The LLM call count is twice what the app made (three times on the session page), on the proxy path only.** Maple counts the proxy's `auth /chat/completions` spans as calls, and the session page also counts its `POST /chat/completions` server span. Token and cost totals are correct. -- **A model call shows an empty reply in the transcript.** That call only requested tools. The tool calls appear as their own rows from your `execute_tool` spans. -- **Session cost is lower than LiteLLM's spend.** Sub-agents report their own `gen_ai.usage.cost` and Maple subtracts it from the parent agent's. Report cost only on the outermost agent span, as in the cost section. -- **Nothing arrives from the proxy.** It is still on the default `console` exporter because no endpoint reached it. Check that `OTEL_EXPORTER_OTLP_ENDPOINT` is set in the proxy's environment, not only in your app's. +- **`ModuleNotFoundError: No module named 'opentelemetry._events'`, or no LiteLLM spans and `Error initializing custom logger` in the log.** OpenTelemetry 1.44+ on LiteLLM 1.103. Pin OpenTelemetry to 1.43.0. +- **Your spans arrive but no `chat` spans.** The code calls sync `litellm.completion()`. Switch to `acompletion()`. +- **Every call is its own session, or spans are named `litellm_request`.** You're on the v1 logger, or no session id was sent. Use the `OpenTelemetryV2` setup and pass `litellm_session_id=` (SDK) or `x-litellm-session-id` (proxy) on every call. +- **Every model call appears twice.** Two loggers, or the proxy and an in-app OpenAI instrumentor both trace the call. Keep one. +- **The last call of a script is missing.** Await `flush_tracing()` before the event loop ends. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/llamaindex.md b/apps/landing/src/content/docs/agent-tracing/llamaindex.md index b205c3e984..91f341ed6f 100644 --- a/apps/landing/src/content/docs/agent-tracing/llamaindex.md +++ b/apps/landing/src/content/docs/agent-tracing/llamaindex.md @@ -1,17 +1,15 @@ --- title: "Trace LlamaIndex agents with OpenTelemetry" -description: "Send LlamaIndex FunctionAgent, AgentWorkflow and Workflow runs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools and sub-agent lanes." +description: "Send LlamaIndex agent and workflow runs to Maple with OpenInference, one Agent Session per conversation." group: "AI Agents" order: 23 navLabel: "LlamaIndex" icon: "llamaindex" --- -LlamaIndex reports what it does through its own instrumentation dispatcher: every agent run, workflow step, model call and tool call opens a dispatcher span and fires events. Two packages turn those into OpenTelemetry spans. LlamaIndex's own `llama-index-observability-otel` copies the span tree but puts the payload in span events, and it loses the model's reply and token usage on every streamed call. OpenInference's `openinference-instrumentation-llama-index` writes model, tokens, messages and tool results as span attributes, which is what Maple reads. Use the OpenInference one. +OpenInference's `openinference-instrumentation-llama-index` turns LlamaIndex agent runs, workflow steps, model calls and tool calls into OpenTelemetry spans. It needs a small span processor to avoid counting each model call two or three times, and you have to wrap every `agent.run()` in a conversation id, or each message becomes its own session. -It needs four adjustments before a chat looks right in Maple. The instrumentor writes OpenInference attribute names unless you turn on its GenAI output, it never sets a conversation id, it doesn't record agent names, and it opens extra spans around every model and tool call, so Maple would count each model call two or three times and each tool call three times. - -This guide covers all four for llama-index-core 0.14.25 with `openinference-instrumentation-llama-index` 4.5.2 on Python 3.10 or later, using `FunctionAgent`, `AgentWorkflow` and custom `Workflow` classes. +Tested with llama-index-core 0.14.25 and `openinference-instrumentation-llama-index` 4.5.2 on Python 3.10 or later, using `FunctionAgent`, `AgentWorkflow` and custom `Workflow` classes. ## Quick setup with a coding agent @@ -25,28 +23,20 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. - -## Why not llama-index-observability-otel - -LlamaIndex's docs point to `LlamaIndexOpenTelemetry` from `llama-index-observability-otel` (0.7.0). It's a bridge: it mirrors dispatcher spans into OpenTelemetry spans and dispatcher events into span events. In Maple that gives you structure and nothing else: - -- **No model, tokens or transcript.** Model settings and the prompt sit inside an `LLMChatStartEvent` span event, and Maple never reads span events. The matching end event, with the reply and the usage, is dropped because the streamed `astream_chat` span closes before the stream is consumed. -- **Every call appears twice.** Each model call has two nested `OpenRouter.astream_chat` spans (or `OpenAI.astream_chat`, after your model class). -- **It ignores the standard exporter variables.** `OTEL_EXPORTER_OTLP_ENDPOINT` does nothing; the default exporter is `ConsoleSpanExporter`, so without an explicit `span_exporter=` everything goes to stdout. +Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. -If you already run it and only want sessions, `instrument_tags({"gen_ai.conversation.id": conversation_id})` around `agent.run()` groups the traces; dotted tag keys become span attributes verbatim. For transcripts and tokens, switch to OpenInference. Don't run both, or every span exists twice. - -## Install the instrumentor and export to Maple +## Install the instrumentor ```bash pip install "llama-index-core>=0.14.25" "openinference-instrumentation-llama-index>=4.5.2" \ "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Add your model package (`llama-index-llms-openai`, `llama-index-llms-anthropic`, `llama-index-llms-openrouter`, ...) as usual. The instrumentor requires llama-index-core 0.14.19 or later. On anything older it logs a dependency conflict and instruments nothing. +Add your model package (`llama-index-llms-openai`, `llama-index-llms-openrouter`, ...) as usual. On llama-index-core older than 0.14.19 the instrumentor logs a dependency conflict and instruments nothing. + +Use this instead of LlamaIndex's own `llama-index-observability-otel`, which puts the prompt and reply in span events that Maple doesn't read. Don't run both. -Point the exporter at Maple with the standard OpenTelemetry variables: +## Point the exporter at Maple ```bash export OTEL_SERVICE_NAME=support-agent @@ -55,9 +45,11 @@ export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" ``` -EU organizations use `https://ingest.eu.maple.dev`. Set the base URL: `OTLPSpanExporter()` with no arguments appends `/v1/traces`. If you pass `endpoint=` in code instead, it's used as is and has to end in `/v1/traces`. +EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, it has to end in `/v1/traces`. -Then add a `tracing.py` and import it at the top of your entry point, before the first `agent.run()`: +## Initialize tracing + +Add a `tracing.py` and import it at the top of your entry point, before the first `agent.run()`: ```py # tracing.py @@ -123,42 +115,15 @@ LlamaIndexInstrumentor().instrument( ) ``` -What each part does: - -- **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.tool.*` and `gen_ai.conversation.id` next to its OpenInference attributes when each span ends. Without it, Maple can count tokens but the session page has no transcript. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, but only if it's set before `TraceConfig` is built. -- **`LlamaIndexForMaple`** wraps the exporting processor and fixes what the instrumentor gets wrong for LlamaIndex. The next sections explain each fix. It also stamps `llamaindex.instrumentor` on every span, because Maple recognises LlamaIndex by a `llamaindex.*` attribute and the OpenInference spans carry none. Without it the sessions still work but show the framework as "Unidentified". -- **One `TracerProvider`.** If the app already has one (from `opentelemetry-instrument`, Logfire or Sentry), don't create a second. Add `LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))` to the existing provider and pass that provider to `instrument()`. - -### Why each model and tool call opens three spans - -LlamaIndex's dispatcher wraps every method that implements an abstract base method, and the instrumentor marks every span on an LLM object as kind `LLM`. A single `FunctionAgent` step with an OpenAI-compatible model (`OpenRouter`, `OpenAILike` and the others built on it) produces: - -```text -BaseWorkflowAgent.run_agent_step -├── OpenRouter._prepare_chat_with_tools kind LLM, builds the request, ~1 ms -└── OpenRouter.astream_chat kind LLM, OpenAILike's override - └── OpenRouter.astream_chat kind LLM, OpenAI's method: messages, usage -``` - -Maple counts every inference span without usage from a reporting ancestor as a model call, so this one call would count three times; in a test run, three model calls counted as nine. A class that implements the call itself, such as `OpenAI`, produces two spans, the helper and the call. Tokens are not doubled, since only the innermost span carries usage. - -`LlamaIndexForMaple` drops the `_prepare_chat_with_tools` helper, copies the inner span's attributes onto its same-named parent and drops the inner span, leaving one `OpenRouter.astream_chat` span per call with the messages and usage. It only merges spans with identical names that nest directly, so a chat engine's `CondensePlusContextChatEngine.chat` calling `OpenAI.chat` is left alone. - -Tool calls have the same problem one level up: +`enable_genai_semconv=True` is required. Without it the session page has no transcript. -```text -BaseWorkflowAgent.call_tool kind CHAIN, workflow step -└── FunctionTool.acall kind TOOL, the tool call -BaseWorkflowAgent.aggregate_tool_results kind CHAIN, workflow step -``` +`LlamaIndexForMaple` wraps the exporter, so always add the exporter through it. It merges the nested spans LlamaIndex opens around each model call into one, keeps workflow steps out of the tool count, copies agent names from `instrument_tags`, and labels the framework as LlamaIndex. -The two step spans have no GenAI operation, and Maple classifies such spans by name, so "tool" in the name makes each of them a tool call. `LlamaIndexForMaple` marks them `gen_ai.operation.name=invoke_workflow`, which Maple reads as agent work, leaving `FunctionTool.acall` as the only tool call. +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))` to that provider and pass it to `instrument()`. ## Group a conversation into one session -LlamaIndex keeps a conversation in a workflow `Context` (`Context(agent)`, or the chat memory you pass), and every `agent.run()` starts a new root span and a new trace. Nothing in those spans says which conversation they belong to, so without a conversation id every message becomes its own one-turn session in Maple, named after its trace id. - -Wrap each `agent.run()` call in OpenInference's `using_session` with your app's conversation id: +Wrap each `agent.run()` call in `using_session` with the conversation id your app stores the chat under, and tag it with the agent's name: ```py from llama_index.core.agent.workflow import AgentStream, FunctionAgent @@ -183,95 +148,15 @@ async def handle_message(conversation_id: str, text: str): await handler ``` -The instrumentor copies the id onto every span it creates as `session.id`, and the GenAI output repeats it as `gen_ai.conversation.id`, the key Maple reads for these spans. `using_session` sets a contextvar, and `agent.run()` starts the workflow's tasks immediately, so the id reaches every step, tool and model call of the run, including parallel workflow steps and the replay after a human-in-the-loop approval. Only the `agent.run()` call has to be inside the `with`; consuming the stream outside it is fine, which keeps the context manager out of your streaming generator. - -Use the id your app already stores the chat under. A new UUID per request gives you one session per message again, and a constant puts every user in one session. Keep one `Context` per conversation too: a shared `Context` shares the chat memory, and the traces would look like one long conversation. - -`llama_index.core.workflow.Context` is not a session id: it never reaches a span. Neither is the per-run `llamaindex.run_id` of the native package. - -## Record prompts, responses and tool calls - -Content capture is on by default. Every model span carries the full message list sent to the model as `gen_ai.input.messages` (system prompt, chat history, tool results) and the reply as `gen_ai.output.messages`, including tool call parts with their ids and arguments. Maple renders these as the session transcript, with the last user message as the turn label. - -The input list grows with the conversation: with a persistent `Context`, every model span repeats the whole chat so far. Maple has no per-attribute limit, and ingest accepts requests up to 20 MiB. - -To keep prompts and outputs out of your traces: - -```py -config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) -``` - -`hide_inputs` drops the input messages and replaces `input.value` with `__REDACTED__`; `hide_outputs` does the same for outputs and tool results. The session still shows turns, model and tool calls, tokens and failures, with an empty transcript. `hide_input_text` and `hide_output_text` keep the message structure and redact only the text. Each switch also has an `OPENINFERENCE_HIDE_*` environment variable. For pattern-based redaction (emails, card numbers), use the `redaction` processor in an OpenTelemetry Collector. - -## Tools, errors and sub-agents - -Each tool call is a `FunctionTool.acall` span of kind `TOOL`, with `gen_ai.operation.name` `execute_tool`, the tool's name in `gen_ai.tool.name`, its description, and its return value (LlamaIndex's `ToolOutput`, with `raw_input` and `raw_output`) in `gen_ai.tool.call.result`. It sits under a `BaseWorkflowAgent.call_tool` step span. - -A tool that raises is marked failed without extra code. `FunctionTool.acall` ends with status `ERROR` and the exception as the status message, for example `RuntimeError: transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. The `call_tool` step stays `OK`, because `FunctionAgent` catches the error and hands it to the model as the tool result. A tool that returns an error string instead of raising looks like a success. +Only the `agent.run()` call has to be inside the `with`. Every step, tool and model call of the run inherits the id, so you can consume the stream outside it. -Two gaps remain on tool spans. `gen_ai.tool.call.arguments` holds the tool's parameter schema, not the call's arguments, because the GenAI output copies OpenInference's `tool.parameters` there; the real arguments are in the model's tool call part in the transcript and in `raw_input` inside the result. And tool spans carry no `gen_ai.tool.call.id`, so Maple can't link a tool span to the exact call in the model's reply. +Keep one `Context` per conversation. The `Context` object itself isn't a session id, and a new UUID per request gives you one session per message. -Human-in-the-loop tools that call `ctx.wait_for_event(...)` run twice: the first call raises LlamaIndex's `WaitingForEvent` to suspend the step, and the step replays once the `HumanResponseEvent` arrives. The instrumentor records the first attempt as a failed tool call. `LlamaIndexForMaple` drops it, so an approved call shows as one successful tool call. +For multi-agent workflows, run the whole workflow inside `using_session(conversation_id)` and wrap each sub-agent's `run()` in its own `instrument_tags({"gen_ai.agent.name": agent.name})`. Maple then shows each agent in its own lane. `AgentWorkflow` handoffs happen inside one span, so they show as a single agent. -### Name your agents +## Get tokens on streamed calls -The instrumentor doesn't record which agent a span belongs to, and without `gen_ai.agent.name` Maple has no agent facet and no lanes. `LlamaIndexForMaple` copies a `gen_ai.agent.name` tag from LlamaIndex's `instrument_tags` onto every span started inside it. Put the tag around each agent's `run()`: - -```py -from llama_index.core.workflow import Context, Event, StartEvent, StopEvent, Workflow, step - - -class WeatherTask(Event): - city: str - - -class TransportTask(Event): - city: str - - -class WorkerDone(Event): - text: str - - -async def run_agent(agent: FunctionAgent, message: str) -> str: - with instrument_tags({"gen_ai.agent.name": agent.name}): - handler = agent.run(user_msg=message) - return str(await handler) - - -class Briefing(Workflow): - @step - async def plan(self, ctx: Context, ev: StartEvent) -> WeatherTask | TransportTask | None: - ctx.send_event(WeatherTask(city=ev.city)) - ctx.send_event(TransportTask(city=ev.city)) - - @step - async def weather(self, ev: WeatherTask) -> WorkerDone: - return WorkerDone(text=await run_agent(weather_worker, f"Weather in {ev.city}?")) - - @step - async def transport(self, ev: TransportTask) -> WorkerDone: - return WorkerDone(text=await run_agent(transport_worker, f"Transport in {ev.city}?")) - - @step - async def summarize(self, ctx: Context, ev: WorkerDone) -> StopEvent | None: - done = ctx.collect_events(ev, [WorkerDone, WorkerDone]) - if done is None: - return None - return StopEvent(result=await run_agent(summary_agent, "\n\n".join(d.text for d in done))) -``` - -Run the whole workflow inside `using_session(conversation_id)`. The workflow is one trace: `Briefing.run` at the root, one span per step, and each worker's `FunctionAgent.run` under its step with its own `gen_ai.agent.name`. Maple opens a lane for every agent span whose name differs from its caller's. `weather` and `transport` run concurrently because `plan` sends both events at once; parallel steps, including `@step(num_workers=N)`, stay in the same trace and keep the session and agent tags. - -The same works for agents called as tools: put `instrument_tags` inside the tool function around the sub-agent's `run()`. The tool span with one `FunctionAgent.run` child then shows as a delegation, with the tool's arguments and result as the lane's input and output. - -`AgentWorkflow` handoffs are different. The whole multi-agent run is one `AgentWorkflow.run` span, and LlamaIndex switches the active agent inside it without a span per agent, so there is nothing to tag. Every span carries the tag you put around `run()`, the handoff itself is a `handoff` tool call, and the run shows as one agent in Maple. If you need lanes, run each agent as its own `FunctionAgent.run()` from a workflow step or a tool, as above. - -## Tokens and cost - -The model span carries input and output tokens from the provider's reply as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus cached input tokens when the provider reports them, next to the OpenInference `llm.token_count.*` originals. The model is the one you configured (`gen_ai.request.model`, for example `openai/gpt-4o-mini`); the instrumentor doesn't record the model name the provider returns, or a response id. - -`FunctionAgent` streams every model call by default, even when you never read `AgentStream` events, and usage on a stream arrives only in the last chunk. OpenRouter always sends it. OpenAI's API sends it only when asked, so pass `stream_options` on OpenAI-compatible models: +`FunctionAgent` streams every model call. OpenAI's API only sends usage on a stream when asked, so pass `stream_options` on OpenAI and OpenAI-compatible models: ```py from llama_index.llms.openai import OpenAI @@ -279,59 +164,25 @@ from llama_index.llms.openai import OpenAI llm = OpenAI(model="gpt-4o-mini", additional_kwargs={"stream_options": {"include_usage": True}}) ``` -LlamaIndex strips the option from non-streaming requests, so it's safe to set once. - -The provider comes from the model class: `OpenRouter` and every `OpenAILike` model report `openai`, even for an Anthropic model behind OpenRouter. - -Maple shows cost only when a span carries one, and neither LlamaIndex nor the instrumentor records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call, and Maple joins them to the same session. These model spans have no `gen_ai.response.id`, so nest the Broadcast spans under them as described in [Join Broadcast to your own traces](/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces), or each call is counted twice. +OpenRouter sends usage without it. With `OpenAILike` or `OpenRouter`, also pass `is_function_calling_model=True`, or the agent never calls tools. -Don't add `openinference-instrumentation-openai` (or another provider instrumentor) next to the LlamaIndex one. It wraps the same HTTP call and gives every model call a second span with its own usage. +## Flush in short-lived processes -## Flush spans before the process exits - -`BatchSpanProcessor` exports every 5 seconds, and the `TracerProvider` flushes on a normal interpreter exit. That doesn't happen when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: - -```py -from tracing import provider - -try: - result = await workflow.run(city="Amsterdam") -finally: - provider.force_flush() # serverless: before returning; notebooks: after each run -``` - -Call `provider.shutdown()` instead when the process is about to exit and won't trace anything else. +The SDK flushes on a normal interpreter exit, and long-running servers need nothing. In serverless handlers, notebooks and task workers, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. ## Check that it works -Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: - -- **One session** for the conversation, with one turn per `agent.run()`. Each turn's trace starts at `FunctionAgent.run` (or `AgentWorkflow.run`, or your workflow class's `.run`). -- **Framework: LlamaIndex**, from the `llamaindex.instrumentor` attribute `LlamaIndexForMaple` adds. -- **The transcript**: your messages, the model's replies and the tool calls it made. -- **Model calls** named after your model class and method, for example `OpenRouter.astream_chat` or `OpenAI.achat`, one per call, each with a model and input and output tokens. -- **Tool calls** named `FunctionTool.acall`, one per call, with the tool's name and result. `BaseWorkflowAgent.call_tool` and `aggregate_tool_results` are agent steps, not tool calls. -- **Agents**: `assistant`, plus one lane per tagged sub-agent. -- **Cost**: unpriced. +Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session labeled **LlamaIndex**, with one turn per `agent.run()`, a transcript, one model call per request with tokens, and `FunctionTool.acall` tool calls. -A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. +Cost shows as **unpriced**, which is expected. Streamed model calls show about 1 ms of duration because the span ends when LlamaIndex hands back the stream. ## Troubleshooting -- **No spans at all.** `instrument()` never ran, or llama-index-core is older than 0.14.19 and the instrumentor skipped itself (look for `DependencyConflict` in the logs). Import `tracing` first in the entry point and look for an `OTLPSpanExporter` error in the logs. -- **Spans print to the console instead of reaching Maple.** You're running `LlamaIndexOpenTelemetry` without `span_exporter=`. Switch to the OpenInference setup above. -- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. -- **Tokens in the list, empty session page.** The GenAI output is off. Pass `TraceConfig(enable_genai_semconv=True)`. -- **One session per message.** `agent.run()` isn't inside `using_session(...)`, or the id changes per request. `Context` and `llamaindex.run_id` are not session ids. -- **Each model or tool call counted two or three times.** `LlamaIndexForMaple` isn't in front of the exporter. Add the exporter through it, not directly with `add_span_processor(BatchSpanProcessor(...))`. -- **Framework shows "Unidentified".** The spans lack a `llamaindex.*` attribute: `LlamaIndexForMaple` is missing, or it's an older copy without the `llamaindex.instrumentor` line. -- **Model calls take 1 ms.** Streamed model spans end when LlamaIndex hands back the stream, not when the last token arrives, so their duration isn't the model's latency, and the session's inference time adds up to a few milliseconds. The enclosing `BaseWorkflowAgent.run_agent_step` span has the real time. If you don't stream tokens to users, `FunctionAgent(..., streaming=False)` records model spans with their full duration. -- **No tokens on streamed calls.** The provider didn't send usage on the stream. Add `stream_options={"include_usage": True}` through `additional_kwargs`. -- **Tool arguments show the tool's schema.** A known gap in the instrumentor's GenAI output; see [Tools, errors and sub-agents](#tools-errors-and-sub-agents). -- **An approved tool call shows as failed first.** `wait_for_event()` suspends the tool by raising `WaitingForEvent`. `LlamaIndexForMaple` drops that attempt; check it's installed. -- **No lanes, no agent names.** Wrap each agent's `run()` in `instrument_tags({"gen_ai.agent.name": agent.name})`, with `LlamaIndexForMaple` installed. `AgentWorkflow` handoffs can't be split into lanes. -- **Tool calling silently doesn't happen.** `OpenAILike` and `OpenRouter` default to `is_function_calling_model=False`, so the agent falls back to prose and emits no tool spans. Pass `is_function_calling_model=True`. -- **Streaming query engine spans land in separate traces.** llama-index-core 0.14.25 defers the model call of a `StreamingResponse` until you consume it, after the query span has ended. OpenInference is fixing this in [PR #3841](https://github.com/Arize-ai/openinference/pull/3841); until it ships, pin `llama-index-core<0.14.25` if you trace streaming query engines. Agents are not affected. +- **No spans at all.** Import `tracing` before the first `agent.run()`, check for `DependencyConflict` in the logs (llama-index-core older than 0.14.19), and look for exporter errors. +- **One session per message.** `agent.run()` isn't inside `using_session(...)`, or the id changes per request. +- **Each model or tool call counted two or three times.** The exporter was added directly. Add it through `LlamaIndexForMaple`. +- **No tokens on streamed calls.** Add `stream_options={"include_usage": True}` through `additional_kwargs`. +- **No lanes or agent names.** Wrap each agent's `run()` in `instrument_tags({"gen_ai.agent.name": agent.name})`. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/mastra.md b/apps/landing/src/content/docs/agent-tracing/mastra.md index 7ab44d8103..776a87eed0 100644 --- a/apps/landing/src/content/docs/agent-tracing/mastra.md +++ b/apps/landing/src/content/docs/agent-tracing/mastra.md @@ -1,17 +1,17 @@ --- title: "Trace Mastra agents and workflows with OpenTelemetry" -description: "Export Mastra's built-in GenAI spans to Maple with @mastra/otel-exporter so each conversation is one Agent Session with its transcript, tool calls, sub-agents and tokens." +description: "Export Mastra's built-in spans to Maple with @mastra/otel-exporter and group each conversation into one Agent Session." group: "AI Agents" order: 11 navLabel: "Mastra" icon: "mastra" --- -Mastra traces itself. Every `agent.generate()` or `agent.stream()` produces an `invoke_agent` span, a `chat` span per model call with the prompt, the reply and the token counts, and an `execute_tool` span per tool call with its arguments and result. The `@mastra/otel-exporter` package turns those spans into OpenTelemetry GenAI spans (semantic conventions v1.38) and sends them to any OTLP endpoint, including Maple. You don't need an OpenTelemetry SDK or an instrumentation package. +Mastra traces every agent run, model call and tool call itself, and `@mastra/otel-exporter` sends those spans to Maple as OpenTelemetry GenAI spans. You don't need an OpenTelemetry SDK or an instrumentation package. -Three things go wrong by default. The exporter ignores the standard `OTEL_EXPORTER_OTLP_*` variables and falls back to OTLP/JSON, so a setup copied from another guide exports nothing and logs a single warning. The conversation id Maple groups by is Mastra's memory thread id: an agent called without `memory: { thread, resource }` sends no id at all, so every message becomes its own session. And Mastra 1.71 exports each model call without its prompt, so the transcript shows the replies but none of the user's messages. A short span processor, shown below, fixes the prompt and two smaller gaps. +The session id is Mastra's memory thread id, so every call of a conversation must pass the same `memory: { thread }`. You also add a short span processor that fills in the prompt Mastra leaves off each model call. -This guide covers Mastra 1.x on Node.js 22.13 or newer. It was written against `@mastra/core` 1.71, `@mastra/observability` 1.18 and `@mastra/otel-exporter` 1.4. +Tested with `@mastra/core` 1.71, `@mastra/observability` 1.18 and `@mastra/otel-exporter` 1.4 on Node.js 22.13 or newer. ## Quick setup with a coding agent @@ -25,25 +25,19 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. -## Export Mastra spans to Maple - -Install the observability packages next to `@mastra/core`: +## Install the observability packages ```bash npm install @mastra/observability@latest @mastra/otel-exporter@latest ``` -The OTLP exporters, including `@opentelemetry/exporter-trace-otlp-proto`, are optional dependencies of `@mastra/otel-exporter` and install with it. Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week. The exporter decides which span is the model call from features both packages report, so mismatched releases change what Maple counts as a model call. - -### Add the Maple span processor +Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week. With mismatched releases, the exporter can pick the wrong span as the model call. -Mastra 1.71 has three gaps in what it exports, and one span processor closes all of them: +## Add the Maple span processor -- The `chat` span of each model call has the reply but not the prompt. Mastra records the messages on the enclosing step span instead. The processor copies them onto the `chat` span, where the exporter turns them into `gen_ai.input.messages`. -- Sub-agents called by a supervisor get their own thread id, so one trace carries several conversation ids. The processor gives every span the thread id of the trace's root span. -- Step spans carry the provider's raw HTTP response as metadata: response headers, including cookies, and the full response body with the reply text. The processor drops both. +This processor fixes three gaps in what Mastra 1.71 exports. It copies each model call's prompt onto its `chat` span, gives sub-agents the conversation's thread id instead of their own, and drops the raw provider response (headers, cookies, full body) from step spans. ```ts // src/mastra/maple-span-processor.ts @@ -73,11 +67,9 @@ export const mapleSpanProcessor: SpanOutputProcessor = { } ``` -A processor has to change the span it receives and return that same object; Mastra drops the span if you return a copy. Mastra's own `SensitiveDataFilter` runs after your processors, so the copied prompt is still redacted. - -### Configure the exporter +## Configure the exporter -Then configure observability on your `Mastra` instance: +Add observability to your `Mastra` instance: ```ts // src/mastra/index.ts @@ -115,34 +107,15 @@ export const mastra = new Mastra({ }) ``` -For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself, and strips it if you include it, so both forms work. - -Three details in this block are load-bearing: - -- `observability` must be an `Observability` instance. A plain `{ configs: ... }` object logs a warning and installs a no-op, so nothing is exported. -- `protocol: "http/protobuf"` must be explicit. The `custom` provider defaults to `http/json`, and each protocol loads a different exporter package; if that package is missing, tracing is disabled with one error line at startup. -- The endpoint and key go in code. `OTEL_EXPORTER_OTLP_ENDPOINT` and `OTEL_EXPORTER_OTLP_HEADERS` are not read by this exporter. The `headers` object is sent as is, so there is no `%20` encoding to get wrong. - -Only agents and workflows registered on this `Mastra` instance, or called with a `tracingContext` from one, are traced. An `Agent` you construct and call on its own, outside `mastra.getAgent()`, has no observability attached. - -### The exporter also sends logs - -`OtelExporter` exports two signals by default: traces to `/v1/traces` and Mastra's own log records (warnings and errors, such as a tool that threw) to `/v1/logs`. Maple accepts both. The logs carry the trace and span ids of the run that wrote them. Agent Sessions reads only the spans. To send traces only: - -```ts -new OtelExporter({ - provider: { custom: { /* as above */ } }, - signals: { logs: false }, -}) -``` +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. -### An app that already has OpenTelemetry +The endpoint, protocol and key have to be in code. This exporter ignores the `OTEL_EXPORTER_OTLP_*` variables and defaults to `http/json`. `observability` must be an `Observability` instance, since a plain object silently installs a no-op. -The exporter runs its own `BatchSpanProcessor` and doesn't touch a global `TracerProvider`, so it coexists with an existing Node SDK setup. The Mastra spans then form their own traces, separate from your HTTP spans. If you want Mastra's spans nested under your request spans, use Mastra's [OpenTelemetry bridge](https://mastra.ai/reference/observability/tracing/bridges/otel) instead of `OtelExporter`, and point your existing exporter at Maple. Use one or the other, not both, or every span arrives twice. +Only agents and workflows registered on this `Mastra` instance are traced. Get them with `mastra.getAgent()` or `mastra.getWorkflow()`. -## Group every turn of a conversation into one session +## Pass the thread id on every call -Maple groups traces into sessions by `gen_ai.conversation.id`. Mastra writes it on every span from the memory thread id of the run, and only from there. Each `generate()` or `stream()` call is its own trace, so a chat that sends one request per user message needs the same thread id on every call: +Maple groups traces into sessions by `gen_ai.conversation.id`, which Mastra sets from the memory thread. Pass the same thread on every call of a conversation: ```ts const agent = mastra.getAgent("supportAgent") @@ -155,22 +128,9 @@ export async function handleMessage(chatId: string, userId: string, text: string } ``` -The thread id must be stable for the whole conversation and different between conversations. Use the chat or conversation id your app already has; a per-process constant merges every user into one session. - -This is the same `memory` option that makes the agent remember earlier messages, so an agent with `memory: new Memory(...)` that already passes it needs nothing more. If you skip it, each message shows up in **Agent Sessions** as its own one-turn session named after the trace id. Mastra's server routes (`mastra dev`, `@mastra/server`) pass the thread the client sends, or the one your middleware sets as `mastra__threadId` in the request context. - -Streaming works the same way. Consume the whole stream: the `invoke_agent` span ends, and gets exported, when the stream finishes. - -```ts -const stream = await agent.stream(text, { memory: { thread: chatId, resource: userId } }) -for await (const chunk of stream.textStream) { - res.write(chunk) -} -``` - -### Workflows and agents without memory +Use the chat id your app already has. It must stay the same for the whole conversation and differ between conversations. An agent that already uses `Memory` with this option needs nothing more. `agent.stream()` takes the same option; read the stream to the end, since the spans are exported when it finishes. -A workflow run has no thread, and neither does an agent you call without `memory`. Put the id in the root span's metadata instead. Mastra copies root metadata to every span in the trace, and the exporter turns `metadata.threadId` into `gen_ai.conversation.id`: +Workflow runs, and agents called without `memory`, have no thread. Put the id in the root span's metadata instead: ```ts const run = await mastra.getWorkflow("briefingWorkflow").createRun() @@ -180,177 +140,25 @@ const result = await run.start({ }) ``` -The same `tracingOptions` works on `agent.generate()` for an agent that has no memory configured. - -## Record prompts, responses and tool calls - -Content capture is on by default. With the span processor in place, each `chat` span carries `gen_ai.input.messages` and `gen_ai.output.messages` as JSON in the GenAI message format, and each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Maple builds the transcript from those attributes. The agent's instructions are the first, `system`, message of every prompt. - -Two details of Mastra's format show up in the transcript. Tool calls from earlier in the conversation appear in later prompts as short `[tool: get_weather]` placeholders; the full arguments and results are on the `execute_tool` spans. And the `invoke_agent` span's `gen_ai.system_instructions` is plain text rather than the JSON the convention asks for, so Maple skips it and shows the instructions from the system message instead. - -Mastra bounds what it serializes into a span. The defaults are 128 KiB per string, 50 items per array, 50 keys per object and 8 levels deep. A cut string ends in `…[truncated]`, which Maple shows as truncated. A message list longer than 50 entries loses the newest messages, so a turn with a long history can be missing its latest user message. Raise the array limit if your agents keep long histories: - -```ts -new Observability({ - configs: { - maple: { - serviceName: "support-agent", - exporters: [mapleExporter], - excludeSpanTypes: [SpanType.MODEL_CHUNK], - spanOutputProcessors: [mapleSpanProcessor], - serializationOptions: { maxArrayLength: 200 }, - }, - }, -}) -``` - -To keep content out of Maple for one request, hide it for the whole trace: - -```ts -await agent.generate(text, { - memory: { thread: chatId, resource: userId }, - tracingOptions: { hideInput: true, hideOutput: true }, -}) -``` - -Sessions keep their turns, models, tool names, tokens and errors, but the transcript is empty and tool calls have no arguments or results. These options hide span input and output only, not metadata. Without the span processor, the full provider response, reply included, still leaves in the step span's `mastra.metadata.body`. - -Mastra also applies a `SensitiveDataFilter` to every span by default. It redacts values under keys like `password`, `token`, `apiKey`, `authorization` and `secret`, including inside JSON strings, and replaces them with `[REDACTED]`. It matches key names, not free text, so a user who types a password into the chat still sends it to Maple. If you need pattern-based redaction of message text, run it in an OpenTelemetry Collector between your app and Maple. - -## Tools, errors and sub-agents - -Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the model's `gen_ai.tool.call.id`, the tool description, and the arguments and result. The transcript shows each call from its `execute_tool` span. The exporter leaves tool calls out of the `chat` span's output messages, so a model reply that only requests tools shows as an empty assistant message. - -A tool that throws is marked failed: status ERROR with the exception message as the status message, `error.type` set to `unknown`, and an `exception` event. The agent keeps running and the model sees the error. A tool that returns an error value instead of throwing stays green, and Maple counts the call as a success, so throw for failures you want to see: - -```ts -import { createTool } from "@mastra/core/tools" -import { z } from "zod" - -export const fetchTransportData = createTool({ - id: "fetch_transport_data", - description: "Fetch public transport options for a city.", - inputSchema: z.object({ city: z.string() }), - execute: async () => { - throw new Error("transport data service unavailable (503)") - }, -}) -``` - -Tools with `requireApproval: true` suspend the run before the tool executes. The resumed run (`approveToolCallGenerate()` or `declineToolCallGenerate()`) continues the same trace; pass the same `memory` to it. A declined call produces no `execute_tool` span. - -Approvals leave one artifact. The model call that asks for the tool is exported twice: once without tokens or reply when the run suspends, and again with its tokens when the resumed run replays it. Maple shows one extra model call with 0 tokens per approval. Token totals are right. - -### Sub-agents: keep one conversation id - -Mastra's multi-agent idiom is a supervisor: an agent with an `agents` property calls each sub-agent through a tool named `agent-`. Each delegation runs the sub-agent inside the supervisor's trace, under that tool span. Give every agent a `name`; it becomes `gen_ai.agent.name`, and Maple draws one lane per agent name. An `execute_tool agent-weather_worker` span whose only child is `invoke_agent weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. - -The `agent-` calls also count as tool calls in the session's totals, next to the sub-agents' own tools. A sub-agent's prompt in the transcript starts with the supervisor's instructions and the user's original message, then the delegated task: Mastra passes the supervisor's conversation along to each sub-agent. - -```ts -export const orchestrator = new Agent({ - id: "orchestrator", - name: "orchestrator", - instructions: "Call weather_worker, budget_worker and transport_worker, then summary.", - model: "openrouter/openai/gpt-4o-mini", - agents: { weather_worker: weatherWorker, budget_worker: budgetWorker, transport_worker: transportWorker, summary }, - memory, -}) -``` - -The catch is memory. When the supervisor runs with a thread, Mastra gives each delegation its own thread id, `-`. The sub-agent's spans carry that id as `gen_ai.conversation.id`, so one trace has several ids. Maple takes the largest one for the whole trace, which is always a sub-agent's, so the run lands in a session of its own and its turn splits into one turn per id. The `mapleSpanProcessor` above rewrites every span to the root span's thread id, so the whole run stays in your conversation's session. +Give every `Agent` a distinct `name`, since Maple draws one lane per agent name. When a workflow step calls an agent, pass it the step's `tracingContext` (`agent.generate(prompt, { tracingContext })`) so the agent joins the workflow's trace. -The processor also covers workflows whose steps call agents with their own `memory`. Pass `tracingContext` from the step's `execute` arguments to `agent.generate()`, so the agent's spans join the workflow's trace instead of starting a new one: - -```ts -const weatherStep = createStep({ - id: "weather_worker", - inputSchema: z.object({ city: z.string() }), - outputSchema: z.object({ findings: z.string() }), - execute: async ({ inputData, tracingContext }) => { - const res = await weatherWorker.generate(`Report the weather in ${inputData.city}.`, { tracingContext }) - return { findings: res.text } - }, -}) -``` +## Flush in scripts and serverless functions -Steps in `.parallel([...])` run concurrently, and their lanes overlap in time in Maple. A workflow run's root span is `invoke_workflow `; Maple treats it as the turn. - -## Tokens and cost - -The `chat` span of each model call carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. The input count includes the cached tokens; Maple shows the cached part separately and counts it once. Only that span carries usage, so the enclosing agent, step and generation spans add nothing to the total. - -Reasoning tokens are exported as `gen_ai.usage.reasoning_tokens`, a key Maple doesn't read. They're still inside the output token count, so totals are right; only the reasoning breakdown is missing. - -Streamed calls report usage too. Time to first token is exported only as `mastra.completion_start_time`, a timestamp Maple doesn't read, so sessions show no time to first token. - -Cost shows as unpriced. Mastra doesn't put a cost attribute on its spans, and Maple never prices tokens itself. Tokens and models are complete, and call counts are too, apart from the extra call per tool approval described above. - -Mastra doesn't export `gen_ai.response.id`. Maple uses it only to merge two reports of the same call, and Mastra reports each call's usage once, so nothing is counted twice. - -## Short-lived processes - -The exporter batches spans and sends them every 5 seconds. A script, CLI, test run or serverless function that exits sooner loses the batch. In a script, shut Mastra down before exiting; that flushes every exporter: - -```ts -try { - await main() -} finally { - await mastra.shutdown() -} -``` - -In a serverless handler, flush at the end of each request instead, and keep the instance for the next one: - -```ts -export async function POST(req: Request) { - const { chatId, userId, text } = await req.json() - try { - const result = await mastra.getAgent("supportAgent").generate(text, { - memory: { thread: chatId, resource: userId }, - }) - return Response.json({ text: result.text }) - } finally { - await mastra.observability.flush() - } -} -``` - -For a streamed response, flush after the stream has finished, for example in Next.js `after()` or Cloudflare's `ctx.waitUntil()`. Flushing when the handler returns the stream is too early: the spans end when the last chunk is sent. +The exporter sends spans every 5 seconds. In a script, call `await mastra.shutdown()` in a `finally` block before exiting. In a serverless handler, call `await mastra.observability.flush()` at the end of each request, after any streamed response has finished. ## Check that it works -Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive, longer if the process is still running and hasn't hit the 5-second export interval. You should see: - -- one session per conversation, with your thread id as the session id, and framework **Mastra**; -- one turn per `generate()` or `stream()` call, labeled with the user's message, and a transcript with the prompts, replies and tool calls; -- `invoke_agent ` spans for runs, `chat ` spans for model calls, and `execute_tool ` spans for tool calls, with `agent_step` and `model_generation` spans in between; -- `invoke_workflow ` as the turn for workflow runs; -- a lane per sub-agent, named after its `name`, with each `agent-` delegation counted as a tool call; -- token counts on every model call, including streamed ones, except one 0-token call per tool approval; -- failed tool calls marked as failed, with the thrown message; -- cost shown as unpriced. +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session named after your thread id with framework **Mastra**, one turn per `generate()` or `stream()` call, and a transcript with the prompts, replies and tool calls. -To see what the exporter does locally, set `logLevel: "debug"` on `OtelExporter`. It prints every queued span and a line per batch, `Export completed: N spans sent successfully` or `Export FAILED` with the reason. +Cost shows as unpriced, because Mastra doesn't report it. If nothing arrives, set `logLevel: "debug"` on `OtelExporter` to log each export as `Export completed` or `Export FAILED` with the reason. ## Troubleshooting -- **Nothing arrives and there is no error.** `observability` is a plain object instead of `new Observability(...)`, or the agent isn't registered on the `Mastra` instance. Check the startup log for a no-op observability warning. -- **`Traces http/json exporter is not installed` or `http/protobuf exporter is not installed` at startup.** The protocol's exporter package is missing. It ships as an optional dependency of `@mastra/otel-exporter`, so an install with `--omit=optional` or `--no-optional` skips it. Set `protocol: "http/protobuf"` and install `@opentelemetry/exporter-trace-otlp-proto`. -- **`Custom configuration requires endpoint. Tracing will be disabled.`** The exporter got no `endpoint`. It doesn't read `OTEL_EXPORTER_OTLP_ENDPOINT`; pass the endpoint in code. -- **`Export FAILED` with 401 or 403 in debug output.** The `Authorization` header is missing or the key is wrong. The header value is `Bearer ` followed by the ingest key. -- **Every message is its own session.** The call has no `memory: { thread, resource }`, or the thread id changes per request. Pass the conversation's id on every call; for workflows, use `tracingOptions.metadata.threadId`. -- **The transcript has the replies but none of the user's messages.** `mapleSpanProcessor` isn't in `spanOutputProcessors`, so the `chat` spans have no `gen_ai.input.messages`. -- **A supervisor run shows several turns, or lands in a session named `-`.** Delegations got their own thread ids. Add `mapleSpanProcessor`. -- **Nothing arrives from a script or serverless function.** The process ended before the batch was exported. Call `mastra.shutdown()` in a script, or `mastra.observability.flush()` at the end of each request. -- **One model call per turn with the turn's summed tokens, instead of one per request.** `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` are from different releases, and the exporter fell back to the older `model_generation` span as the model call. Update the three together. -- **One model call per tool approval has no tokens and no reply.** Expected: Mastra exports the call that requested the tool again when the run resumes, and that copy carries the tokens. -- **Dozens of tiny `model_chunk` spans per reply.** Mastra exports one per streamed chunk by default, for `generate()` too. Add `excludeSpanTypes: [SpanType.MODEL_CHUNK]`. -- **Hundreds of `workflow_step` spans per turn.** `includeInternalSpans: true` is set. Mastra runs its agent loop as internal workflows; the flag exports all of them, around 5 times the spans and 25 times the bytes per turn. Leave it off. -- **A workflow's agents show up as separate traces.** The step called `agent.generate()` without `tracingContext`. Pass it from the step's `execute` arguments. -- **The latest user message is missing from a long conversation's transcript.** The message list hit `maxArrayLength` (50). Raise it in `serializationOptions`. -- **A failed tool shows as successful.** The tool returned an error value instead of throwing. Throw an `Error`. -- **Spans show up twice.** Two exporters send the same spans to Maple, for example `OtelExporter` plus the OpenTelemetry bridge, or `OtelExporter` plus an OpenLLMetry or OpenInference instrumentation of the AI SDK underneath. Keep `OtelExporter` and remove the other. +- **Nothing arrives and there is no error.** `observability` is a plain object instead of `new Observability(...)`, or the agent isn't registered on the `Mastra` instance. +- **`http/protobuf exporter is not installed` at startup.** The install skipped optional dependencies. Install `@opentelemetry/exporter-trace-otlp-proto`. +- **Every message is its own session.** The call has no `memory: { thread, resource }`, or the thread id changes per request. For workflows, use `tracingOptions.metadata.threadId`. +- **The transcript has replies but no user messages, or a supervisor run lands in a session named `-`.** `mapleSpanProcessor` is missing from `spanOutputProcessors`. +- **A failed tool shows as successful.** The tool returned an error value. Throw an `Error` instead. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md index ad8fc62930..e0b9ade0df 100644 --- a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md +++ b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md @@ -1,21 +1,19 @@ --- title: "Trace Microsoft Agent Framework and Semantic Kernel agents with OpenTelemetry" -description: "Send Microsoft Agent Framework and Semantic Kernel traces to Maple so each conversation becomes one Agent Session with its transcript, tool calls, tokens and failures, in Python and .NET." +description: "Send Microsoft Agent Framework and Semantic Kernel traces from Python or .NET to Maple as one Agent Session per conversation." group: "AI Agents" order: 30 navLabel: "Microsoft Agent Framework" icon: "dotnet" --- -Microsoft Agent Framework (MAF) ships its own OpenTelemetry instrumentation in Python and .NET. Every `agent.run()` produces an `invoke_agent` span, a `chat` span per model call and an `execute_tool` span per tool call, using the current GenAI semantic conventions, with tokens (including cache and reasoning buckets) and, once you switch content capture on, the full prompts and replies as span attributes. Workflows add `workflow.run`, `executor.process` and `edge_group.process` spans. +Microsoft Agent Framework (MAF) emits OpenTelemetry spans for every agent run, model call and tool call in Python and .NET. It does not set a conversation id when your app keeps chat history itself, so you add one with a short span processor, or every turn shows up as its own session. -What it doesn't emit is a conversation id. MAF only sets `gen_ai.conversation.id` when the model provider stores the conversation server side, so with Chat Completions, OpenRouter or any local chat history, every turn arrives as its own one-turn session. This guide adds the id with a 15-line span processor. - -This guide covers `agent-framework` 1.19 (Python), `Microsoft.Agents.AI` 1.22 (.NET), and Semantic Kernel 1.44 (Python), the framework MAF replaces, which has its own section below. +Tested with `agent-framework` 1.19 (Python), `Microsoft.Agents.AI` 1.22 (.NET) and Semantic Kernel 1.44 (Python). ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-microsoft-agent-framework](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-microsoft-agent-framework) skill, which contains every step of this guide. +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-microsoft-agent-framework](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-microsoft-agent-framework) skill. ```text Set up Maple agent tracing for Microsoft Agent Framework in this project. @@ -25,19 +23,17 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. -## Export Agent Framework traces to Maple (Python) +## Export traces from Python -MAF depends on the OpenTelemetry API and SDK but installs no exporter. Add the HTTP one: +MAF installs no exporter, so add the OTLP/HTTP one: ```bash pip install "agent-framework-core>=1.19.0" "agent-framework-openai>=1.14.4" opentelemetry-exporter-otlp-proto-http ``` -The `agent-framework` meta-package works too; it pulls in every connector (Azure, Anthropic, Bedrock and more), so install the core and the connectors you use instead. - -Call `configure_otel_providers()` once at startup, before you create agents: +Call `configure_otel_providers()` once at startup, before you create agents. `enable_sensitive_data=True` records prompts, replies and tool arguments and results; without it the transcript is empty. ```py # telemetry.py @@ -60,24 +56,11 @@ configure_otel_providers( trace.get_tracer_provider().add_span_processor(ConversationIdProcessor()) ``` -`ConversationIdProcessor` comes from the next section. Instrumentation itself is on by default, so this is the whole setup: `configure_otel_providers()` builds the tracer, meter and logger providers and an OTLP exporter for each. It appends `/v1/traces`, `/v1/metrics` and `/v1/logs` to the endpoint for you. - -**Set the protocol explicitly.** MAF defaults to gRPC, unlike the OpenTelemetry spec's `http/protobuf`. Without `otlp_protocol` (or `OTEL_EXPORTER_OTLP_PROTOCOL`), the exporter fails with an import error if the gRPC package isn't installed, or silently retries a gRPC connection that never succeeds if it is. - -The same configuration through environment variables, with a bare `configure_otel_providers()` call: - -```bash -export OTEL_SERVICE_NAME="support-agent" -export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" -export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" -export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" -export ENABLE_SENSITIVE_DATA="true" -export ENABLE_MESSAGE_EVENTS="false" -``` +Keep `otlp_protocol="http/protobuf"`. MAF defaults to gRPC, which Maple doesn't accept. -If your app already configures OpenTelemetry (Azure Monitor, Logfire, your own `TracerProvider`), don't call `configure_otel_providers()`. Add Maple's exporter and `ConversationIdProcessor` to the provider you have, and call `enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)` from `agent_framework.observability`. +If your app already has a `TracerProvider` (Azure Monitor, Logfire, your own), don't call `configure_otel_providers()`. Add Maple's exporter and `ConversationIdProcessor` to your provider, then call `enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)` from `agent_framework.observability`. -## Export Agent Framework traces to Maple (.NET) +## Export traces from .NET ```bash dotnet add package Microsoft.Agents.AI --version 1.22.0 @@ -85,7 +68,7 @@ dotnet add package Microsoft.Agents.AI.OpenAI --version 1.22.0 dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol --version 1.19.1 ``` -Instrument the agent with `UseOpenTelemetry()`. In 1.22 it also instruments the agent's chat client, so you get the `chat` and `execute_tool` spans without wrapping the `IChatClient` yourself: +`UseOpenTelemetry()` on the agent also instruments its chat client: ```csharp using System.ClientModel; @@ -123,24 +106,11 @@ AIAgent agent = openAi.GetChatClient("gpt-4o-mini").AsIChatClient() .Build(); ``` -Pass each tool a `name`. `AIFunctionFactory.Create` otherwise uses the method name, and a local function in a top-level `Program.cs` compiles to something like `_Main_g_GetWeather_0_3`, which is what Maple then shows as the tool. +Keep the leading `*` in `AddSource`: the default source names start with `Experimental.`, so `AddSource("Microsoft.Agents.AI.*")` matches nothing. Keep `/v1/traces` in the endpoint and the `HttpProtobuf` line. Pass each tool a `name`, or a local function in `Program.cs` shows up as something like `_Main_g_GetWeather_0_3`. -The wildcards matter. With no `sourceName`, the agent emits on `Experimental.Microsoft.Agents.AI`, and `Microsoft.Extensions.AI` chat clients on `Experimental.Microsoft.Extensions.AI`. A plain `AddSource("Microsoft.Agents.AI.*")`, as some samples use, matches neither, and you get zero spans. Workflows emit on `Microsoft.Agents.AI.Workflows` once you call `.WithOpenTelemetry()` on the `WorkflowBuilder`. +## Group each conversation into one session -When you set `Endpoint` in code, it must include `/v1/traces`, and the .NET exporter also defaults to gRPC, so keep the `HttpProtobuf` line. .NET agent spans show up under the **Unidentified** framework in Maple (the Python instrumentation scope is what identifies MAF); sessions, transcripts, tokens and tools work the same. The agent span is named `invoke_agent support_agent()` in .NET, while `gen_ai.agent.name` stays `support_agent`. - -## Group every turn of a conversation into one session - -Maple groups traces into a session by `gen_ai.conversation.id`. On an ordinary `agent.run()`, MAF writes that attribute only from the provider's own conversation id (`AgentSession.service_session_id`), which exists when the Responses API stores history (`store=True`) or a Foundry agent owns the thread. With Chat Completions, OpenRouter, Ollama, or anything that keeps history in an `AgentSession` in your process, it's never set. The session's own `session_id` isn't exported either. - -Skip this and every `agent.run()` is its own session in Maple, named after its trace id: a ten-message chat becomes ten one-turn sessions, and a human-in-the-loop approval splits a turn in two. - -Two things look like fixes and aren't: - -- **`gen_ai.agent.id`** is a random id per `Agent` object. A server that builds one agent at startup stamps the same value on every user's conversation. -- **The `conversation_id` chat option** (`Agent(..., conversation_id=...)` or `options={"conversation_id": ...}`) does reach the `invoke_agent` span, but it's a provider setting, not a label. It turns off the in-memory history MAF adds to a session, and the Responses API client sends it as `previous_response_id`. - -Instead, add this processor. It stamps the id on every span started inside a `conversation()` block, including the `chat`, `execute_tool` and workflow spans, and it follows `asyncio` tasks the block creates: +Maple groups turns by `gen_ai.conversation.id`. MAF sets it only when the provider stores the conversation (Responses API with `store=True`, Foundry agents). This processor stamps your id on every span started inside a `conversation()` block: ```py # maple_tracing.py @@ -169,21 +139,13 @@ def conversation(conversation_id: str): _conversation_id.reset(token) ``` -Wrap each request in it. The session's `session_id` is a good id: it's stable for the life of the conversation and survives `to_dict()`/`from_dict()` if you persist sessions. +Wrap each request in it. `session.session_id` works as the id, or create the session with your own chat id: `agent.create_session(session_id=chat_id)`. ```py -from agent_framework import Agent, AgentSession -from agent_framework.openai import OpenAIChatCompletionClient +from agent_framework import AgentSession from maple_tracing import conversation -agent = Agent( - OpenAIChatCompletionClient(model="gpt-4o-mini"), - "You are a helpful assistant.", - name="support_agent", - tools=[get_weather, calculate], -) - async def handle_message(session: AgentSession, text: str) -> str: with conversation(session.session_id): @@ -191,15 +153,9 @@ async def handle_message(session: AgentSession, text: str) -> str: return response.text ``` -Create the session with your own chat id if you have one: `agent.create_session(session_id=chat_id)`. For streaming, keep the whole `async for` inside the block, because the `chat` span starts on the first pull: - -```py -with conversation(session.session_id): - async for update in agent.run(text, session=session, stream=True): - print(update.text, end="") -``` +When streaming, keep the whole `async for` loop inside the block. Build workflows inside it too, or `WorkflowBuilder.build()` shows up as a separate one-span session. Don't use `gen_ai.agent.id` or the `conversation_id` chat option as the id; the first is shared by every user and the second turns off MAF's in-memory history. -In .NET, the same idea is an `Activity` processor with an `AsyncLocal`: +In .NET, use an `Activity` processor with an `AsyncLocal` and set it before `RunAsync`: ```csharp using System.Diagnostics; @@ -222,64 +178,9 @@ ConversationIdProcessor.Current.Value = chatId; // once per request, before RunA var response = await agent.RunAsync(userMessage, session); ``` -## Record prompts, responses and tool calls - -Content is off by default in both languages. Without it, Maple shows the model, tokens and timing of each call, but the transcript is empty and tool calls have no arguments or results. - -- **Python:** `enable_sensitive_data=True` or `ENABLE_SENSITIVE_DATA=true`. Content lands on the spans as `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, which is the shape Maple reads. -- **.NET:** `EnableSensitiveData = true` in the `UseOpenTelemetry` callback, or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=true`. - -Two Python settings change where content goes: - -- `OTEL_SEMCONV_STABILITY_OPT_IN`: MAF uses the current GenAI conventions only while this variable is unset or includes `gen_ai_latest_experimental`. If another library in your app sets it to something like `http`, MAF drops back to the v1.36 conventions: the message attributes and tool arguments disappear from the spans and the content goes to log events, which Maple doesn't read for sessions. Use `OTEL_SEMCONV_STABILITY_OPT_IN=http,gen_ai_latest_experimental`. -- `ENABLE_MESSAGE_EVENTS` defaults to `true` and sends a second copy of every message as OTLP log records. Maple builds the transcript from span attributes, so turn it off unless another tool reads those logs. - -With content on, everything the user types and every tool result is stored in Maple. To keep a service's content out, leave sensitive data off there and keep the spans; sessions, tokens, tool names and failures still work. Tool definitions (`gen_ai.tool.definitions`, the JSON schema of each tool) are sent on every `invoke_agent` span even with sensitive data off. - -## Tools, errors and sub-agents - -Each function tool call is an `execute_tool ` span with `gen_ai.tool.name`, `gen_ai.tool.call.id`, the arguments and the result. When a tool raises, MAF marks that span ERROR with `error.type` (the exception class) and the message in the status, then hands the model an error string as the tool result so the run continues. Maple counts the failure on the tool, and the agent and `chat` spans above it stay green, which is correct: the turn itself succeeded. .NET does the same (`error.type=System.InvalidOperationException`, for example). - -Python MAF also logs each tool failure as an ERROR and a WARN record. `configure_otel_providers()` exports those to Maple as logs, next to the span. - -Give every agent a distinct `name`. Maple opens a sub-agent lane when an `invoke_agent` span's `gen_ai.agent.name` differs from its parent's, and an agent with no name gets its UUID as the name. +## Flush before a script exits -The simplest multi-agent pattern is agents as tools. The orchestrator's `execute_tool weather_worker` span has the worker's `invoke_agent weather_worker` span as its child, which Maple shows as a delegation with the task and the answer: - -```py -weather_worker = Agent(client, "Report current weather.", name="weather_worker", tools=[get_weather]) -budget_worker = Agent(client, "Estimate trip costs.", name="budget_worker", tools=[calculate]) - -orchestrator = Agent( - client, - "Delegate each part of the briefing to the right worker, then summarize.", - name="orchestrator", - tools=[weather_worker.as_tool(), budget_worker.as_tool()], -) - -with conversation(session.session_id): - result = await orchestrator.run(brief, session=session) -``` - -Workflows (`WorkflowBuilder`, or the `SequentialBuilder`, `ConcurrentBuilder`, `HandoffBuilder` and `MagenticBuilder` orchestrations) trace each executor as `executor.process ` with the agent's `invoke_agent` span inside it. Fan-in is recorded as span links, not parent-child, so parallel workers appear as siblings under `workflow.run`. The executors run as `asyncio` tasks and inherit the conversation id from the block. - -Build the workflow inside the `conversation()` block too. `WorkflowBuilder.build()` emits its own one-span `workflow.build` trace. Inside the block it joins the session as a short extra trace with no model calls. Maple shows it as an empty first turn with no label or agent, the real run is turn 2, and the session has no title. Built once at import time, it becomes a one-span session of its own instead. - -Tools that need approval (`@tool(approval_mode="always_require")`) end the run with a pending request. The resume is a new `agent.run()` and a new trace; with the processor in place, it joins the same session. A rejected call emits no `execute_tool` span at all; the rejection only appears in the next `chat` span's input messages. - -## Tokens and cost - -Every `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `cache_read.input_tokens`, `cache_creation.input_tokens` and `reasoning.output_tokens` when the provider returns them. Streaming works without extra settings: the OpenAI chat completions client requests `include_usage` itself. - -The `invoke_agent` span repeats the total of that agent's own `chat` calls. A sub-agent called as a tool reports its calls on its own `invoke_agent` span, not the orchestrator's. Maple subtracts the child `chat` spans, so the session total counts each call once. - -In .NET, streamed `chat` spans also carry `gen_ai.response.time_to_first_chunk`, which Maple shows as time to first token. Python MAF doesn't record it. - -MAF emits no cost attribute, and Maple doesn't price tokens, so sessions show as unpriced. `gen_ai.provider.name` on `chat` spans is the client type (`openai` for the OpenAI client, even when it points at OpenRouter or Ollama), and `server.address` holds the real base URL. - -## Flush before a short-lived process exits - -`configure_otel_providers()` uses batch processors, and MAF has no flush helper. A script, CLI, notebook cell or serverless handler that exits without flushing loses its last turns. Shut the three providers down in a `finally`: +`configure_otel_providers()` batches spans. Scripts, CLIs and notebooks that exit without flushing lose their last turns, so shut the providers down in a `finally`: ```py from opentelemetry import _logs, metrics, trace @@ -296,22 +197,16 @@ finally: shutdown_telemetry() ``` -For a long-running server, don't shut down per request. In a serverless handler that is frozen between invocations, call `trace.get_tracer_provider().force_flush()` before returning. - -In .NET, `using var tracerProvider` flushes on dispose at the end of `Main`. In a hosted app, `AddOpenTelemetry()` flushes on graceful shutdown. +In a serverless handler, call `trace.get_tracer_provider().force_flush()` before returning. In .NET, `using var tracerProvider` flushes when `Main` ends. ## Semantic Kernel -Semantic Kernel (SK) 1.44 is in maintenance while Microsoft moves agent work to MAF, and its telemetry differs in four ways. - -**It's gated by environment variables read at import time.** Set them before the first `import semantic_kernel`, or SK emits no GenAI spans and logs no warning. SK brings no exporter or provider, so set up your own: +Semantic Kernel (SK) reads its telemetry switches at import time, so set them before the first `import semantic_kernel`. SK brings no exporter, so configure your own provider with the same `ConversationIdProcessor`: ```bash pip install "semantic-kernel>=1.44.1" opentelemetry-sdk opentelemetry-exporter-otlp-proto-http ``` -`semantic-kernel` 1.44 depends on a pre-release `azure-ai-agents`, so `uv` needs `--prerelease=allow`; pip resolves it as is. - ```py # telemetry.py: import this before anything that imports semantic_kernel import os @@ -340,54 +235,19 @@ provider.add_span_processor( trace.set_tracer_provider(provider) ``` -**Model-call content goes to Python logging, not spans.** The `chat ` spans carry model, tokens and finish reason; the prompts and replies are log records Maple doesn't read. The transcript comes from the agent instead: `ChatCompletionAgent` puts the messages you pass in and the reply on its `invoke_agent` span as `gen_ai.input.messages` and `gen_ai.output.messages`. Pass the messages positionally, `await agent.get_response(text, thread=thread)`. SK reads them from the second positional argument, and `get_response(messages=text, ...)` records an empty input. Code that calls the kernel directly (`kernel.invoke_prompt`, a chat service without an agent) has no transcript in Maple. - -**There's no conversation id either.** `ChatHistoryAgentThread.id` isn't exported, so use the same `ConversationIdProcessor` and wrap each request in `with conversation(thread_id):`. - -**The agent runtime splits orchestrations across traces.** `ConcurrentOrchestration`, `SequentialOrchestration` and the other orchestrations run on `InProcessRuntime`, which starts each message on a new trace. One run became 10 traces in our test. Open your own span around the run and start the runtime inside the `conversation()` block, so the runtime's tasks inherit both: - -```py -tracer = trace.get_tracer("support-agent") - -with conversation(conversation_id), tracer.start_as_current_span("briefing"): - runtime = InProcessRuntime() - runtime.start() - result = await orchestration.invoke(task=brief, runtime=runtime) - output = await result.get() - await runtime.stop_when_idle() -``` - -Smaller differences: tool spans are named `execute_tool -`, failing tools get ERROR status and `error.type` but no result attribute, finish reasons are Python enum names (`FinishReason.STOP`) so Maple's reply-length and refusal checks can't read them, and a `temperature` of `0` is left off the span. Flush with `provider.shutdown()` as above. - -In .NET, SK reads the same variables, or the `AppContext` switches `Microsoft.SemanticKernel.Experimental.GenAI.EnableOTelDiagnostics` and `...EnableOTelDiagnosticsSensitive`. Add `AddSource("Microsoft.SemanticKernel*")` to the tracer provider. SK .NET spans show up under the **Unidentified** framework in Maple. +Wrap each turn in `with conversation(thread_id):`. The transcript comes from `ChatCompletionAgent`, so pass messages positionally (`await agent.get_response(text, thread=thread)`); the `messages=` keyword records an empty input. Code that calls the kernel without an agent has no transcript in Maple. ## Check that it works -Run one conversation of two or three turns, one of which calls a tool. Sessions appear in **Agent Sessions** within about a minute. You should see: - -- **One session per conversation**, labeled **Microsoft Agent Framework** (or **Semantic Kernel**) for Python, whose id is your conversation id, not `trace:…`. A second conversation is a second session. -- **One turn per `agent.run()`**, each rooted at `invoke_agent support_agent` with `chat gpt-4o-mini` and `execute_tool get_weather` spans under it. An approval resume is its own turn in the same session. In .NET the root is `invoke_agent support_agent()`. -- **A transcript** with the system instructions, your messages and the replies, labeled by the first line of each user message. Semantic Kernel shows only each turn's message and the agent's reply, with tool calls as tool spans. -- **Tool calls** with arguments and results, and failed tools counted under **Tool errors**. -- **Tokens** on every model call, including streamed ones. Cost shows as unpriced. -- For multi-agent runs, a lane per agent name. +Run a conversation of two or three turns where one turn calls a tool. Within about a minute, **Agent Sessions** shows one session for it, labeled **Microsoft Agent Framework** or **Semantic Kernel** (.NET shows **Unidentified**), with one turn per `agent.run()`, the transcript, and tool calls with their arguments and results. Cost shows as unpriced because MAF doesn't emit one. ## Troubleshooting -- **Nothing arrives, no error.** The protocol is still gRPC. Set `otlp_protocol="http/protobuf"` or `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf`. If you set `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`, it's used as is and needs the `/v1/traces` path. -- **`ImportError: opentelemetry-exporter-otlp-proto-grpc is required`.** Same cause: the protocol defaulted to gRPC. -- **Every turn is its own session.** No `gen_ai.conversation.id`. Register `ConversationIdProcessor` and wrap the call, including the whole `async for` when streaming, in `conversation()`. -- **Every user shares one session.** The id is a process-wide constant: `gen_ai.agent.id`, a module-level string, or `conversation()` called once at startup. Use a per-conversation id. -- **An extra session with a single `workflow.build` span.** The workflow was built outside `conversation()`. Build it per request inside the block. -- **Spans but an empty transcript (Python).** Sensitive data is off, or `OTEL_SEMCONV_STABILITY_OPT_IN` is set without `gen_ai_latest_experimental`. -- **The agent forgets earlier turns after adding tracing.** You passed `conversation_id` as a chat option, which disables the in-memory history. Remove it and use the processor. -- **Every message part is one character.** `Message("user", text)` iterates a bare string; pass a list: `Message("user", [text])`. -- **OpenRouter rejects the second turn with a `previous_response_id` error.** `OpenAIChatClient` is the Responses API client. Use `OpenAIChatCompletionClient` for OpenRouter and other Chat Completions endpoints. -- **A WARN log "Ignored an approval response ... did not match the active approval occurrence identity", but the tool ran.** Seen on every approval resume with `to_function_approval_response()` in 1.19. The `execute_tool` span shows the approved call ran once; the log is noise. -- **.NET: no spans at all.** `AddSource` doesn't match the default `Experimental.` prefix. Use `AddSource("*Microsoft.Agents.AI*")`. -- **.NET: tools named like `_Main_g_GetWeather_0_3`.** The tool is a local function in a top-level `Program.cs`. Pass `name:` to `AIFunctionFactory.Create`. -- **Semantic Kernel: `AutoFunctionInvocationLoop` spans but no `chat` or `invoke_agent`.** The `SEMANTICKERNEL_EXPERIMENTAL_GENAI_*` variables were set after `semantic_kernel` was imported. -- **Semantic Kernel: an orchestration shows as many small sessions or traces.** Wrap the run in your own span and `conversation()` and start the runtime inside it. +- **Nothing arrives, or `ImportError: opentelemetry-exporter-otlp-proto-grpc is required`.** The protocol defaulted to gRPC. Set `otlp_protocol="http/protobuf"` or `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf`. +- **Every turn is its own session.** Register `ConversationIdProcessor` and wrap each call, including the whole streaming loop, in `conversation()`. +- **Spans but an empty transcript.** Sensitive data is off, or `OTEL_SEMCONV_STABILITY_OPT_IN` is set without `gen_ai_latest_experimental` (use `http,gen_ai_latest_experimental`). +- **.NET: no spans at all.** Use `AddSource("*Microsoft.Agents.AI*")` with the leading `*`. +- **Semantic Kernel: no `chat` or `invoke_agent` spans.** The `SEMANTICKERNEL_EXPERIMENTAL_GENAI_*` variables were set after `semantic_kernel` was imported. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/openai-agents.md b/apps/landing/src/content/docs/agent-tracing/openai-agents.md index f3b05b11af..45585c0c0e 100644 --- a/apps/landing/src/content/docs/agent-tracing/openai-agents.md +++ b/apps/landing/src/content/docs/agent-tracing/openai-agents.md @@ -1,17 +1,17 @@ --- title: "Trace OpenAI Agents SDK runs with OpenTelemetry" -description: "Send OpenAI Agents SDK runs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools, handoffs and agents as tools." +description: "Send OpenAI Agents SDK runs to Maple as one Agent Session per conversation, with the transcript, tool calls and tokens." group: "AI Agents" order: 12 navLabel: "OpenAI Agents SDK" icon: "openai" --- -The OpenAI Agents SDK traces every run out of the box, but not with OpenTelemetry. Its tracing pipeline builds its own traces and spans (agent, generation, function, handoff, guardrail) and uploads them to the OpenAI dashboard. To get them into Maple you swap that uploader for OpenInference's `openinference-instrumentation-openai-agents`, which turns each SDK span into an OpenTelemetry span as it ends. +The OpenAI Agents SDK traces every run, but it uploads those traces to the OpenAI dashboard instead of exporting OpenTelemetry. OpenInference's `openinference-instrumentation-openai-agents` bridges them to OpenTelemetry so Maple can read them. -The SDK's own way to tie the traces of one chat together, `group_id`, never reaches OpenTelemetry. The bridge drops it, so a ten-message conversation shows up in Maple as ten one-turn sessions until you wrap each run in OpenInference's `using_session`. A few more defaults need changing: the bridge writes OpenInference attributes that Maple's session page doesn't decode for this framework, and streamed calls to any provider other than OpenAI lose their tokens and their reply. +The part to get right is the session id. The SDK's `group_id` never reaches OpenTelemetry, so you wrap each run in OpenInference's `using_session`, or every message shows up as its own session. -This guide covers `openai-agents` 0.22 with `openinference-instrumentation-openai-agents` 2.5 on Python 3.10 to 3.14. TypeScript (`@openai/agents`) works with less detail; see [TypeScript](#typescript-openaiagents). +Tested with `openai-agents` 0.22 and `openinference-instrumentation-openai-agents` 2.5 on Python 3.10 to 3.14. ## Quick setup with a coding agent @@ -25,17 +25,15 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is under **Settings → Ingestion**. EU organizations should say EU region. -## Install the bridge and export to Maple +## Install the bridge and point it at Maple ```bash pip install "openai-agents>=0.22" "openinference-instrumentation-openai-agents>=2.5" \ "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Point the exporter at Maple with the standard OpenTelemetry variables: - ```bash export OTEL_SERVICE_NAME=support-agent export OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production @@ -43,9 +41,11 @@ export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" ``` -EU organizations use `https://ingest.eu.maple.dev`. `OTLPSpanExporter()` with no arguments appends `/v1/traces` to that base URL and sends `http/protobuf`. If you pass `endpoint=` in code instead, it's used as is, so it has to end in `/v1/traces`. +EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, give the full URL ending in `/v1/traces`. + +## Register the bridge -Then add a `tracing.py` and import it at the top of your entry point: +Add a `tracing.py` and import it at the top of your entry point, before the first `Runner.run`: ```py # tracing.py @@ -109,43 +109,19 @@ OpenAIAgentsInstrumentor().instrument( ) ``` -What each part does: - -- **`enable_genai_semconv=True`** makes OpenInference write the OpenTelemetry GenAI attributes (`gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*`, `gen_ai.agent.name`, `gen_ai.tool.*`) next to its own `llm.*` attributes when each span ends. Without it, the Agent Sessions list shows token counts but the session page has no transcript. `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, but only if it's set before `TraceConfig` is built; passing it in code avoids that trap. -- **`MapleSpanFixes`** fills three gaps in what the bridge exports. The GenAI dual-write fills `gen_ai.tool.call.arguments` from `tool.parameters`, which is the tool's JSON schema, so every tool call would show `{"properties": {"city": ...}}` instead of `{"city": "Berlin"}`; the dual-write never overwrites a key that's already set, so the processor sets the real arguments first. Handoff spans get no tool name, so it names them `transfer_to_`, the tool the model called. And a streamed Chat Completions call records its reply as a Responses API object the bridge can't parse, so without the fix the streamed turn has no reply in the transcript; the processor rewrites it into a Chat Completions message before the bridge reads it. -- **`set_trace_processors([...])` plus `exclusive_processor=False`** puts `MapleSpanFixes` ahead of the OpenInference processor, which the ordering requires. It also removes the SDK's default processor, so nothing is uploaded to the OpenAI dashboard and you don't need an OpenAI key for tracing. To keep that upload too, pass `set_trace_processors([MapleSpanFixes(), default_processor()])` with `default_processor` from `agents.tracing.processors`. - -The bridge hooks into the SDK's processor list rather than patching imports, so import order only matters in one way: `tracing.py` has to run before the first `Runner.run`. - -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), don't create a second one. Add the OTLP exporter to the existing provider and pass that provider to `instrument()`. - -Don't call `set_tracing_disabled(True)`, set `OPENAI_AGENTS_DISABLE_TRACING=1` or pass `RunConfig(tracing_disabled=True)` to stop the upload to OpenAI. Those switch off the SDK's whole tracing pipeline, which is where the bridge gets its data, and you get zero spans. +`enable_genai_semconv=True` makes the bridge write the OpenTelemetry GenAI attributes that Maple builds the transcript from. Without it, the session page is empty. -### Models from other providers +`MapleSpanFixes` corrects three things the bridge records wrong: tool arguments (it would copy the tool's JSON schema), handoff tool names, and the reply of streamed Chat Completions calls. It has to run before the OpenInference processor, which is why `set_trace_processors` comes first and `instrument()` gets `exclusive_processor=False`. -With an `OPENAI_API_KEY`, agents use OpenAI's Responses API and need nothing extra. To use another provider through an OpenAI-compatible endpoint (OpenRouter, LiteLLM, vLLM, Ollama), give the agent a `OpenAIChatCompletionsModel` and turn on `include_usage`: +`set_trace_processors` also stops the upload to OpenAI, so tracing needs no OpenAI key. Don't use `set_tracing_disabled(True)` or `OPENAI_AGENTS_DISABLE_TRACING=1` for that. They turn off the pipeline the bridge reads from, and you get no spans. -```py -import os - -from agents import Agent, ModelSettings, OpenAIChatCompletionsModel -from openai import AsyncOpenAI - -client = AsyncOpenAI(base_url="https://openrouter.ai/api/v1", api_key=os.environ["OPENROUTER_API_KEY"]) - -agent = Agent( - name="assistant", - instructions="You are a helpful assistant. Be brief.", - model=OpenAIChatCompletionsModel(model="openai/gpt-4o-mini", openai_client=client), - model_settings=ModelSettings(include_usage=True), -) -``` +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add the OTLP exporter to it and pass it to `instrument()` instead of creating a second one. -The SDK only asks for token usage on streamed Chat Completions calls when the client points at `api.openai.com`. For any other base URL it sends no `stream_options`, the provider returns no usage chunk, and every `Runner.run_streamed` turn has zero tokens. `include_usage=True` sends `stream_options={"include_usage": true}`; non-streamed calls report usage either way. +If an agent uses `OpenAIChatCompletionsModel` with a base URL other than OpenAI's (OpenRouter, LiteLLM, vLLM, Ollama), give it `model_settings=ModelSettings(include_usage=True)`. Otherwise streamed turns report zero tokens. -## Group a conversation into one session +## Group each conversation into one session -Each `Runner.run` is its own trace. The SDK has two ids that look like the answer, and neither reaches OpenTelemetry: `RunConfig(group_id=...)` and the `session_id` of a `SQLiteSession` stay inside the SDK. Maple groups this framework's traces by the `session.id` attribute, and the bridge only sets it inside OpenInference's `using_session`: +Maple groups this framework's traces by `session.id`, and the bridge only sets it inside `using_session`. Wrap every run: ```py from agents import RunConfig, Runner, SQLiteSession @@ -163,85 +139,15 @@ async def handle_message(conversation_id: str, text: str) -> str: return result.final_output ``` -Use your app's own conversation id, the one it already stores the chat under, and give the SDK session the same one. A new UUID per request gives you one session per message again, and a constant gives every user one shared session. - -`using_session` stores the id in a Python contextvar. The bridge copies it onto every span it creates inside the block, including tool calls that run concurrently and agents called as tools, because asyncio tasks inherit the context. For streaming, call `Runner.run_streamed` inside the block; the SDK starts its background task there, so iterating `stream_events()` afterwards is fine: - -```py -with using_session(conversation_id): - result = Runner.run_streamed(agent, text, session=SQLiteSession(conversation_id, "chats.db")) - async for event in result.stream_events(): - ... -``` - -`workflow_name` names the trace's root span. The default is `Agent workflow`, which is the same for every run in your app. Keep `chat`, `completion` and `tool` out of the name. The bridge also gives each run a bookkeeping span with that name and no operation, and Maple classifies such spans by name: `support chat` would add one phantom LLM call per run, and `tool` in the name a phantom tool call. A name ending in `workflow` or `agent` is safe. - -If you skip `using_session`, every run shows up in **Agent Sessions** as its own one-turn session named after its trace id. Setting `gen_ai.conversation.id` yourself doesn't help either: Maple reads `session.id` first for this framework, and the dual-write already copies it to `gen_ai.conversation.id`. - -## Record prompts, responses and tool calls - -Content capture is on by default, in two places. The SDK records model inputs and outputs and tool arguments and results on its spans (`trace_include_sensitive_data`, default `True`), and OpenInference copies them onto the OpenTelemetry span. With the dual-write on, each model span carries the content three times: as flattened `llm.input_messages.N.*` keys, as the `input.value` JSON, and as `gen_ai.input.messages`, which is the one Maple renders. Every model span also repeats the whole conversation so far. Maple has no per-attribute limit and accepts requests up to 20 MiB, so this costs bandwidth rather than data. - -To keep prompts and outputs out of your traces, turn capture off at the source: - -```bash -export OPENAI_AGENTS_TRACE_INCLUDE_SENSITIVE_DATA=false -``` - -or per run with `RunConfig(trace_include_sensitive_data=False)`. The session still shows turns, models, tool names, tokens and failures, with an empty transcript. Tool error messages are replaced by `Tool execution failed. Error details are redacted.` - -OpenInference's own switches (`TraceConfig(hide_inputs=True, hide_outputs=True)` or `OPENINFERENCE_HIDE_INPUTS` / `OPENINFERENCE_HIDE_OUTPUTS`) redact at the bridge instead, and also empty the transcript, because the GenAI messages are derived from the attributes they hide. For partial redaction, such as masking emails, drop or rewrite attributes in an OpenTelemetry Collector with the `transform` or `redaction` processor. - -## Tools, errors, handoffs and agents as tools - -Each function tool call is a span named after the tool, with `gen_ai.operation.name` `execute_tool`, `gen_ai.tool.name`, the tool's description, its arguments (with `MapleSpanFixes`) and its result in `gen_ai.tool.call.result`. - -Maple shows that result on the tool span only when it's JSON, as it is for a tool that returns a dict or a list. A plain-text result shows as not captured there: a tool that returns a string, an agent called as a tool, a handoff, or a failed tool's error text. The transcript still shows it, as the tool message the model received on its next call. - -A tool that raises is marked failed without extra code. The SDK catches the exception, sends the model `An error occurred while running the tool. Please try again. Error: ...` as the tool result, and records the error on its span. The bridge turns that into status `ERROR` with a message like `Error running tool (non-fatal): {'tool_name': 'fetch_transport_data', 'error': 'transport data service unavailable (503)'}`, and Maple counts the call as failed on the session and on the tool's page. A tool that returns an error string instead of raising counts as a success. - -Every agent the run enters gets a span named after it, with `gen_ai.agent.name`, and Maple opens a lane for each agent whose name differs from its caller's. Give every `Agent` a distinct `name`. The spans in between are the bridge's bookkeeping: a `CHAIN` span named after the workflow per `Runner.run`, and a `turn` span per step of the agent loop. - -The two multi-agent idioms look different in a trace: - -- **Agents as tools** (`agent.as_tool(tool_name=..., tool_description=...)`): the calling agent's tool span contains the whole nested run, with the sub-agent's model and tool calls inside. The tool's arguments and result are the sub-agent's input and output. -- **Handoffs** (`handoffs=[...]`): the model calls a `transfer_to_` tool, which shows up as a `handoff to ` span with the source and target agent names as input and output. The target agent's span is a sibling of the source agent's, not a child, and the conversation continues there. Maple counts each handoff as a tool call, named `transfer_to_` by `MapleSpanFixes`. - -Parallel sub-agents work if every run is inside one SDK trace and one `using_session` block, so they share a trace id: - -```py -import asyncio - -from agents import trace - -with using_session(conversation_id), trace("amsterdam briefing"): - weather, budget = await asyncio.gather( - Runner.run(weather_worker, "Weather in Amsterdam?"), - Runner.run(budget_worker, "3-day budget for Amsterdam?"), - ) -``` - -Without the `trace()` block, each `Runner.run` starts its own trace, and the briefing shows up as several turns of the session instead of one. - -Tools that need approval (`@function_tool(needs_approval=True)`) show up twice for one approved call. The run that pauses records a tool span with arguments but no result, and the resumed `Runner.run(agent, state)` records the real execution. The resume is its own trace, so the session shows two turns for the approval: the pause and the resumed run, both labelled with the user's message. Wrap the resume in the same `using_session` id or it becomes a separate session. - -Two gaps remain. Tool spans have no `gen_ai.tool.call.id`, because the SDK's function span doesn't carry the model's call id, so Maple can't tie a tool span to the exact call in the model's reply. And on the Chat Completions path, model spans have no `gen_ai.response.id`; the Responses API path has one. +Use the id your app already stores the chat under, and give the SDK session the same one. A new UUID per request gives you one session per message, and a constant puts every user in one session. -## Tokens and cost +For streaming, call `Runner.run_streamed` inside the `with` block. Iterating `stream_events()` after the block is fine. -Every model span carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals. On the Responses API, cached input tokens are recorded too, and reasoning tokens as `llm.token_count.completion_details.reasoning`. Streamed Chat Completions calls need `include_usage=True`, as above. +`workflow_name` names each trace's root span. Keep `chat`, `completion` and `tool` out of it, or Maple counts a phantom model or tool call per run. A name ending in `workflow` or `agent` is safe. -Model spans are named `generation` on the Chat Completions path and `response` on the Responses path. On the Chat Completions path the model is the id you configured (`openai/gpt-4o-mini`). On the Responses path it's the name OpenAI returns, which is usually a dated snapshot such as `gpt-4o-mini-2024-07-18`. The provider is `openai` for both, even for an Anthropic model behind OpenRouter, because the SDK only speaks the OpenAI API. Maple uses the OpenAI token convention for it, where cached tokens are part of the input total, which is also what OpenRouter returns. +## Flush in short-lived processes -The SDK also totals usage per run and per turn, but the bridge doesn't export those totals, so nothing is counted twice. - -Maple shows cost only when a span carries one, and neither the SDK nor the bridge records cost. Sessions show as **unpriced**, with token counts. If you route through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call. Chat Completions model spans have no `gen_ai.response.id`, so nest the Broadcast spans under them as described in [Join Broadcast to your own traces](/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces), or each call is counted twice. - -Don't add `openinference-instrumentation-openai` next to the Agents bridge. It patches the `openai` client the SDK calls, so every model call gets a second model span under the `generation` span. The same goes for Logfire's `instrument_openai_agents()`, Langfuse's or Traceloop's Agents instrumentation, and the OpenTelemetry project's `opentelemetry-instrumentation-genai-openai-agents`: pick one. - -## Flush spans before the process exits - -`BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` registers an `atexit` handler that flushes on a normal interpreter exit, which covers most scripts and CLIs. It doesn't run when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: +The provider flushes on a normal interpreter exit. A killed process, a serverless invocation or a notebook needs an explicit flush: ```py from tracing import provider @@ -252,66 +158,25 @@ finally: provider.force_flush() # serverless: before returning; notebooks: after each run ``` -Call `provider.shutdown()` instead when the process is about to exit. The SDK's own `flush_traces()` doesn't help here: the bridge's `force_flush()` is a no-op, and the spans wait in the OpenTelemetry batch processor. +The SDK's `flush_traces()` doesn't help here, because the spans wait in the OpenTelemetry batch processor. ## Check that it works -Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: - -- **One session** for the conversation, framework **OpenAI Agents SDK**, with one turn per `Runner.run`. Each turn's trace starts at a span named after your `workflow_name`. -- **The transcript**: the agent's instructions as the system message, your messages, the model's replies (streamed ones included) and its tool calls. -- **Model calls** named `generation` (Chat Completions) or `response` (Responses API), as many as the app made, each with a model and input and output tokens, streamed turns included. -- **Tool calls** named after your tools, such as `get_weather`, with arguments, and results where the tool returned JSON. A tool that raised is marked failed, and the session's verdict names it under **Tool availability**. -- **Agents**: one lane per agent name, such as `assistant`, plus sub-agents and handoff targets. -- **Cost**: unpriced. - -A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. - -## TypeScript (`@openai/agents`) - -The same bridge exists for the TypeScript SDK as `@arizeai/openinference-instrumentation-openai-agents`. We ran it with `@openai/agents` 0.18 and bridge 0.2.15, and it works with Maple with three differences. Maple doesn't identify it as the OpenAI Agents SDK, so the framework shows as **Unidentified**. The TypeScript bridge has no GenAI dual-write, so the transcript is built from the OpenInference `input.value` and `output.value` JSON. Inputs render as messages, but `output.value` is the raw API response, which Maple can't read yet. Each model call's reply and the tool calls it asked for are missing from that call; an earlier reply only appears as history in the next call's input, so the last reply of every turn is missing. Tool spans show their input and output as messages but have no arguments and result fields, and there's no `gen_ai.agent.name`, so no agent lanes. And the session id has to be `gen_ai.conversation.id`, because Maple doesn't read `session.id` for unidentified OpenInference spans: - -```ts -import * as agents from "@openai/agents" -import { context } from "@opentelemetry/api" -import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" -import { detectResources, envDetector } from "@opentelemetry/resources" -import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace-node" -import { setAttributes } from "@arizeai/openinference-core" -import { OpenAIAgentsInstrumentation } from "@arizeai/openinference-instrumentation-openai-agents" - -export const provider = new NodeTracerProvider({ - resource: detectResources({ detectors: [envDetector] }), // OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES - spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())], // OTEL_EXPORTER_OTLP_* variables -}) -provider.register() -new OpenAIAgentsInstrumentation({ tracerProvider: provider }).manuallyInstrument(agents) - -export function handleMessage(conversationId: string, text: string) { - const ctx = setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }) - return context.with(ctx, () => agents.run(agent, text, { session })) -} -``` +Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. You should see one session with the framework **OpenAI Agents SDK**, one turn per `Runner.run`, a transcript with the tool calls and their arguments, and input and output tokens on every model call. + +Cost shows as unpriced, because neither the SDK nor the bridge records it. -The `envDetector` line matters: the 2.x `NodeTracerProvider` doesn't read `OTEL_SERVICE_NAME` on its own, and every span arrives as `unknown_service:node`. The same environment variables as above configure the exporter. In a script, `await provider.forceFlush()` before exiting. +## TypeScript -Models, tokens and tool names come through as in Python. If you need the full session page in TypeScript today, the [provider SDKs guide](/docs/agent-tracing/provider-sdks) shows how to emit the GenAI attributes yourself. +The TypeScript bridge, `@arizeai/openinference-instrumentation-openai-agents`, works with less detail: the framework shows as **Unidentified**, model replies are missing from the transcript, there are no agent lanes, and the session id has to be set as `gen_ai.conversation.id`. The [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) has the setup. For the full session page in TypeScript, emit the GenAI attributes yourself as in the [provider SDKs guide](/docs/agent-tracing/provider-sdks). ## Troubleshooting -- **No spans at all.** `set_tracing_disabled(True)`, `OPENAI_AGENTS_DISABLE_TRACING=1` or `RunConfig(tracing_disabled=True)` is set somewhere, or `tracing.py` ran after the first `Runner.run`. Remove the switch and import `tracing` first. -- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. -- **Tokens in the list, empty session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)`, or set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` before `instrument()` runs. Also check the bridge is 2.5 or later: older versions don't record agent names. +- **No spans at all.** Tracing is disabled somewhere (`set_tracing_disabled`, `OPENAI_AGENTS_DISABLE_TRACING`, `RunConfig(tracing_disabled=True)`), or `tracing.py` ran after the first `Runner.run`. - **One session per message.** The run isn't inside `using_session(...)`, or the id changes per request. `group_id` and `SQLiteSession` ids don't reach Maple. -- **Streamed turns have zero tokens.** The model goes through a non-OpenAI base URL without `ModelSettings(include_usage=True)`. -- **Tool arguments show the tool's JSON schema.** `MapleSpanFixes` isn't registered, or it runs after the OpenInference processor. Call `set_trace_processors([MapleSpanFixes()])` before `instrument(..., exclusive_processor=False)`. -- **A streamed turn has tokens but no reply in the transcript.** Same cause: `MapleSpanFixes` isn't first. The streamed Chat Completions reply is otherwise only in `output.value` as a Responses object. -- **More LLM calls than the app made.** `workflow_name` contains `chat` or `completion`, so Maple counts each run's bookkeeping span as a model call. Rename it, for example to `support workflow`. -- **`UserError: Unknown prefix: anthropic`, or OpenRouter gets `gpt-4o-mini` without its prefix.** `Agent(model="vendor/model")` strings go through the SDK's provider prefixes. Pass `OpenAIChatCompletionsModel(model=..., openai_client=...)` instead. -- **`Tracing client error 401` in the logs.** The SDK's default processor is still uploading to OpenAI without a valid key. `set_trace_processors` in `tracing.py` removes it; make sure nothing calls `add_trace_processor` or re-adds `default_processor()` without a key. -- **Every model call appears twice.** `openinference-instrumentation-openai`, Logfire or another Agents instrumentation is also active. Keep one. -- **A multi-agent run is split over several turns.** Parallel `Runner.run` calls outside an SDK `trace()` block each start a trace. Wrap them in one `with trace("...")`. -- **Tool errors show as redacted.** `OPENAI_AGENTS_TRACE_INCLUDE_SENSITIVE_DATA=false` also redacts error details. That's the privacy switch working. +- **Tokens in the list, empty session page.** Pass `TraceConfig(enable_genai_semconv=True)` and use bridge 2.5 or later. +- **Streamed turns have zero tokens.** Add `ModelSettings(include_usage=True)` to agents on a non-OpenAI base URL. +- **Every model call appears twice.** Another instrumentation (`openinference-instrumentation-openai`, Logfire, Langfuse) is also active. Keep one. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/openrouter.md b/apps/landing/src/content/docs/agent-tracing/openrouter.md index 33dd9cf396..8e67a2bbd5 100644 --- a/apps/landing/src/content/docs/agent-tracing/openrouter.md +++ b/apps/landing/src/content/docs/agent-tracing/openrouter.md @@ -1,17 +1,15 @@ --- title: "Trace OpenRouter calls in Maple with Broadcast" -description: "Send every OpenRouter model call to Maple as an OpenTelemetry trace with tokens and real cost, grouped into one Agent Session per conversation, and join it to your app's own traces." +description: "Send every OpenRouter model call to Maple with OpenRouter Broadcast, grouped into one Agent Session per conversation." group: "AI Agents" order: 41 navLabel: "OpenRouter" icon: "openrouter" --- -OpenRouter Broadcast exports a trace for every request that goes through your OpenRouter account. You configure it once in the OpenRouter dashboard, with no SDK and no code in your app. Each trace carries the model, the provider that served it, input, output, cached and reasoning tokens, the cost OpenRouter charged you, and the prompt and completion. Maple reads all of it and shows these traces under **Agent Sessions** as vendor **OpenRouter**. +OpenRouter Broadcast exports a trace for every request made with your OpenRouter account, with the model, tokens, the cost OpenRouter charged, and the prompt and completion. You set it up once in the OpenRouter dashboard. The one thing to change in your code is the `session_id` field: without it, every model call is its own session. -Out of the box, though, every model call is its own trace and its own session. Broadcast only knows what is in the request body. Unless your code sends a `session_id`, a ten-turn conversation shows up as thirty one-call "sessions". - -This guide covers the dashboard setup, the two request fields that fix the grouping (`session_id` and `trace`), and what Broadcast can't see: your tools and your agent structure. Code samples use the `openai` SDK (npm 7.23, PyPI 3.20), `@openrouter/ai-sdk-provider` 3.1 for the Vercel AI SDK, and `@openrouter/sdk` 1.3. +Code samples use the `openai` SDK (npm 7.23, PyPI 3.20), `@openrouter/ai-sdk-provider` 3.1 and `@openrouter/sdk` 1.3. ## Quick setup with a coding agent @@ -25,48 +23,32 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. - -The agent can change your code, but not your OpenRouter dashboard. It finishes by telling you the exact values to paste into the Broadcast destination below. +Your ingest key is in **Settings → Ingestion**. The agent can't change your OpenRouter dashboard, so it ends by printing the values to enter in the next section. ## Point Broadcast at Maple -1. In OpenRouter, open [Settings → Observability](https://openrouter.ai/settings/observability) and turn on **Enable Broadcast**. In an organization account, only an organization admin can edit this. -2. Click the edit icon next to **OpenTelemetry Collector**. -3. Set **Endpoint** to the full traces URL. OpenRouter posts to it verbatim and does not append `/v1/traces`: +1. In OpenRouter, open [Settings → Observability](https://openrouter.ai/settings/observability) and turn on **Enable Broadcast**. In an organization account, only an admin can change this. +2. Click the edit icon next to **OpenTelemetry Collector** and set **Endpoint** to the full traces URL. OpenRouter doesn't append `/v1/traces`. EU organizations use `https://ingest.eu.maple.dev/v1/traces`. ```text https://ingest.maple.dev/v1/traces ``` - EU organizations use `https://ingest.eu.maple.dev/v1/traces`. - -4. Set **Headers** to a JSON object with your Maple ingest key: +3. Set **Headers** to your Maple ingest key: ```json { "Authorization": "Bearer YOUR_INGEST_KEY" } ``` -5. Leave the sampling rate at 1.0 and Privacy Mode off for now (see [privacy](#prompts-completions-and-privacy-mode) below). -6. Click **Test Connection**. OpenRouter only saves the destination if the test passes. - -OpenRouter sends OTLP over HTTP with JSON encoding only. Maple's ingest accepts JSON on `/v1/traces`, so no collector is needed in between. - -Three destination settings silently decide what reaches Maple: - -- **API key filter.** If the destination lists API keys, only requests made with those keys are exported. Excluded keys always win. Leave it empty to export every key. -- **Sampling rate.** Sampling is per `session_id`, so a session is either complete or absent. Anything below 1.0 drops whole conversations. -- **Data regions.** A destination only receives requests served in its regions. If your app calls `eu.openrouter.ai`, the destination needs the Europe region (the default is global). +4. Click **Test Connection**. OpenRouter only saves the destination if the test passes. -Destinations can also be created through OpenRouter's [observability API](https://openrouter.ai/docs/api/api-reference/observability/create-an-observability-destination) (`type: "otel-collector"`) with a management key, if you manage OpenRouter as code. +Leave the sampling rate at 1.0. Sampling is per session, so a lower rate drops whole conversations. Leave the API key filter empty to export every key, and if your app calls `eu.openrouter.ai`, add the Europe data region. -## Group each conversation into one session +## Send a session id with every request -Maple builds a session from the `session.id` attribute on OpenRouter's spans. OpenRouter copies it from the `session_id` field of your request (up to 256 characters) or from the `x-session-id` header. The body wins if you send both. +Maple groups calls into a session by the `session_id` field of your request body (or the `x-session-id` header). Send the same id on every request of a conversation, such as your chat thread id, and a new one for each conversation. -Send the same id on every request of a conversation: your chat thread id, conversation id or agent run id. Use a new one for each conversation. A process-wide constant merges every user into one session. - -With the `openai` SDK in TypeScript, `session_id` isn't in the types, so it needs a `@ts-expect-error`. The SDK sends unknown fields as-is: +With the `openai` SDK in TypeScript, the field isn't in the types, so it needs a `@ts-expect-error`: ```ts import OpenAI from "openai" @@ -84,15 +66,6 @@ const completion = await client.chat.completions.create({ }) ``` -Or skip the type workaround and send the header, which the SDK types allow per request: - -```ts -const completion = await client.chat.completions.create( - { model: "openai/gpt-4o-mini", messages }, - { headers: { "x-session-id": conversationId } }, -) -``` - In Python, use `extra_body`: ```py @@ -109,7 +82,7 @@ completion = client.chat.completions.create( ) ``` -With the Vercel AI SDK and `@openrouter/ai-sdk-provider`, everything under `providerOptions.openrouter` is merged into the request body: +With the Vercel AI SDK, pass it under `providerOptions.openrouter`: ```ts import { createOpenRouter } from "@openrouter/ai-sdk-provider" @@ -124,22 +97,13 @@ const result = streamText({ }) ``` -OpenRouter's own SDKs have typed fields: `sessionId` in `@openrouter/sdk` (`openRouter.chat.send({ model, messages, sessionId })`) and `session_id=` in the `openrouter` Python package. +OpenRouter's own SDKs have a typed field: `sessionId` in `@openrouter/sdk` and `session_id=` in the `openrouter` Python package. -`session_id` also makes OpenRouter route a session's requests to the same provider, so prompt caches hit more often. +## Nest Broadcast under your own traces -If you skip it, each call lands in its own session, named `trace:` in Maple, with one turn and one model call. +Skip this if OpenRouter is your only source of traces. If your app already sends its own traces to Maple, every model call is otherwise recorded twice. Pass the active span's ids in the `trace` field, and OpenRouter places its spans inside your trace. -## Join Broadcast to your own traces - -By default, each OpenRouter request becomes a separate trace with an `LLM Generation` root span. That has two consequences in Maple: - -- **Turns.** Maple counts one turn per trace, so a user message that makes three model calls (one per tool round trip) shows as three turns. -- **Duplicates.** If your app already sends its own traces to Maple (the Vercel AI SDK, OpenAI Agents SDK, LangChain and so on), every model call is recorded twice: once by your app and once by Broadcast. - -The `trace` request field fixes both. OpenRouter uses `trace.trace_id` and `trace.parent_span_id` verbatim as the OTLP trace id and parent span id, so the Broadcast spans land inside your trace. Use W3C ids: 32 lowercase hex characters for the trace id and 16 for the span id. - -In TypeScript, the least invasive place is a `fetch` wrapper. It reads the span that is active when the SDK sends the HTTP request, which is your framework's model-call span if it has one: +In TypeScript, wrap `fetch` and pass it to the client (`new OpenAI({ baseURL, apiKey, fetch: openRouterFetch })` or `createOpenRouter({ apiKey, fetch: openRouterFetch })`): ```ts import { trace } from "@opentelemetry/api" @@ -156,9 +120,7 @@ export const openRouterFetch: typeof fetch = (input, init) => { } ``` -Pass it to the client: `new OpenAI({ baseURL, apiKey, fetch: openRouterFetch })` or `createOpenRouter({ apiKey, fetch: openRouterFetch })`. Both SDKs send the body as a JSON string, which is what the wrapper expects. - -In Python, add the current span's ids next to `session_id`: +In Python, build the body next to `session_id`: ```py from opentelemetry import trace @@ -178,100 +140,23 @@ def openrouter_extra_body(conversation_id: str) -> dict: client.chat.completions.create(model=model, messages=messages, extra_body=openrouter_extra_body(conversation_id)) ``` -This parents the Broadcast spans to whatever span is current at the call site, usually your turn or agent span. - -Once the spans share a trace, Maple counts each model call once: - -- If the Broadcast `LLM Generation` span is a descendant of your app's model-call span, Maple counts the usage at the deepest span that reports it. -- If they are siblings, or in different traces, Maple matches them by `gen_ai.response.id`. OpenRouter's is the `gen-…` id in the response. Instrumentations that record the response id (the Vercel AI SDK, OpenTelemetry's `openai-v2` instrumentation) match; ones that don't are counted twice. - -Use the same value for `session_id` as for your framework's conversation id. A trace that carries two different session ids is assigned to one of them, the lexically larger, without a warning. - -## Prompts, completions and Privacy Mode - -Broadcast includes content by default. The `LLM Generation` span carries the request messages in `gen_ai.prompt` and the reply in `gen_ai.completion`, both as JSON strings. - -They are wrapped objects, not message arrays: `{"messages": [...]}` for the prompt and `{"completion": "...", "reasoning": "..."}` for the reply. Maple shows each one as a single raw JSON block on the span, so the transcript is readable but not rendered as chat bubbles, and turns have no label from the user's message. If your app also emits `gen_ai.input.messages` on its own spans, that transcript renders normally. +Use the same value for `session_id` as your framework's conversation id. -In our captures, the completion object also echoes the request body, including your tool definitions and the `user`, `session_id` and `trace` fields. - -To keep content out of Maple, turn on **Privacy Mode** on the destination. OpenRouter strips prompts and completions; tokens, cost, timing, model and metadata still arrive, so sessions, counts and cost still work. Privacy Mode does not remove `user`, `session_id` or custom `trace` metadata, so don't put emails or names in them. - -OpenRouter shortens any single value over 10,000,000 characters and adds a `.truncated` attribute. Maple's ingest accepts request bodies up to 20 MiB. - -## Tools, errors and provider fallbacks - -Broadcast sees HTTP requests to OpenRouter, not your agent. Your tools run in your process, so a Broadcast-only setup has **no tool spans**: tool calls appear inside the completion JSON, but Maple's tool counts, tool pages and tool failure groups stay empty, and a tool that throws is invisible. There is no agent name either, so sub-agents don't get lanes. - -For tools and agent structure, instrument the app with its framework guide from [Agent tracing](/docs/agent-tracing) and nest Broadcast under it as shown above. Broadcast then adds what most frameworks lack: cost and the provider routing. - -Model failures do show up. Each request's trace has an `LLM Generation` root and a `provider attempt N: ` child per upstream provider OpenRouter tried: - -- A failed attempt that OpenRouter recovered from by falling back to another provider is a retry. Maple does not count it as a failure. -- If every attempt fails, `LLM Generation` has status Error with the message `Provider returned error`, and Maple counts it as a `provider_error` in the session. Such a call carries no token usage, and Maple then counts the root and each failed attempt as separate model calls, so one failed request with one attempt shows as two LLM calls. - -On that error path, OpenRouter currently drops the `trace` object, so the failed call lands in its own trace. It keeps `session.id`, so it still joins the right session. - -## Tokens and cost - -Every `LLM Generation` span carries: - -| Attribute | What Maple does with it | -| --- | --- | -| `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` | Input and output tokens | -| `gen_ai.usage.input_tokens.cached` | Cache-read tokens (included in input) | -| `gen_ai.usage.output_tokens.reasoning` | Reasoning tokens (included in output) | -| `gen_ai.usage.total_cost` | Cost in USD, OpenRouter's actual charge | -| `gen_ai.request.model`, `gen_ai.response.model` | Model, as the OpenRouter slug (`openai/gpt-4o-mini`) | -| `gen_ai.response.finish_reasons` | Truncation and refusal checks | - -Cost is the main reason to use Broadcast even when your app is already instrumented. Maple never prices tokens itself, and most frameworks export no cost, so their sessions read "unpriced". When the app's span and the Broadcast span describe the same call, Maple keeps the larger cost of the two, which is OpenRouter's. - -Streaming doesn't matter here: OpenRouter accounts usage server-side, so streamed calls carry tokens and cost without `stream_options.include_usage`. - -Not read by Maple: - -- `gen_ai.usage.input_tokens.cache_write` (cache writes). Totals are unaffected, but the cache-write column stays empty. -- `trace.metadata.openrouter.first_token_ms` (time to first token). Maple reads TTFT only from `gen_ai.response.time_to_first_chunk`. -- The **Cost** option under **Additional generation metadata**, which adds `span.metadata.openrouter_generation.*` attributes. You don't need it for Maple. - -`gen_ai.provider.name` is the model's author (`openai`, `anthropic`), not the provider that served the request. That one is in `trace.metadata.openrouter.provider_name` (for example `Amazon Bedrock`). - -One known gap for Claude models: OpenRouter's input count includes cached tokens for every model, but Maple reads an `anthropic` input count as excluding them, so cache reads are counted twice in Claude sessions' token totals. In our test, a Claude session with 26,730 input tokens (4,248 of them cached) showed 32,573 total tokens instead of 28,325. Cost is unaffected, because it comes from OpenRouter's charge. - -## Short-lived processes - -Broadcast needs no flush. OpenRouter sends traces from its own servers after each request completes, so a script, a serverless function or a CLI that exits right after the response loses nothing. - -Expect about a minute between the response and the trace in Maple. - -If your app also exports its own spans, those still need the usual flush on exit (`sdk.shutdown()` in Node, `provider.force_flush()` in Python). Otherwise Maple shows the Broadcast spans with a missing parent. +Broadcast only sees requests to OpenRouter, so it has no tool calls or agent names. For those, instrument your app with its [framework guide](/docs/agent-tracing) and nest Broadcast under it as shown here. ## Check that it works -Run one conversation of three or more turns, with a tool call, sending the same `session_id` on every request. Wait about a minute, then open **Agent Sessions** in Maple and filter by service `openrouter`. - -- **One session for the conversation**, named by your `session_id`, with vendor **OpenRouter**. A second conversation is a second session. -- **Model calls.** Each call has an `LLM Generation` span with one or more `provider attempt N: ` children, and sometimes `generation` or `moderation` children. Only `LLM Generation` counts as a model call, except on a request where every provider attempt failed (see above). -- **Tokens and cost** on the session and per model, with cost in USD. -- **Transcript**: each call's prompt and completion as a JSON block, unless Privacy Mode is on. -- **Turns**: one per trace. Broadcast-only, that's one per model call, listed as unlabeled segments (Segment 1, Segment 2, ...). Nested under your own traces, it's one per turn of your agent. -- **Tools**: none from Broadcast. Tool spans come from your app's instrumentation. +Broadcast sends traces from OpenRouter's servers, so your app needs no flush, but expect about a minute of delay. Run a conversation of three turns with the same `session_id`, then open **Agent Sessions** and filter by service `openrouter`. -The service name on Broadcast spans is always `openrouter`, and there is no environment attribute. Custom keys in the `trace` object arrive as `trace.metadata.` attributes, which you can search in [Traces](/docs/explore/traces) but which don't set Maple's environment or service. +You should see one session named after your `session_id`, with vendor **OpenRouter**, an `LLM Generation` span per model call, and tokens and cost in USD. The transcript shows each call's prompt and completion as a raw JSON block. To keep content out of Maple, turn on **Privacy Mode** on the destination. Tokens and cost still arrive. ## Troubleshooting -- **Test Connection fails.** The endpoint must be the full `https://ingest.maple.dev/v1/traces` URL and the headers valid JSON with `"Authorization": "Bearer YOUR_INGEST_KEY"`. EU organizations use `ingest.eu.maple.dev`. -- **Test Connection passes but nothing arrives.** Check the destination's API key filter and data regions against the key and endpoint your app actually uses, and that **Enable Broadcast** is on for the account or organization your app's key belongs to. A placeholder key such as `MAPLE_TEST` passes the test, but Maple discards everything it sends. -- **Every call is its own session, named `trace:`.** The request has no `session_id`. Check the outgoing body, not just your code: a wrapper or framework may drop unknown fields. -- **A session named `trace:00000000000000000000000000000001` with no model calls.** That's the `openrouter-connection-test` span from **Test Connection**. Every test reuses that trace id, so the session gains a span per click. Ignore it. -- **A session with an `openai/gpt-4-turbo` call you never made, trace name `Test Trace - OpenRouter Observability`.** OpenRouter's sample trace from the destination settings, with sample tokens and cost. -- **Tokens or LLM calls are about double what you expect.** Your app's instrumentation and Broadcast both report the call, in different traces and without a shared response id. Nest Broadcast with `trace.trace_id` and `trace.parent_span_id`, or filter that service's API key out of the destination. -- **Broadcast spans show up as a separate trace despite `trace.trace_id`.** The id isn't W3C hex, or the call failed: on the all-providers-failed path OpenRouter drops the `trace` object. -- **Turns have no label and the transcript is raw JSON.** Expected for Broadcast content (see [above](#prompts-completions-and-privacy-mode)). -- **Sessions are missing entirely at random.** The destination's sampling rate is below 1.0. -- **Tool counts are zero.** Expected for Broadcast-only. Add your framework's instrumentation for tool spans. +- **Test Connection fails.** Use the full `https://ingest.maple.dev/v1/traces` URL and valid JSON headers. +- **Test Connection passes but nothing arrives.** Check the destination's API key filter and data regions against the key and endpoint your app uses, and that you're not using a placeholder key like `MAPLE_TEST`. +- **Every call is its own session, named `trace:`.** The request has no `session_id`. Log the outgoing body, since some wrappers drop unknown fields. +- **A session named `trace:00000000000000000000000000000001` with no model calls.** That's the **Test Connection** trace. Ignore it. +- **Tokens or LLM calls are about double.** Your app and Broadcast both record each call. Nest Broadcast with the `trace` field as shown above. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md index 771f7f3bf0..8eb70beebf 100644 --- a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md +++ b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md @@ -1,17 +1,15 @@ --- title: "Trace any AI agent with the OpenTelemetry GenAI conventions" -description: "Write the three OpenTelemetry spans Maple needs by hand, in any language, so a hand-rolled agent loop shows up in Agent Sessions with its transcript, tool calls, tokens and cost." +description: "Write the OpenTelemetry GenAI spans Maple reads by hand, in any language, so a custom agent loop shows up in Agent Sessions." group: "AI Agents" order: 51 navLabel: "Any language (OTel GenAI)" icon: "opentelemetry" --- -If your agent is a loop you wrote yourself, runs on a framework without OpenTelemetry support, or lives in Go, Rust, Ruby or Elixir, nothing emits agent spans for you. You write them. Maple reads the OpenTelemetry GenAI semantic conventions, so three kinds of span are enough: one `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. +If your agent loop is hand-written, runs on a framework without OpenTelemetry support, or lives in Go, Rust, Ruby or Elixir, you write the agent spans yourself. Maple needs three kinds: an `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. The details that break most setups are the conversation id and sending messages as JSON strings. -What goes wrong is the detail. The GenAI conventions are still in Development status and have renamed attributes several times, and a span that looks right can still arrive with an empty transcript: messages sent as plain text, content put in span events, a JSON string cut off by an attribute length limit, or a fresh conversation id on every request. - -This guide lists the exact keys and value formats Maple reads, with full examples in TypeScript and Python and a shorter one in Go. It follows the conventions in [`semantic-conventions-genai`](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. The TypeScript and Python examples were run against OpenRouter with the OpenTelemetry JS SDK 2.11 (OTLP exporter 0.222) on Node.js 26 and the Python SDK 1.45 on Python 3.14, using the OpenAI SDK 7.23 and 3.20. The Go example targets the Go SDK 1.46 and was not compiled for this guide. +Tested with the OpenTelemetry JS SDK 2.11 on Node.js 26 and the Python SDK 1.45 on Python 3.14, calling OpenRouter, against the [GenAI conventions](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. If you use a framework, check the [framework guides](/docs/agent-tracing) first. Most of them emit these spans for you. @@ -27,9 +25,9 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. -## The three spans Maple needs +## The spans and attributes Maple reads One user message produces one trace: @@ -40,45 +38,33 @@ invoke_agent support gen_ai.conversation.id = chat_42 └── chat openai/gpt-4o-mini model call: final answer ``` -Maple classifies a span by `gen_ai.operation.name`, not by its name. A span without that attribute is ignored by Agent Sessions, even if it carries a model or token counts. The span names above follow the spec (`invoke_agent {agent}`, `chat {model}`, `execute_tool {tool}`) and are what you'll see in the trace view. - -**`invoke_agent`** (span kind `INTERNAL`), one per agent run: +Maple classifies a span by `gen_ai.operation.name`, not by its name. Agent Sessions ignores a span without that attribute, even if it carries a model or token counts. -| Attribute | Value | What Maple does with it | +| Span (kind) | Attribute | Value | | --- | --- | --- | -| `gen_ai.operation.name` | `invoke_agent` | Marks the span as an agent run. A root agent span starts a turn. | -| `gen_ai.agent.name` | `support` | Agent facet, and a separate lane for each sub-agent. | -| `gen_ai.conversation.id` | your chat or thread id | Groups the trace into a session. | -| `gen_ai.input.messages`, `gen_ai.output.messages` | JSON string, see below | Optional. The turn's prompt and answer, and a sub-agent lane's input and output. | - -**`chat`** (span kind `CLIENT`), one per model call. `generate_content` and `text_completion` work too. - -| Attribute | Value | What Maple does with it | -| --- | --- | --- | -| `gen_ai.operation.name` | `chat` | Counts the span as an LLM call. | -| `gen_ai.provider.name` | `openai`, `anthropic`, `gcp.gemini`, `openrouter`... | Decides how token counts are read (see [Tokens and cost](#tokens-and-cost)). | -| `gen_ai.request.model`, `gen_ai.response.model` | model ids | Model facet. The response model wins when both are set. | -| `gen_ai.response.id` | the provider's response id | Counts two spans for the same response as one call. | -| `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` | int | Token totals. | -| `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` | int | Cache and reasoning breakdown. | -| `gen_ai.usage.cost` | double, USD | Session cost. Not part of the OpenTelemetry spec. | -| `gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages` | JSON string | The transcript. | -| `gen_ai.response.finish_reasons` | string array, e.g. `["stop"]` | Refusal (`content_filter`) and truncation (`length`) checks. | -| `gen_ai.response.time_to_first_chunk` | double, **seconds** | Time to first token on streamed calls. | - -**`execute_tool`** (span kind `INTERNAL`), one per tool call: - -| Attribute | Value | What Maple does with it | -| --- | --- | --- | -| `gen_ai.operation.name` | `execute_tool` | Counts the span as a tool call. | -| `gen_ai.tool.name` | `get_weather` | Tool pages and facet. | -| `gen_ai.tool.call.id` | the model's tool call id | Matches the call to the model message that requested it. | -| `gen_ai.tool.call.arguments` | JSON string of an object | Arguments in the transcript. | -| `gen_ai.tool.call.result` | JSON string of an object or array | Result in the transcript. A bare string is dropped. | - -A failed span of any kind gets status `ERROR` with the error message and an `error.type` attribute. Maple counts a span as failed when either is present. - -Older spellings still work, so spans from an emitter written against an earlier version of the conventions don't need a rewrite: `gen_ai.system` (renamed to `gen_ai.provider.name` in 1.37), `gen_ai.usage.prompt_tokens` and `completion_tokens`, and whole-value `gen_ai.prompt` and `gen_ai.completion`. When both old and new keys are set, the new one wins. Use the current names in new code. +| `invoke_agent` (`INTERNAL`) | `gen_ai.operation.name` | `invoke_agent` | +| | `gen_ai.agent.name` | `support`. Each sub-agent gets a lane named after it. | +| | `gen_ai.conversation.id` | your chat or thread id, see [sessions](#group-turns-into-one-session) | +| | `gen_ai.input.messages`, `gen_ai.output.messages` | optional, JSON string | +| `chat` (`CLIENT`) | `gen_ai.operation.name` | `chat` (`generate_content` and `text_completion` also work) | +| | `gen_ai.provider.name` | the API you called: `openai`, `anthropic`, `gcp.gemini`, `openrouter`... | +| | `gen_ai.request.model`, `gen_ai.response.model` | model ids | +| | `gen_ai.response.id` | the provider's response id | +| | `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` | int | +| | `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` | int, optional | +| | `gen_ai.usage.cost` | double in USD, optional (not part of the spec) | +| | `gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages` | JSON string | +| | `gen_ai.response.finish_reasons` | string array, e.g. `["stop"]` | +| | `gen_ai.response.time_to_first_chunk` | double, in **seconds**, streamed calls | +| `execute_tool` (`INTERNAL`) | `gen_ai.operation.name` | `execute_tool` | +| | `gen_ai.tool.name` | `get_weather` | +| | `gen_ai.tool.call.id` | the id the model gave the tool call | +| | `gen_ai.tool.call.arguments` | JSON string of an object | +| | `gen_ai.tool.call.result` | JSON string of an object or array (a bare string is dropped) | + +Mark a failed span of any kind with status `ERROR` and an `error.type` attribute. When a tool fails, you can still return the error to the model as its result; the span status is what Maple counts. + +Put token usage on `chat` spans only, copied from the provider's response as is. Maple reads the input count by `gen_ai.provider.name`: for `anthropic`, send Anthropic's raw `input_tokens`, which exclude cache, or cached tokens are counted twice. On OpenAI's streaming API, set `stream_options: { include_usage: true }`, or streamed calls report no tokens. Maple never prices tokens, so a session without `gen_ai.usage.cost` shows as unpriced. OpenRouter returns the cost in `usage.cost`. ### The message format @@ -95,9 +81,11 @@ Older spellings still work, so spans from an emitter written against an earlier ] ``` -Output messages add a `finish_reason` to each message. `gen_ai.system_instructions` is an array of parts without a role: `[{"type":"text","content":"You are a concise assistant."}]`. Maple also renders `reasoning` parts, and accepts `content` (a string or a part array) in place of `parts`. +Output messages add a `finish_reason` to each message. `gen_ai.system_instructions` is an array of parts without a role: `[{"type":"text","content":"You are a concise assistant."}]`. + +Always set these as a string holding JSON. Maple drops plain text, and it doesn't read messages from span events, logs or indexed keys like `gen_ai.prompt.0.content`. Leave `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` and `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` unset: a long history cut mid-string no longer parses and Maple drops the whole attribute. -Always set these as a **string** holding JSON. Maple drops a plain-text value like `"What's the weather?"`. The spec also allows structured attribute values, but Maple receives those as a list of strings and renders them as raw text. +To keep prompts and results out of Maple, skip the five content attributes. Models, tokens, tool names and errors still show up, with an empty transcript. ## Export spans to Maple @@ -160,133 +148,17 @@ provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) ``` -Import `tracing` first in every entry point: the web server, each worker and each script. If your app already has a `TracerProvider` (from Sentry, Datadog, `opentelemetry-instrument` or `NodeSDK`), don't create a second one. Add the `BatchSpanProcessor` to the existing provider instead. +If your app already has a `TracerProvider` (from Sentry, Datadog, `opentelemetry-instrument` or `NodeSDK`), add the `BatchSpanProcessor` to it instead of creating a second one. -Name the tracer after your app. Maple recognizes some frameworks by their instrumentation scope name, and a tracer named `openrouter` or `langsmith`, for example, makes Maple read the session from that framework's key and ignore `gen_ai.conversation.id`. +Name the tracer after your app, like `support-agent`. A tracer named after a framework or gateway, such as `openrouter` or `langsmith`, makes Maple read that framework's session key and ignore `gen_ai.conversation.id`. -## Instrument the agent loop in TypeScript +## Instrument the agent loop -The examples call an OpenAI-compatible Chat Completions API through OpenRouter. Any provider works; change `baseURL`, the model ids and `PROVIDER`. Each model call streams, so the same code records time to first chunk and works for streamed chat replies. +Complete, tested loops that stream, call tools and record every attribute above are in the skill: [TypeScript](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/typescript.md), [Python](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/python.md) and [Go](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/go.md) (not compiled; run `go vet`). Keep your own loop and wrap its existing calls. -```ts -// agent.ts -import { type Span, SpanKind, SpanStatusCode, trace } from "@opentelemetry/api" -import OpenAI from "openai" -import type { - ChatCompletionAssistantMessageParam, - ChatCompletionMessageFunctionToolCall, - ChatCompletionMessageParam, - ChatCompletionTool, -} from "openai/resources/chat/completions" - -const tracer = trace.getTracer("support-agent") -const client = new OpenAI({ baseURL: "https://openrouter.ai/api/v1", apiKey: process.env.OPENROUTER_API_KEY }) -const PROVIDER = "openrouter" // gen_ai.provider.name: who you send the request to - -type Message = ChatCompletionMessageParam -type ToolCall = ChatCompletionMessageFunctionToolCall -type Tool = { definition: ChatCompletionTool; run: (args: Record) => unknown } -export type Agent = { name: string; model: string; instructions: string; tools: Record } - -const json = (value: unknown) => JSON.stringify(value) - -// OpenAI message -> GenAI semconv message: { role, parts: [...] } -function toSemconv(message: Message) { - if (message.role === "tool") { - return { role: "tool", parts: [{ type: "tool_call_response", id: message.tool_call_id, response: message.content }] } - } - const parts: object[] = typeof message.content === "string" && message.content ? [{ type: "text", content: message.content }] : [] - if (message.role === "assistant") { - for (const call of message.tool_calls ?? []) { - if (call.type !== "function") continue - parts.push({ type: "tool_call", id: call.id, name: call.function.name, arguments: JSON.parse(call.function.arguments || "{}") }) - } - } - return { role: message.role, parts } -} - -function markFailed(span: Span, error: unknown) { - const err = error instanceof Error ? error : new Error(String(error)) - span.setStatus({ code: SpanStatusCode.ERROR, message: err.message }) - span.setAttribute("error.type", err.name) -} - -// One model call = one `chat` span. Streams, so time to first chunk is recorded too. -async function chat(agent: Agent, messages: Message[], onText?: (delta: string) => void) { - return tracer.startActiveSpan( - `chat ${agent.model}`, - { - kind: SpanKind.CLIENT, - attributes: { - "gen_ai.operation.name": "chat", - "gen_ai.provider.name": PROVIDER, - "gen_ai.request.model": agent.model, - "gen_ai.system_instructions": json([{ type: "text", content: agent.instructions }]), - "gen_ai.input.messages": json(messages.map(toSemconv)), - }, - }, - async (span) => { - try { - const started = performance.now() - const stream = await client.chat.completions.create({ - model: agent.model, - messages: [{ role: "system", content: agent.instructions }, ...messages], - tools: Object.values(agent.tools).map((tool) => tool.definition), - stream: true, - stream_options: { include_usage: true }, // without it, streamed calls report no tokens - }) - let id = "" - let model = agent.model - let text = "" - let finishReason = "stop" - let usage: (OpenAI.CompletionUsage & { cost?: number }) | undefined - const calls: ToolCall[] = [] - for await (const chunk of stream) { - if (!id) span.setAttribute("gen_ai.response.time_to_first_chunk", (performance.now() - started) / 1000) - id = chunk.id - model = chunk.model - if (chunk.usage) usage = chunk.usage - const choice = chunk.choices[0] - if (!choice) continue - if (choice.finish_reason) finishReason = choice.finish_reason - if (choice.delta.content) { - text += choice.delta.content - onText?.(choice.delta.content) - } - for (const delta of choice.delta.tool_calls ?? []) { - const call = (calls[delta.index] ??= { id: "", type: "function", function: { name: "", arguments: "" } }) - if (delta.id) call.id = delta.id - call.function.name += delta.function?.name ?? "" - call.function.arguments += delta.function?.arguments ?? "" - } - } - const reply: ChatCompletionAssistantMessageParam = { role: "assistant", content: text, ...(calls.length ? { tool_calls: calls } : {}) } - span.setAttributes({ - "gen_ai.response.id": id, - "gen_ai.response.model": model, - "gen_ai.response.finish_reasons": [finishReason], - "gen_ai.output.messages": json([{ ...toSemconv(reply), finish_reason: finishReason }]), - }) - if (usage) { - span.setAttributes({ - "gen_ai.usage.input_tokens": usage.prompt_tokens, - "gen_ai.usage.output_tokens": usage.completion_tokens, - "gen_ai.usage.cache_read.input_tokens": usage.prompt_tokens_details?.cached_tokens ?? 0, - "gen_ai.usage.reasoning.output_tokens": usage.completion_tokens_details?.reasoning_tokens ?? 0, - }) - if (usage.cost !== undefined) span.setAttribute("gen_ai.usage.cost", usage.cost) // OpenRouter returns USD cost - } - return reply - } catch (error) { - markFailed(span, error) - throw error - } finally { - span.end() - } - }, - ) -} +This is the `execute_tool` span from the TypeScript version. The `chat` span follows the same pattern around each model call: +```ts // One tool call = one `execute_tool` span. A failure is marked on the span and returned to the model. async function runTool(agent: Agent, call: ToolCall) { const name = call.function.name @@ -318,78 +190,15 @@ async function runTool(agent: Agent, call: ToolCall) { }, ) } - -// One agent run = one `invoke_agent` span. For a user turn it is the root of the trace. -export async function runAgent( - agent: Agent, - messages: Message[], - options: { conversationId?: string; onText?: (delta: string) => void } = {}, -): Promise { - const input = messages.at(-1) - return tracer.startActiveSpan( - `invoke_agent ${agent.name}`, - { - kind: SpanKind.INTERNAL, - attributes: { - "gen_ai.operation.name": "invoke_agent", - "gen_ai.agent.name": agent.name, - ...(options.conversationId ? { "gen_ai.conversation.id": options.conversationId } : {}), - ...(input ? { "gen_ai.input.messages": json([toSemconv(input)]) } : {}), - }, - }, - async (span) => { - try { - for (let step = 0; step < 10; step++) { - const reply = await chat(agent, messages, options.onText) - messages.push(reply) - if (!reply.tool_calls?.length) { - span.setAttribute("gen_ai.output.messages", json([toSemconv(reply)])) - return typeof reply.content === "string" ? reply.content : "" - } - for (const call of reply.tool_calls) { - if (call.type !== "function") continue - messages.push({ role: "tool", tool_call_id: call.id, content: await runTool(agent, call) }) - } - } - throw new Error("agent exceeded 10 steps") - } catch (error) { - markFailed(span, error) - throw error - } finally { - span.end() - } - }, - ) -} ``` -A chat backend keeps one message list per conversation and calls `runAgent` once per user message: +Start the `chat` and `execute_tool` spans inside the `invoke_agent` span's callback, so they land in the same trace. For a sub-agent, run its loop inside the delegating tool's `execute_tool` span and give it a distinct `gen_ai.agent.name`. -```ts -// main.ts -import { provider } from "./tracing" // first import: sets up the provider -import type { ChatCompletionMessageParam } from "openai/resources/chat/completions" -import { type Agent, runAgent } from "./agent" - -const assistant: Agent = { - name: "support", - model: "openai/gpt-4o-mini", - instructions: "You are a concise assistant.", - tools: { - get_weather: { - definition: { - type: "function", - function: { - name: "get_weather", - description: "Current weather for a city", - parameters: { type: "object", properties: { city: { type: "string" } }, required: ["city"] }, - }, - }, - run: ({ city }) => ({ city, temperature_c: 21, condition: "partly cloudy" }), - }, - }, -} +## Group turns into one session +Set `gen_ai.conversation.id` on the `invoke_agent` span of every turn, using the id your app already has for the conversation. Every span in that trace joins the session. A chat backend passes it once per user message: + +```ts // One history per conversation. Store it in your database in a real backend. const histories = new Map() @@ -399,350 +208,11 @@ export async function handleMessage(chatId: string, text: string, onText?: (delt history.push({ role: "user", content: text }) return runAgent(assistant, history, { conversationId: chatId, onText }) } - -// In a script: two messages of one conversation, then flush -try { - await handleMessage("chat_42", "Hi! Briefly introduce yourself.") - await handleMessage("chat_42", "What's the weather in Berlin?", (delta) => process.stdout.write(delta)) -} finally { - await provider.shutdown() -} ``` -Run it with `npx tsx main.ts` or your bundler. The files are ES modules (`"type": "module"` in `package.json`) for the top-level `await`, and use extensionless imports, which Node's built-in type stripping doesn't resolve. - -## Instrument the agent loop in Python - -The same loop with the OpenAI Python SDK: - -```py -# agent.py -import json -import os -import time -from dataclasses import dataclass, field -from typing import Any, Callable - -from openai import OpenAI, omit -from opentelemetry import trace -from opentelemetry.trace import SpanKind, Status, StatusCode - -tracer = trace.get_tracer("support-agent") -client = OpenAI(base_url="https://openrouter.ai/api/v1", api_key=os.environ["OPENROUTER_API_KEY"]) -PROVIDER = "openrouter" # gen_ai.provider.name: who you send the request to - - -@dataclass -class Tool: - definition: dict[str, Any] - run: Callable[..., Any] - - -@dataclass -class Agent: - name: str - model: str - instructions: str - tools: dict[str, Tool] = field(default_factory=dict) - - -def to_semconv(message: dict[str, Any]) -> dict[str, Any]: - """OpenAI message -> GenAI semconv message: {role, parts: [...]}.""" - if message["role"] == "tool": - part = {"type": "tool_call_response", "id": message["tool_call_id"], "response": message["content"]} - return {"role": "tool", "parts": [part]} - parts: list[dict[str, Any]] = [] - if message.get("content"): - parts.append({"type": "text", "content": message["content"]}) - for call in message.get("tool_calls") or []: - fn = call["function"] - parts.append({"type": "tool_call", "id": call["id"], "name": fn["name"], "arguments": json.loads(fn["arguments"] or "{}")}) - return {"role": message["role"], "parts": parts} - - -def mark_failed(span: trace.Span, error: Exception) -> None: - span.set_status(Status(StatusCode.ERROR, str(error))) - span.set_attribute("error.type", type(error).__qualname__) - - -def chat(agent: Agent, messages: list[dict[str, Any]], on_text: Callable[[str], None] | None = None) -> dict[str, Any]: - """One model call = one `chat` span. Streams, so time to first chunk is recorded too.""" - attributes = { - "gen_ai.operation.name": "chat", - "gen_ai.provider.name": PROVIDER, - "gen_ai.request.model": agent.model, - "gen_ai.system_instructions": json.dumps([{"type": "text", "content": agent.instructions}]), - "gen_ai.input.messages": json.dumps([to_semconv(m) for m in messages]), - } - with tracer.start_as_current_span(f"chat {agent.model}", kind=SpanKind.CLIENT, attributes=attributes) as span: - try: - started = time.perf_counter() - stream = client.chat.completions.create( - model=agent.model, - messages=[{"role": "system", "content": agent.instructions}, *messages], - tools=[tool.definition for tool in agent.tools.values()] or omit, - stream=True, - stream_options={"include_usage": True}, # without it, streamed calls report no tokens - ) - response_id, model, text, finish_reason, usage = "", agent.model, "", "stop", None - calls: dict[int, dict[str, Any]] = {} - for chunk in stream: - if not response_id: - span.set_attribute("gen_ai.response.time_to_first_chunk", time.perf_counter() - started) - response_id, model = chunk.id, chunk.model - usage = chunk.usage or usage - if not chunk.choices: - continue - choice = chunk.choices[0] - finish_reason = choice.finish_reason or finish_reason - if choice.delta.content: - text += choice.delta.content - if on_text: - on_text(choice.delta.content) - for delta in choice.delta.tool_calls or []: - call = calls.setdefault(delta.index, {"id": "", "type": "function", "function": {"name": "", "arguments": ""}}) - call["id"] = delta.id or call["id"] - if delta.function: - call["function"]["name"] += delta.function.name or "" - call["function"]["arguments"] += delta.function.arguments or "" - reply: dict[str, Any] = {"role": "assistant", "content": text} - if calls: - reply["tool_calls"] = [calls[i] for i in sorted(calls)] - span.set_attributes({ - "gen_ai.response.id": response_id, - "gen_ai.response.model": model, - "gen_ai.response.finish_reasons": [finish_reason], - "gen_ai.output.messages": json.dumps([{**to_semconv(reply), "finish_reason": finish_reason}]), - }) - if usage: - span.set_attributes({ - "gen_ai.usage.input_tokens": usage.prompt_tokens, - "gen_ai.usage.output_tokens": usage.completion_tokens, - "gen_ai.usage.cache_read.input_tokens": getattr(usage.prompt_tokens_details, "cached_tokens", None) or 0, - "gen_ai.usage.reasoning.output_tokens": getattr(usage.completion_tokens_details, "reasoning_tokens", None) or 0, - }) - cost = getattr(usage, "cost", None) # OpenRouter returns USD cost - if cost is not None: - span.set_attribute("gen_ai.usage.cost", cost) - return reply - except Exception as error: - mark_failed(span, error) - raise - - -def run_tool(agent: Agent, call: dict[str, Any]) -> str: - """One tool call = one `execute_tool` span. A failure is marked on the span and returned to the model.""" - name, arguments = call["function"]["name"], call["function"]["arguments"] or "{}" - attributes = { - "gen_ai.operation.name": "execute_tool", - "gen_ai.tool.name": name, - "gen_ai.tool.type": "function", - "gen_ai.tool.call.id": call["id"], - "gen_ai.tool.call.arguments": arguments, - } - with tracer.start_as_current_span(f"execute_tool {name}", kind=SpanKind.INTERNAL, attributes=attributes) as span: - try: - result = agent.tools[name].run(**json.loads(arguments)) - # Maple reads a JSON object or array here; a bare string is dropped - output = json.dumps(result if isinstance(result, (dict, list)) else {"result": result}) - span.set_attribute("gen_ai.tool.call.result", output) - return output - except Exception as error: - mark_failed(span, error) - return json.dumps({"error": str(error)}) - - -def run_agent( - agent: Agent, - messages: list[dict[str, Any]], - conversation_id: str | None = None, - on_text: Callable[[str], None] | None = None, -) -> str: - """One agent run = one `invoke_agent` span. For a user turn it is the root of the trace.""" - attributes = {"gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agent.name} - if conversation_id: - attributes["gen_ai.conversation.id"] = conversation_id - if messages: - attributes["gen_ai.input.messages"] = json.dumps([to_semconv(messages[-1])]) - with tracer.start_as_current_span(f"invoke_agent {agent.name}", kind=SpanKind.INTERNAL, attributes=attributes) as span: - try: - for _ in range(10): - reply = chat(agent, messages, on_text) - messages.append(reply) - if not reply.get("tool_calls"): - span.set_attribute("gen_ai.output.messages", json.dumps([to_semconv(reply)])) - return reply["content"] - for call in reply["tool_calls"]: - messages.append({"role": "tool", "tool_call_id": call["id"], "content": run_tool(agent, call)}) - raise RuntimeError("agent exceeded 10 steps") - except Exception as error: - mark_failed(span, error) - raise -``` - -```py -# main.py -from tracing import provider # first import: sets up the provider - -from agent import Agent, Tool, run_agent - -get_weather = Tool( - definition={ - "type": "function", - "function": { - "name": "get_weather", - "description": "Current weather for a city", - "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}, - }, - }, - run=lambda city: {"city": city, "temperature_c": 21, "condition": "partly cloudy"}, -) -assistant = Agent("support", "openai/gpt-4o-mini", "You are a concise assistant.", {"get_weather": get_weather}) - -# One history per conversation. Store it in your database in a real backend. -histories: dict[str, list[dict]] = {} +Without the id, each trace becomes its own one-turn session named `trace:`. A UUID generated per request, or the trace id, has the same effect. Don't give sub-agents their own id: they inherit the session from the trace, and with two ids in one trace Maple keeps only one. - -def handle_message(chat_id: str, text: str, on_text=None) -> str: - history = histories.setdefault(chat_id, []) - history.append({"role": "user", "content": text}) - return run_agent(assistant, history, conversation_id=chat_id, on_text=on_text) - - -if __name__ == "__main__": - # In a script: two messages of one conversation, then flush - try: - handle_message("chat_42", "Hi! Briefly introduce yourself.") - handle_message("chat_42", "What's the weather in Berlin?", on_text=lambda d: print(d, end="", flush=True)) - finally: - provider.shutdown() -``` - -`start_as_current_span` keeps the span in a context variable, so the `chat` and `execute_tool` spans nest under `invoke_agent` in the same trace. That holds across `await` in asyncio code. It doesn't hold in a new thread: code submitted to a `ThreadPoolExecutor` starts with an empty context and becomes a separate trace. Pass the context along with `contextvars.copy_context().run(...)` if your tools run in threads. - -## Go and other languages - -Any OpenTelemetry SDK works, as long as it sets the same attributes. In Go, start the turn span and wrap your existing client call: - -```go -// genai.go -package agent - -import ( - "context" - "encoding/json" - "fmt" - - "go.opentelemetry.io/otel" - "go.opentelemetry.io/otel/attribute" - "go.opentelemetry.io/otel/codes" - "go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp" - "go.opentelemetry.io/otel/sdk/resource" - sdktrace "go.opentelemetry.io/otel/sdk/trace" - "go.opentelemetry.io/otel/trace" -) - -var tracer = otel.Tracer("support-agent") - -// SetupTracing reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. -// Call Shutdown on the returned provider before the process exits. -func SetupTracing(ctx context.Context) (*sdktrace.TracerProvider, error) { - exporter, err := otlptracehttp.New(ctx) - if err != nil { - return nil, err - } - tp := sdktrace.NewTracerProvider( - sdktrace.WithBatcher(exporter), - sdktrace.WithResource(resource.NewSchemaless( - attribute.String("service.name", "support-agent"), - attribute.String("deployment.environment.name", "production"), - )), - ) - otel.SetTracerProvider(tp) - return tp, nil -} - -// Message is the GenAI semconv shape: {role, parts}. -type Message struct { - Role string `json:"role"` - Parts []map[string]any `json:"parts"` - FinishReason string `json:"finish_reason,omitempty"` -} - -// ChatResult holds what your provider returned, copied verbatim. -type ChatResult struct { - ID, Model, FinishReason string - Output Message - InputTokens, OutputTokens int64 - CostUSD float64 // 0 when the provider doesn't return a cost -} - -func jsonAttr(key string, value any) attribute.KeyValue { - b, _ := json.Marshal(value) - return attribute.String(key, string(b)) -} - -func fail(span trace.Span, err error) { - span.SetStatus(codes.Error, err.Error()) - span.SetAttributes(attribute.String("error.type", fmt.Sprintf("%T", err))) -} - -// StartTurn opens the invoke_agent span for one user message. End it when the turn is done. -func StartTurn(ctx context.Context, agentName, conversationID string) (context.Context, trace.Span) { - return tracer.Start(ctx, "invoke_agent "+agentName, trace.WithAttributes( - attribute.String("gen_ai.operation.name", "invoke_agent"), - attribute.String("gen_ai.agent.name", agentName), - attribute.String("gen_ai.conversation.id", conversationID), - )) -} - -// Chat wraps one model call (your existing client code goes in call). -func Chat(ctx context.Context, model string, input []Message, call func(context.Context) (ChatResult, error)) (ChatResult, error) { - ctx, span := tracer.Start(ctx, "chat "+model, trace.WithSpanKind(trace.SpanKindClient), trace.WithAttributes( - attribute.String("gen_ai.operation.name", "chat"), - attribute.String("gen_ai.provider.name", "openai"), - attribute.String("gen_ai.request.model", model), - jsonAttr("gen_ai.input.messages", input), - )) - defer span.End() - res, err := call(ctx) - if err != nil { - fail(span, err) - return res, err - } - res.Output.FinishReason = res.FinishReason - span.SetAttributes( - attribute.String("gen_ai.response.id", res.ID), - attribute.String("gen_ai.response.model", res.Model), - attribute.StringSlice("gen_ai.response.finish_reasons", []string{res.FinishReason}), - attribute.Int64("gen_ai.usage.input_tokens", res.InputTokens), - attribute.Int64("gen_ai.usage.output_tokens", res.OutputTokens), - jsonAttr("gen_ai.output.messages", []Message{res.Output}), - ) - if res.CostUSD > 0 { - span.SetAttributes(attribute.Float64("gen_ai.usage.cost", res.CostUSD)) - } - return res, nil -} -``` - -This snippet was not compiled for this guide; run `go vet ./...` after adding it. An `execute_tool` span follows the same pattern with the attributes from the table. Rust (`opentelemetry` crate), Ruby (`opentelemetry-sdk`), Elixir (`opentelemetry_api`), Java and .NET follow the same pattern: set the attributes from the tables above as strings, ints, doubles and string arrays, and serialize every message and tool payload to a JSON string first. - -## Group turns into one session - -Maple files a trace under the session id it finds in `gen_ai.conversation.id`. Setting it on the `invoke_agent` span of each turn is enough: every span of that trace joins the session, including HTTP and database spans. Use the id your app already has for the conversation, like a chat id or thread id, and pass the same value for every message of that conversation. - -Without it, each trace is its own one-turn session, named `trace:`. A chat backend that handles one message per request then shows one session per message. - -A few things break grouping: - -- **A new id per request.** A UUID generated in the request handler, or the trace id, is different for every message. The GenAI spec says the same: when there's no real conversation id, leave the attribute out rather than inventing one. -- **Different ids in one trace.** If a sub-agent sets its own id, the trace carries two, and Maple silently keeps the one that sorts last. Let sub-agents inherit the session from the trace, as the examples do, or give them the same id. -- **An id only on a span without `gen_ai.operation.name`.** Maple only reads the id from spans it classifies as AI spans. - -### When a framework's session key isn't read: `maple_ai.session.id` - -Some frameworks write their session id under a key Maple doesn't read for that framework, for example OpenInference's `session.id` from LangChain or LlamaIndex instrumentation, or the Vercel AI SDK's `runtimeContext`. The framework guides fix those three without a wrapper. For anything they don't cover, add a wrapper span of your own around each turn, carrying Maple's own session key: +If a framework writes its session id under a key Maple doesn't read, wrap each turn in your own span that carries `maple_ai.session.id`, and run the framework inside it: ```ts // chatId comes from your request; frameworkAgent is the framework's agent @@ -765,116 +235,23 @@ await tracer.startActiveSpan( ) ``` -`maple_ai.session.id` works on any span and overrides framework detection for that span, so the session shows the framework as **Maple**. Put it only on your own wrapper span, never on the framework's spans: a span carrying it loses the framework's attribute decoding, and Maple reads its token counts as if they include cache. Use the same value the framework would use, because with two different session ids in one trace, the one that sorts last wins. You don't need it for spans written by hand with `gen_ai.conversation.id`. - -## Record prompts, responses and tool calls - -The transcript comes from five attributes: `gen_ai.system_instructions`, `gen_ai.input.messages` and `gen_ai.output.messages` on `chat` spans, and `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result` on `execute_tool` spans. Each has to be a JSON string on the span itself. - -- **Span events and logs aren't read.** Older versions of the conventions put messages in events like `gen_ai.user.message`, and some SDKs send them as log records. Agent Sessions only reads span attributes. -- **Indexed keys aren't read.** `gen_ai.prompt.0.content` and `llm.input_messages.0.message.content` are separate dialects. Maple doesn't reassemble them. -- **Attribute length limits break the JSON.** The SDKs don't limit attribute length by default. If `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` or `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` is set (some platforms and distributions set one), a long history is cut mid-string, no longer parses, and Maple drops the whole attribute. Leave them unset. To cap size yourself, drop the oldest messages whole and serialize what's left. -- **Batches over 20 MiB are rejected.** Maple's ingest endpoint returns 413 for larger requests. Base64 images in every message's history add up fast; replace them with a placeholder part before serializing. - -The spec treats all of this as opt-in, because prompts and tool results often contain personal data. To keep metadata only, skip those five attributes; models, tokens, timings, tool names and errors still work, and the transcript stays empty. To keep content but remove specific values, redact them in the `toSemconv` / `to_semconv` function before serializing, or with an OpenTelemetry Collector `redaction` or `transform` processor. Keep API keys out of tool arguments and results. - -## Tools, errors and sub-agents - -When a tool throws, the examples mark its span failed and return the error to the model as the tool result, so the agent can recover: +Put `maple_ai.session.id` only on your own wrapper span. On a framework's spans it replaces the framework's attribute decoding. -- status `ERROR`, with the error message as the status description; -- `error.type`, the exception class (`RuntimeError`, `TypeError`) or an error code; -- no `gen_ai.tool.call.result`, because the tool didn't return one. +## Flush before a short-lived process exits -Maple counts a span as failed if it has `ERROR` status, a non-empty `error.type`, or `gen_ai.response.status` set to `failed`. A tool that catches its own error and returns `{"error": "..."}` without marking the span shows as successful. The tool pages group failures by their message, with ids and numbers masked, so the status description should say what failed. - -Set `gen_ai.tool.call.id` to the id the model gave the call. Maple matches each result to its call by id, which keeps parallel calls from the same model reply apart. - -For a sub-agent, run it inside a tool call: - -```ts -// weatherWorker is an Agent like `assistant` above, with get_weather -const orchestrator: Agent = { - name: "orchestrator", - model: "openai/gpt-4o-mini", - instructions: "Delegate weather questions to the weather worker.", - tools: { - ask_weather_worker: { - definition: { - type: "function", - function: { - name: "ask_weather_worker", - description: "Ask the weather worker a question", - parameters: { type: "object", properties: { question: { type: "string" } }, required: ["question"] }, - }, - }, - // No conversationId: the sub-agent's spans are in this trace, so it inherits the session - run: ({ question }) => runAgent(weatherWorker, [{ role: "user", content: String(question) }]), - }, - }, -} -``` - -Maple shows an `execute_tool` span whose only child is an `invoke_agent` span as a delegation, in a lane named after the sub-agent's `gen_ai.agent.name`, with the tool's arguments and result as the lane's input and output. Give every agent a distinct name: two agents called `agent` share one lane, and an agent span without a name gets no lane at all. - -## Tokens and cost - -Put usage on `chat` spans only. Maple sums it per model call and nets a parent's usage against its children, but an agent span that reports a running total for the whole conversation is counted on top. - -Maple reads five buckets: `input_tokens`, `output_tokens`, `cache_read.input_tokens`, `cache_write.input_tokens` and `reasoning.output_tokens`, all under `gen_ai.usage.`. It also accepts `gen_ai.usage.cache_creation.input_tokens`, the spelling before `cache_write`. It doesn't read `total_tokens`, `gen_ai.usage.reasoning_tokens` or `gen_ai.usage.cache_read_input_tokens`. - -Providers disagree on whether input includes cached tokens, so Maple interprets the numbers by `gen_ai.provider.name`. Copy the provider's raw numbers and set the provider to the API you called: - -| `gen_ai.provider.name` | Input tokens | Output tokens | -| --- | --- | --- | -| `anthropic` | Anthropic's `input_tokens`, **excluding** cache | includes thinking | -| `gcp.gemini`, `gcp.vertex_ai` | `promptTokenCount`, including cache | `candidatesTokenCount`, excluding thoughts | -| `openai`, `openrouter`, anything else | `prompt_tokens`, including cache | includes reasoning | - -The Anthropic row differs from the spec, which asks for the inclusive total on every provider. If you send Anthropic's input as `input_tokens + cache_read + cache_write`, Maple counts cached tokens twice. A call to Claude through OpenRouter's OpenAI-compatible API uses `openrouter`, and OpenAI-shaped numbers: in our test, a cached Claude Haiku 4.5 call reported 11,076 `prompt_tokens` with 11,058 of them in `cached_tokens`, so nothing is counted twice. OpenRouter also reports `prompt_tokens_details.cache_write_tokens`; copy it to `gen_ai.usage.cache_write.input_tokens` if you want cache writes shown separately. - -Streaming needs `stream_options: { include_usage: true }` on OpenAI's API, or the stream carries no usage. OpenAI sends it in an extra last chunk with no choices. OpenRouter always sends usage and cost, on the chunk that carries the finish reason. Read `chunk.usage` before skipping chunks without choices, as the examples do, and both work. - -Maple never prices tokens. A session shows cost only when `chat` spans carry `gen_ai.usage.cost` in USD (`gen_ai.usage.total_cost` also works); otherwise it's shown as unpriced. OpenRouter returns the cost in `usage.cost`, which the examples copy. OpenAI and Anthropic don't return a cost, so compute it from your own price table or leave it out. - -Set `gen_ai.response.id`. If the same call is also exported by a gateway, for example [OpenRouter Broadcast](/docs/agent-tracing/openrouter), both spans carry the same response id and Maple counts the call once. - -## Short-lived processes - -`BatchSpanProcessor` exports every few seconds. A script, CLI, Lambda or notebook can exit before the last batch is sent. Flush explicitly: - -- **TypeScript:** `await provider.shutdown()` at the end of a script, or `await provider.forceFlush()` before a serverless handler returns (inside `waitUntil` or `after()` on platforms that have one). -- **Python:** `provider.shutdown()` at the end of a script, or `provider.force_flush()` in a `finally` in a Lambda handler or after each notebook cell. -- **Go:** `defer tp.Shutdown(context.Background())` in `main`. +`BatchSpanProcessor` exports every few seconds, so a script, CLI or Lambda can exit before the last batch is sent. At the end of a script, call `await provider.shutdown()` (TypeScript), `provider.shutdown()` (Python) or `defer tp.Shutdown(context.Background())` (Go). In a serverless handler, call `provider.forceFlush()` (`force_flush()` in Python) before returning. ## Check that it works -Run one conversation with two or more messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: - -- one session per conversation, with your conversation id as the session id, and the framework shown as **Unidentified** (or **Maple** with `maple_ai.session.id`); -- one turn per user message, labeled with the message, and a transcript with the system instructions, prompts, replies and tool calls; -- `invoke_agent `, `chat ` and `execute_tool ` spans, with every `chat` and `execute_tool` span inside its turn's `invoke_agent` span; -- input and output tokens on every model call, including streamed ones; -- the tool that failed marked as failed, with its message, and every other tool marked successful; -- a lane per sub-agent, named after its `gen_ai.agent.name`; -- cost on each call if your spans carry `gen_ai.usage.cost`, otherwise unpriced. +Run one conversation with two messages and a tool call, then open **Agent Sessions** in Maple. After a few seconds you should see one session with your conversation id, one turn per message with its transcript, and `chat` and `execute_tool` spans nested under each turn's `invoke_agent` span. Hand-written spans show the framework as **Unidentified**, which is expected. ## Troubleshooting -- **Nothing shows up in Agent Sessions, but the trace is in Traces.** No span has `gen_ai.operation.name`. Maple only picks up traces with at least one classified span. -- **Every message is its own session.** `gen_ai.conversation.id` is missing, or it changes per request. Pass the conversation's id on the `invoke_agent` span of every turn. -- **A session merged two conversations.** The id is a constant, or comes from a module-level variable shared across users. Take it from the request. -- **The transcript is empty, but tokens are there.** The messages are plain text, are in span events or logs, or were cut by an attribute length limit. Set them as JSON strings on the span and unset `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT`. -- **The transcript shows raw JSON instead of messages.** The value isn't an array of `{role, parts}` messages, or it was set as a structured attribute instead of a string. -- **Streamed replies have no tokens.** Add `stream_options: { include_usage: true }` and read usage from the final chunk. -- **Anthropic calls show too many input tokens.** `input_tokens` includes cache while the provider is `anthropic`. Send Anthropic's raw `input_tokens`. -- **Tool calls are missing from the tool pages.** The span has no `gen_ai.operation.name: execute_tool` or no `gen_ai.tool.name`. -- **A failed tool shows as successful.** The span has neither `ERROR` status nor `error.type`. -- **Model and tool spans are separate traces.** The context was lost: the spans weren't started inside the `invoke_agent` span's callback, the provider wasn't registered with a context manager (use `provider.register()` in Node), or the work ran in a new thread. -- **Every model call appears twice.** A provider auto-instrumentation (OpenAI, Anthropic, OpenLLMetry, OpenInference) is also active. Keep either your `chat` spans or the instrumentation, not both. See [provider SDKs](/docs/agent-tracing/provider-sdks) for the instrumentation route. -- **Tokens are missing although the span has usage.** The key is a spelling Maple doesn't read, like `gen_ai.usage.total_tokens`, `gen_ai.usage.reasoning_tokens` or `gen_ai.usage.cache_read_input_tokens`. Use the keys in the table above. -- **Exports fail with 401.** The header is missing or malformed. Some SDKs need the space encoded: `Authorization=Bearer%20YOUR_INGEST_KEY`. -- **Nothing arrives from a script.** The process exited before the batch was sent. Call `shutdown()` at the end. +- **Nothing shows up in Agent Sessions, but the trace is in Traces.** No span has `gen_ai.operation.name`. Add it to every span. +- **Every message is its own session.** `gen_ai.conversation.id` is missing or changes per request. Pass the conversation's id on each turn's `invoke_agent` span. +- **The transcript is empty, but tokens are there.** The messages are plain text, in span events, or cut by an attribute length limit. Set them as JSON strings and unset `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT`. +- **Model and tool spans are separate traces.** The spans weren't started inside the `invoke_agent` span, the Node provider wasn't registered with `provider.register()`, or the work ran in a new Python thread (pass the context with `contextvars.copy_context().run(...)`). +- **Every model call appears twice.** A provider auto-instrumentation (OpenAI, Anthropic, OpenLLMetry, OpenInference) is also active. Keep your `chat` spans or the instrumentation, not both. See [provider SDKs](/docs/agent-tracing/provider-sdks). ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md index 3616c26754..b712506426 100644 --- a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md +++ b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md @@ -1,17 +1,15 @@ --- title: "Trace agents built on the OpenAI, Anthropic and Gemini SDKs" -description: "Trace your own agent loop on the OpenAI, Anthropic or Google Gen AI SDK so each conversation is one Maple Agent Session with its transcript, tool calls, failures and tokens, in Python or TypeScript." +description: "Trace your own agent loop on the OpenAI, Anthropic or Google Gen AI SDK so each conversation becomes one Maple Agent Session, in Python or TypeScript." group: "AI Agents" order: 50 navLabel: "OpenAI, Anthropic & Gemini SDKs" icon: "openai" --- -If your agent is your own loop around `client.chat.completions.create`, `client.messages.create` or `client.models.generate_content`, an instrumentation library can record each model call: the prompt, the reply, the model and the tokens. It can't know where a user's turn starts, which conversation it belongs to, or that your code ran a tool between two calls. Those spans are yours to add, and it's about 40 lines of code. +If your agent is your own loop around `client.chat.completions.create`, `client.messages.create` or `client.models.generate_content`, an instrumentation library can record each model call. It can't see where a turn starts, which conversation it belongs to, or the tools your code runs, so you add those spans yourself and put the conversation id on the turn span. You also have to turn on content capture, which is off by default. -Without them, every model call is its own trace, and Maple files every trace without a conversation id as its own session. A four-message chat with two tool calls shows up as six one-call sessions, with no tool calls and nothing tying them together. The second surprise is that the instrumentations record no prompts or replies until you turn content capture on. - -This guide covers Python 3.10+ with `openai` 3.x, `anthropic` 1.x and `google-genai` 2.x, and TypeScript on Node.js with `openai` 7.x (the same pattern works for `@anthropic-ai/sdk` and `@google/genai`). We ran the OpenAI and Anthropic paths end to end (Python `openai` 3.20.0 and `anthropic` 1.8.0 with the 1.2b0 instrumentations, TypeScript `openai` 7.23.0). The Gemini path follows the instrumentation's documentation and hasn't been run against a live Gemini model yet. If you use an agent framework on top of these SDKs, such as the OpenAI Agents SDK, LangChain or Pydantic AI, use [that framework's guide](/docs/agent-tracing) instead. +Tested with Python `openai` 3.20.0 and `anthropic` 1.8.0 (instrumentations 1.2b0) and TypeScript `openai` 7.23.0. The Gemini path follows the instrumentation's documentation and hasn't been run against a live model yet. If you use an agent framework on top of these SDKs, use [that framework's guide](/docs/agent-tracing) instead. ## Quick setup with a coding agent @@ -25,28 +23,11 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. - -## Which instrumentation to use - -Maple builds the transcript from GenAI span attributes: `gen_ai.input.messages` and `gen_ai.output.messages` as JSON arrays of `{role, parts}`. It doesn't read span events or OpenTelemetry logs. That rules out several popular options: - -| Language | Use | Why not the alternatives | -| --- | --- | --- | -| Python | The OpenTelemetry GenAI instrumentations: `opentelemetry-instrumentation-genai-openai`, `opentelemetry-instrumentation-genai-anthropic`, `opentelemetry-instrumentation-google-genai` | They write the messages, tokens (including cache and reasoning), response id and finish reasons onto the span in the exact shape Maple reads. | -| TypeScript | A small helper that records the model call itself (below) | `@opentelemetry/instrumentation-openai` 0.20 only patches `openai` below 7.0 and writes messages to log events. There is no OpenTelemetry instrumentation for `@anthropic-ai/sdk` or `@google/genai`. | - -Other packages you'll find: +Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. -- **`opentelemetry-instrumentation-openai-v2`** is deprecated. Its README says it gets security fixes only and points to `opentelemetry-instrumentation-genai-openai`. -- **`pip install opentelemetry-instrumentation-openai`** (without `genai`) installs OpenLLMetry by Traceloop, a different project. The same goes for `opentelemetry-instrumentation-anthropic`. OpenLLMetry records prompts and replies by default. Pick one family and don't install both. -- **OpenInference** (`openinference-instrumentation-openai`) is recognized by Maple as "OpenInference · OpenAI" and its tokens count, but the transcript is the raw request JSON as one message, with no turn labels. The Anthropic and Gemini OpenInference packages show as Unidentified with the same raw transcript. +## Configure the exporter -The GenAI instrumentations aren't a vendor Maple recognizes by name either, so sessions show **Unidentified** in the framework column. Everything else (transcript, tools, tokens, failures) is read in full. - -## Export spans to Maple - -Both languages use the standard OpenTelemetry exporter variables: +Both languages read the standard OpenTelemetry variables: ```bash export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" @@ -55,11 +36,11 @@ export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" export OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT="SPAN_ONLY" ``` -For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. `SPAN_ONLY` is explained in [Record prompts, replies and tool calls](#record-prompts-replies-and-tool-calls). +For an EU organization, use `https://ingest.eu.maple.dev`. `SPAN_ONLY` records prompts and replies as span attributes, which Maple needs for the transcript. Leave it unset to keep content out of Maple. `EVENT_ONLY` or `true` leaves the transcript empty. -### Python +## Python: install the GenAI instrumentation -Install the SDK, the exporter and the instrumentation for each provider you call: +Use the official OpenTelemetry GenAI packages. Note the `genai` in the name: `opentelemetry-instrumentation-openai` is a different project (OpenLLMetry), and `opentelemetry-instrumentation-openai-v2` is deprecated. ```bash pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ @@ -68,8 +49,6 @@ pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1 # Gemini: opentelemetry-instrumentation-google-genai>=1.2b0 ``` -Set up tracing once, before your first model call: - ```py # tracing.py from opentelemetry import trace @@ -89,35 +68,11 @@ OpenAIInstrumentor().instrument() # from opentelemetry.instrumentation.google_genai import GoogleGenAiSdkInstrumentor ``` -Import `tracing` at the top of your entry point. The instrumentors patch the SDK classes, so they cover every client, but the patch has to be in place before the first request. If your app already has a `TracerProvider` (Sentry, Logfire, Datadog), add the `BatchSpanProcessor` to that one instead of creating a second. - -### TypeScript - -```bash -npm install @opentelemetry/api @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-proto -``` - -```ts -// instrumentation.ts -import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" -import { NodeSDK, tracing } from "@opentelemetry/sdk-node" - -// The exporter reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. -export const spanProcessor = new tracing.BatchSpanProcessor(new OTLPTraceExporter()) - -export const sdk = new NodeSDK({ serviceName: "support-agent", spanProcessors: [spanProcessor] }) -sdk.start() -``` - -Import it first in your entry point. The helper below creates its spans through `@opentelemetry/api`, so nothing gets monkey-patched and import order only matters for the provider being registered before the first span. - -## Group turns into one session +Import `tracing` first in your entry point so the SDK is patched before the first request. If your app already has a `TracerProvider` (Sentry, Logfire, Datadog), add the `BatchSpanProcessor` to it instead of creating a second. -Maple groups traces into a session by `gen_ai.conversation.id`. It only needs the id on one AI span per trace, and the natural place is a span that wraps the whole turn: `invoke_agent`, with the agent's name. The model calls and tool calls made inside it become its children, so a turn is one trace. +## Python: wrap each turn and tool call -Use the id your app already has for the conversation: the chat thread's database id, a support ticket id, the Slack thread timestamp. Don't generate a UUID per request (every turn becomes its own session) or per process (every user shares one). - -### Python +Maple groups traces into a session by `gen_ai.conversation.id`. Put it on an `invoke_agent` span that wraps the whole turn, so the model and tool calls inside become one trace. Use the id your app already stores for the conversation (thread id, ticket id), not a new UUID per request. ```py # agent_tracing.py @@ -138,7 +93,6 @@ CAPTURE_CONTENT = os.environ.get( @contextmanager def agent_span(agent_name: str, conversation_id: str | None = None): - """One agent run. With a conversation id, it's the turn Maple files under that session.""" attributes = {"gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agent_name} if conversation_id: attributes["gen_ai.conversation.id"] = conversation_id @@ -147,7 +101,6 @@ def agent_span(agent_name: str, conversation_id: str | None = None): def run_tool(call_id: str, name: str, arguments: str, tool) -> str: - """One tool call. A failing tool returns its error to the model, and the span still says it failed.""" with tracer.start_as_current_span( f"execute_tool {name}", attributes={ @@ -171,7 +124,7 @@ def run_tool(call_id: str, name: str, arguments: str, tool) -> str: return result ``` -Your loop then looks like this. Only the two `with` and `run_tool` lines are tracing: +In your loop, the tracing is the `with agent_span(...)` line and the `run_tool(...)` call: ```py # agent.py @@ -187,7 +140,6 @@ TOOLS = {"get_weather": get_weather, "fetch_transport_data": fetch_transport_dat def chat_turn(conversation_id: str, history: list, user_text: str) -> str: - """One user message in, one reply out. `history` is this conversation's stored messages.""" with agent_span("support_agent", conversation_id): history.append({"role": "user", "content": user_text}) while True: @@ -201,11 +153,27 @@ def chat_turn(conversation_id: str, history: list, user_text: str) -> str: history.append({"role": "tool", "tool_call_id": call.id, "content": result}) ``` -With Anthropic, the loop is the same shape: run each `tool_use` block with `run_tool(block.id, block.name, json.dumps(block.input), ...)` and send the results back as `tool_result` blocks. With Gemini's automatic function calling, the SDK runs your Python functions itself and the instrumentation records an `execute_tool` span for each, so wrap the turn in `agent_span` but don't call `run_tool`. +With Anthropic, run each `tool_use` block with `run_tool(block.id, block.name, json.dumps(block.input), ...)`. With Gemini's automatic function calling, the instrumentation records the tool spans itself, so wrap the turn in `agent_span` and skip `run_tool`. + +## TypeScript: set up the exporter and helper + +There is no OpenTelemetry instrumentation that works with `openai` 7 or writes messages where Maple reads them, so a small helper records the turn, each tool call and each model call. + +```bash +npm install @opentelemetry/api @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-proto +``` + +```ts +// instrumentation.ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { NodeSDK, tracing } from "@opentelemetry/sdk-node" -### TypeScript +// The exporter reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. +export const spanProcessor = new tracing.BatchSpanProcessor(new OTLPTraceExporter()) -The helper records all three spans: the turn, each tool call, and each model call with the attributes the Python instrumentations would write. +export const sdk = new NodeSDK({ serviceName: "support-agent", spanProcessors: [spanProcessor] }) +sdk.start() +``` ```ts // agent-tracing.ts @@ -214,19 +182,16 @@ import type OpenAI from "openai" const tracer = trace.getTracer("support-agent") -// Same switch as the Python instrumentations, so one env var controls content everywhere. const captureContent = ["SPAN_ONLY", "SPAN_AND_EVENT"].includes( (process.env.OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT ?? "").toUpperCase(), ) -/** One agent run. With a conversation id, it's the turn Maple files under that session. */ export function agentSpan(agentName: string, conversationId: string | undefined, fn: () => Promise) { const attributes: Attributes = { "gen_ai.operation.name": "invoke_agent", "gen_ai.agent.name": agentName } if (conversationId) attributes["gen_ai.conversation.id"] = conversationId return withSpan(`invoke_agent ${agentName}`, SpanKind.INTERNAL, attributes, () => fn()) } -/** One tool call. A failing tool returns its error to the model, and the span still says it failed. */ export function runTool(callId: string, name: string, args: string, tool: (args: any) => unknown) { const attributes = { "gen_ai.operation.name": "execute_tool", "gen_ai.tool.name": name, "gen_ai.tool.call.id": callId } return withSpan(`execute_tool ${name}`, SpanKind.INTERNAL, attributes, async (span) => { @@ -331,6 +296,10 @@ function markFailed(span: Span, error: unknown) { } ``` +## TypeScript: wrap each turn + +Pass the conversation id your app already stores to `agentSpan`, and route model and tool calls through `tracedChat` and `runTool`: + ```ts // agent.ts import "./instrumentation.ts" @@ -341,7 +310,6 @@ const client = new OpenAI() const model = "gpt-4o-mini" // tools: Record unknown> and toolSchemas: OpenAI.Chat.ChatCompletionTool[] are your own. -/** One user message in, one reply out. `history` is this conversation's stored messages. */ export function chatTurn( conversationId: string, history: OpenAI.Chat.ChatCompletionMessageParam[], @@ -365,112 +333,31 @@ export function chatTurn( } ``` -For `@anthropic-ai/sdk` or `@google/genai`, copy `tracedChat` and change what it reads. Keep each provider's raw token counts: Maple knows that Anthropic's `input_tokens` excludes cached tokens and Gemini's `candidatesTokenCount` excludes thinking tokens, and it picks the right arithmetic from `gen_ai.provider.name`. - -| Attribute | Anthropic Messages | Gemini `generateContent` | -| --- | --- | --- | -| `gen_ai.operation.name` | `chat` | `generate_content` | -| `gen_ai.provider.name` | `anthropic` | `gcp.gemini` (`gcp.vertex_ai` on Vertex) | -| `gen_ai.response.id` / `.model` | `id` / `model` | `responseId` / `modelVersion` | -| `gen_ai.response.finish_reasons` | `[stop_reason]` | `candidates[].finishReason` | -| `gen_ai.usage.input_tokens` | `usage.input_tokens` | `usageMetadata.promptTokenCount` | -| `gen_ai.usage.output_tokens` | `usage.output_tokens` | `usageMetadata.candidatesTokenCount` | -| `gen_ai.usage.cache_read.input_tokens` | `usage.cache_read_input_tokens` | `usageMetadata.cachedContentTokenCount` | -| `gen_ai.usage.cache_write.input_tokens` | `usage.cache_creation_input_tokens` | not reported | -| `gen_ai.usage.reasoning.output_tokens` | not reported separately | `usageMetadata.thoughtsTokenCount` | - -For messages, map text blocks to `{type: "text", content}`, tool calls (`tool_use`, `functionCall`) to `{type: "tool_call", id, name, arguments}` and tool results (`tool_result`, `functionResponse`) to `{type: "tool_call_response", id, response}`. Pass the system prompt as `gen_ai.system_instructions`, a JSON array of text parts. - -If you skip the turn span, each model call is its own trace and its own session, and tool calls float as separate one-span traces. If you keep the span but drop the id, you get one session per turn. - -## Record prompts, replies and tool calls - -The instrumentations record no message content by default. `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` takes four values: - -| Value | Where content goes | In Maple | -| --- | --- | --- | -| `NO_CONTENT` (default) | Nowhere | Turns, models, tokens and tools, with an empty transcript | -| `SPAN_ONLY` | Span attributes | Full transcript. Use this. | -| `EVENT_ONLY` | Log events | Empty transcript: Maple doesn't read logs | -| `SPAN_AND_EVENT` | Both | Full transcript, plus a second copy in your logs if you export them | - -The 1.x GenAI packages (July 2026 onward) don't need `OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental`: the current GenAI conventions are the only ones they emit. Older guides that tell you to set it, or to set the capture variable to `true`, describe the deprecated `openai-v2` package, where `true` meant log events. - -The helpers above read the same variable, so tool arguments, tool results and (in TypeScript) messages follow one switch. - -Content capture sends everything your users type and everything the model answers. To keep it off in production, leave the variable unset there: you still get the session, turns, tools, tokens and failures. To keep content but drop secrets, redact in your code before the call, or run an OpenTelemetry Collector with a `redaction` or `transform` processor on `gen_ai.input.messages` and `gen_ai.output.messages`. Don't lower `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` to trim them: it cuts the JSON mid-string, and Maple drops a message attribute that doesn't parse. - -## Tools, errors and sub-agents - -Each `run_tool` call is an `execute_tool ` span with the tool name, the model's call id, the arguments and the result. The call id is what lets Maple pair the tool span with the tool call in the model's reply. - -Most agent loops catch a tool's exception and hand the error back to the model, which is right for the agent but hides the failure from tracing: the exception never escapes, so the span would end as a success. The helpers mark the span failed themselves (status `ERROR`, `error.type`, the exception recorded), and Maple counts it under the session's tool errors and on the Tools page, grouped by the error message. - -A sub-agent is another agent run inside a tool call. Call it through `run_tool` and wrap its loop in `agent_span` with its own name: - -```py -def ask_weather_worker(task: str) -> str: - # Runs inside the orchestrator's execute_tool span, in the same trace. - with agent_span("weather_worker"): - ... # its own model calls and tools -``` - -It doesn't need the conversation id, because it's in the orchestrator's trace. Maple opens a separate lane for each distinct `gen_ai.agent.name`, and an `execute_tool` span whose only child is an agent span shows as a delegation with the task and the answer. Two agents with the same name merge into one lane. +For `@anthropic-ai/sdk` or `@google/genai`, copy `tracedChat` and map that SDK's response fields. The [skill's TypeScript reference](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-provider-sdks/references/typescript.md) has the attribute table. -If you run tools in parallel, `asyncio.gather` and `Promise.all` keep the trace context. A `ThreadPoolExecutor` doesn't: submit `contextvars.copy_context().run` with your function, or the tool spans start new traces outside the turn. +## Sub-agents, streaming and cost -## Tokens and cost +A sub-agent is an agent run inside a tool call. Call it through `run_tool` and wrap its loop in `agent_span` with its own name and no conversation id. Maple draws one lane per agent name. -The Python instrumentations record input and output tokens on every model call, plus cached input and reasoning tokens when the provider reports them. The TypeScript helper does the same for OpenAI. +When you stream OpenAI Chat Completions in Python, pass `stream_options={"include_usage": True}`, or the call shows 0 tokens. The TypeScript helper sets it for you. -Streaming is the exception. OpenAI's Chat Completions API only sends usage on streamed calls when you ask for it, in a last chunk with no `choices`: +Python sessions show as unpriced, because the instrumentations record no cost. The TypeScript helper records OpenRouter's `usage.cost` when you call through OpenRouter. -```py -stream = client.chat.completions.create( - model=MODEL, messages=history, stream=True, stream_options={"include_usage": True} -) -reply = "".join(chunk.choices[0].delta.content or "" for chunk in stream if chunk.choices) -``` - -Without `stream_options`, the streamed call shows 0 tokens in Maple. Anthropic and Gemini always send usage on a stream. The Python OpenAI instrumentation and the TypeScript helper also record time to first chunk on streamed calls, which Maple shows per model call. The Anthropic instrumentation (1.2b0) doesn't record it for `messages.stream()`. - -If you call Claude or Gemini models through OpenRouter's OpenAI-compatible endpoint with the `openai` SDK, the spans say `gen_ai.provider.name=openai`. That is correct: the usage arrives in OpenAI's shape, and Maple does the arithmetic for that shape. +## Flush before a short-lived process exits -None of the Python instrumentations record cost, and Maple doesn't price tokens itself, so those sessions show as **unpriced**. In TypeScript you own the span: the helper copies OpenRouter's `usage.cost` (USD, sent on streamed calls too) to `gen_ai.usage.cost`, which Maple reads as the call's cost. Called directly, OpenAI sends no cost and the session stays unpriced. - -## Short-lived processes - -`BatchSpanProcessor` sends spans every 5 seconds, so the last turn of a script can still be in memory when the process exits. - -- **Python:** the `TracerProvider` flushes when the interpreter exits normally. In a serverless handler, a notebook or anything that ends with `os._exit`, call `provider.force_flush()` after each turn. -- **TypeScript:** call `await sdk.shutdown()` before a CLI or script exits. In a serverless handler, call `await spanProcessor.forceFlush()` before returning, or in `waitUntil` once a streamed response is sent. +In Python, a serverless handler or notebook should call `provider.force_flush()` after each turn. In TypeScript, call `await sdk.shutdown()` before a script exits, or `await spanProcessor.forceFlush()` before a serverless handler returns. ## Check that it works -Run one conversation of two or three messages with at least one tool call, flush, and open **Agent Sessions** in Maple. Sessions appear within a few seconds of the export. - -- **One session per conversation**, with the conversation id you passed. The framework column says **Unidentified**, which is expected for this setup. -- **One turn per user message**, labeled with the first line of that message (with content capture on). Each turn is an `invoke_agent support_agent` span (your agent name) holding the rest. -- **Model calls** named `chat ` for OpenAI and Anthropic (`chat gpt-4o-mini`) or `generate_content ` for Gemini, with input and output tokens. -- **Tool calls** named `execute_tool get_weather` with their arguments and results, and a failing tool counted as a tool error. -- **A transcript** with the user messages, the replies and the tool calls between them. -- **Cost** shown as unpriced, except for TypeScript calls through OpenRouter, where the helper records OpenRouter's price. +Run a conversation of two or three messages with a tool call, flush, and open **Agent Sessions** in Maple. You should see one session with your conversation id, one turn per user message, and a transcript with the replies and tool calls. The framework column says **Unidentified**, which is expected for this setup. ## Troubleshooting -- **Every model call is its own session.** The call ran outside `agent_span`, or the span was ended before the call. Wrap the whole turn, including the tool loop, in one `agent_span` with the conversation id. -- **One session per turn instead of per conversation.** The conversation id changes per request. Pass the id stored with the conversation, not a new UUID. -- **Turns show models and tokens but the transcript is empty.** Content capture is off, or set to `EVENT_ONLY` or `true`. Set `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` in the process that makes the calls. -- **No model spans in Python, only yours.** The instrumentor wasn't installed or `instrument()` ran after the first request. Check that `tracing` is imported first, and that you installed `opentelemetry-instrumentation-genai-openai`, not `opentelemetry-instrumentation-openai`. -- **Every model call appears twice.** Two instrumentations wrap the same SDK: the GenAI package plus OpenLLMetry, OpenInference, the deprecated `openai-v2` package, `logfire.instrument_openai()`, Sentry's OpenAI integration or a framework's own tracing. `opentelemetry-instrument` loads every instrumentation package that's installed, so uninstall the extras rather than just not calling them. -- **Twin traces for every call when you use OpenRouter.** OpenRouter Broadcast is also exporting the same calls. Keep one source, or nest Broadcast under your spans as the [OpenRouter guide](/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces) shows. -- **The streamed turn has 0 tokens.** Add `stream_options={"include_usage": True}` (OpenAI Chat Completions). -- **A failing tool shows as successful.** The tool's exception was caught outside `run_tool`. Let `run_tool` catch it, or set the span status and `error.type` where you catch it. -- **Sub-agent calls land in the orchestrator's lane.** The sub-agent's `agent_span` has the same name as the orchestrator's, or it was never wrapped. Give each agent its own name. -- **Two conversation ids in one trace.** With the OpenAI Responses API and a `conversation` parameter, the instrumentation also sets `gen_ai.conversation.id` to that `conv_...` id. Pass the same id to `agent_span`, or Maple picks one of the two and can split the turn. -- **Cached Anthropic calls show more input tokens than you were billed for (Python).** The Anthropic GenAI instrumentation reports `gen_ai.usage.input_tokens` as the raw input plus cache reads plus cache writes (a call that sent 330 new tokens and wrote 7,581 to the cache reports 7,911), while Maple reads Anthropic's figure as excluding the cache and adds both buckets again. In our test, a three-turn session that processed 40,239 tokens showed 78,144 in Maple. Calls without prompt caching are unaffected. -- **Using the Anthropic SDK through OpenRouter.** Point it at `https://openrouter.ai/api` (no `/v1`): `Anthropic(base_url="https://openrouter.ai/api", api_key=OPENROUTER_API_KEY)`. OpenRouter accepts the key as `api_key` or `auth_token`. Model ids are OpenRouter's, such as `anthropic/claude-haiku-4.5`, and the spans say `gen_ai.provider.name=anthropic`. -- **Nothing arrives, and the exporter logs 401.** The key or region is wrong. EU keys only work with `ingest.eu.maple.dev`. +- **Every model call is its own session.** The call ran outside `agent_span`. Wrap the whole turn, including the tool loop. +- **One session per turn.** The conversation id changes per request. Pass the id stored with the conversation. +- **The transcript is empty.** Set `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY` in the process that makes the calls. +- **No model spans in Python.** `tracing` wasn't imported first, or you installed `opentelemetry-instrumentation-openai` instead of `opentelemetry-instrumentation-genai-openai`. +- **Every model call appears twice.** A second instrumentation (OpenLLMetry, OpenInference, `logfire.instrument_openai()`, Sentry's OpenAI integration) wraps the same SDK. Uninstall the extra one. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md index cd08b4b0d5..18ec0f81e6 100644 --- a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md @@ -1,17 +1,15 @@ --- title: "Trace Pydantic AI agents with OpenTelemetry" -description: "Send Pydantic AI's built-in OpenTelemetry spans to Maple so each conversation is one Agent Session with its transcript, tool calls, sub-agents and tokens, with or without Logfire." +description: "Send Pydantic AI's built-in OpenTelemetry spans to Maple so each conversation shows up as one Agent Session." group: "AI Agents" order: 20 navLabel: "Pydantic AI" icon: "pydantic" --- -Pydantic AI has its own OpenTelemetry instrumentation, and it's good. Every `agent.run()` emits an `invoke_agent` span, a `chat` span per model request with the prompt, the reply and the token counts, and an `execute_tool` span per tool call with its arguments and result. All of it follows the GenAI semantic conventions Maple reads, so you don't need Logfire or an extra instrumentation package. You give Pydantic AI a `TracerProvider` that exports to Maple. +Pydantic AI emits OpenTelemetry spans for every run, model call and tool call, with prompts, replies and token counts. You give it a `TracerProvider` that exports to Maple. The one thing to get right is the conversation id: unless you pass one, Pydantic AI generates a new id for every run, and each message becomes its own session. -The part that goes wrong is the conversation id. Every span carries `gen_ai.conversation.id`, so the traces look grouped, but unless you pass an id, Pydantic AI generates a new UUID7 for each run. A chat backend that handles one message per request gets one session per message, and an agent that delegates to other agents gets a different id for every delegate. - -This guide covers Pydantic AI 2.x on Python 3.10 or newer. It was tested with `pydantic-ai-slim` 2.51.0 and the OpenTelemetry Python SDK 1.45.0, and with Logfire 5.1.1 for the Logfire setup. +Tested with `pydantic-ai-slim` 2.51.0, the OpenTelemetry Python SDK 1.45.0, Logfire 5.1.1 and Python 3.10+. ## Quick setup with a coding agent @@ -25,19 +23,17 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. -## Export Pydantic AI spans to Maple +## Export spans to Maple -Install the OpenTelemetry SDK and the OTLP/HTTP exporter next to Pydantic AI: +Install the OpenTelemetry SDK and the OTLP/HTTP exporter next to Pydantic AI. Swap `[openai]` for the extras of the providers you use. ```bash pip install "pydantic-ai-slim[openai]>=2.51" "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Swap `[openai]` for the extras of the providers you use (`anthropic`, `google`, `openrouter`, ...). The full `pydantic-ai` package works the same way. - -Point the exporter at Maple with the standard OpenTelemetry variables: +Point the exporter at Maple. For an EU organization, use `https://ingest.eu.maple.dev`. ```bash export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" @@ -45,9 +41,7 @@ export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" ``` -For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. - -Then set up tracing once, when your process starts: +Set up tracing once, when your process starts: ```py # tracing.py @@ -76,15 +70,13 @@ Agent.instrument_all( ) ``` -Import `tracing` at the top of your entry point (`main.py`, the FastAPI app module, the worker). `Agent.instrument_all()` sets the default for every agent, including agents created before it runs, so the order relative to your agent modules doesn't matter. What matters is that it runs before the first `agent.run()`. - -If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Sentry or your own setup), don't create a second one. Add the `BatchSpanProcessor` above to the existing provider and call `Agent.instrument_all(InstrumentationSettings(include_content=True, include_binary_content=False))` without `tracer_provider`, so Pydantic AI uses the global one. +Import `tracing` at the top of your entry point (`main.py`, the FastAPI app module, the worker). It must run before the first `agent.run()`. Agents created earlier are still covered. -Leave `version` at its default. Pydantic AI's instrumentation format is versioned separately from the package: 5 is the default, 2 to 4 are deprecated and emit a warning, and 6 is an opt-in that sends tool results with `role: "tool"`. Version 2 used different span names; this guide was tested with 5. +If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Sentry or your own setup), add the `BatchSpanProcessor` to it instead of creating a second one, and call `Agent.instrument_all(InstrumentationSettings(include_content=True, include_binary_content=False))` without `tracer_provider`. -### Already using Logfire +### If you already use Logfire -Logfire sets up the same instrumentation and can export to any OTLP backend. It brings its own OpenTelemetry SDK and OTLP exporter, so skip the `pip install` line above: Logfire 5.1 requires `opentelemetry-sdk` below 1.45, and the `>=1.45` pins make the install fail to resolve. Keep the three `OTEL_EXPORTER_OTLP_*` variables and configure Logfire like this: +Logfire brings its own OpenTelemetry SDK and exporter, so skip the `pip install` above (Logfire 5.1 pins `opentelemetry-sdk` below 1.45 and the install won't resolve). Keep the three environment variables. Logfire exports to Maple whenever `OTEL_EXPORTER_OTLP_ENDPOINT` is set. ```py import logfire @@ -93,19 +85,11 @@ logfire.configure(service_name="support-agent", environment="production", send_t logfire.instrument_pydantic_ai() ``` -Logfire adds an OTLP exporter whenever `OTEL_EXPORTER_OTLP_ENDPOINT` is set. With `send_to_logfire=True` (or a Logfire token in the environment), spans go to both Logfire and Maple. - -Logfire scrubs attributes by default, which changes what reaches Maple. The conversation id and the prompt and reply messages are exempt, but tool arguments, tool results and the run's `final_result` are not. A tool result that contains `session`, `auth`, `password`, `cookie` or `secret` anywhere in its text arrives as `[Scrubbed due to 'session']`. See [Record prompts, responses and tool calls](#record-prompts-responses-and-tool-calls) for how to keep it. - -## Group every turn of a conversation into one session - -Maple groups traces into sessions by `gen_ai.conversation.id`. Pydantic AI puts it on every span it emits, and picks the value in this order: +Logfire scrubs tool arguments and results by default. If they arrive as `[Scrubbed due to ...]`, see Troubleshooting. -1. the `conversation_id=` you pass to `run()`, `run_stream()` or `iter()`; -2. the id stamped on the last message of `message_history`; -3. a new UUID7. +## Pass the conversation id on every run -So a chat backend that stores the message history and passes it back keeps one id, as long as the first run of the conversation got one. A backend that starts each request without history, or that rebuilds the history from its own database, gets a new session for every message. Pass your own id on every run, from the chat or thread id your app already has: +Maple groups traces into sessions by `gen_ai.conversation.id`. Pydantic AI takes it from the `conversation_id=` you pass, then from the last message of `message_history`, and otherwise generates a new UUID7. Pass the chat or thread id your app already has on every `run()`, `run_stream()` and `iter()`: ```py from pydantic_ai import Agent @@ -118,11 +102,9 @@ async def handle_message(chat_id: str, text: str, history: list) -> str: return result.output ``` -The id must be stable for the whole conversation and different between conversations. A per-process constant merges every user into one session. +The id must stay the same for the whole conversation and differ between conversations. -If you skip this, each request shows up in **Agent Sessions** as its own one-turn session, named after a UUID. - -Streaming works the same way. Pass `conversation_id=` to `run_stream()`, and keep the `async with` block open until the stream is finished, because the `invoke_agent` span ends when the block exits: +When streaming, keep the `async with` block open until the stream is finished. The run's span ends when the block exits. ```py async def stream_reply(chat_id: str, text: str, history: list): @@ -131,86 +113,9 @@ async def stream_reply(chat_id: str, text: str, history: list): yield delta ``` -## Record prompts, responses and tool calls - -Content capture is on by default in Pydantic AI (`include_content=True`). Each `chat` span carries `gen_ai.input.messages`, `gen_ai.output.messages` and `gen_ai.system_instructions` as JSON in the GenAI message format, and each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Maple builds the transcript from those attributes. - -Two settings are worth changing: - -- `include_binary_content=False` keeps images, audio and documents out of the spans. With it on, a single uploaded PDF can put megabytes of base64 into every later `chat` span of that run, since each request repeats the history. -- `include_content=False` drops prompts, replies, tool arguments and tool results. The messages keep their roles and part types, so the transcript shows the shape of the conversation with no text. Turns, models, tool names, tokens and errors are unaffected. Exception messages are dropped too; only the exception type is kept. - -To turn content off for one agent only, set it on that agent: - -```py -from pydantic_ai import Agent, InstrumentationSettings -from pydantic_ai.capabilities import Instrumentation - -billing = Agent( - "openai:gpt-4o-mini", - name="billing", - capabilities=[Instrumentation(settings=InstrumentationSettings(include_content=False))], -) -``` - -An agent with its own `Instrumentation` capability ignores `Agent.instrument_all()`, and without `tracer_provider=` it uses the global provider you set in `tracing.py`. - -Logfire's scrubbing does not protect prompts. The message attributes are exempt from it, so a user who types a password into the chat sends it to Maple either way. If you need pattern-based redaction of message content, run it in an OpenTelemetry Collector between your app and Maple. - -With Logfire, keep tool content intact by letting those attributes through the scrubber: - -```py -import logfire - -TOOL_CONTENT = {"gen_ai.tool.call.arguments", "gen_ai.tool.call.result", "final_result"} - - -def keep_tool_content(match: logfire.ScrubMatch): - if len(match.path) > 1 and match.path[1] in TOOL_CONTENT: - return match.value # keep; returning None redacts - - -logfire.configure( - service_name="support-agent", - send_to_logfire=False, - scrubbing=logfire.ScrubbingOptions(callback=keep_tool_content), -) -``` - -## Tools, errors and sub-agents - -Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the provider's `gen_ai.tool.call.id`, and the arguments and result. Maple matches each call to the model reply that requested it by that id. - -How a tool fails decides what Maple shows: +### Sub-agents -| In your tool | Span status | Run continues | In Maple | -| --- | --- | --- | --- | -| `raise ToolFailed("...")` | ERROR | yes, the model sees the message | failed call, message as its result | -| `raise ModelRetry("...")` | ERROR | yes, the model retries | one failed call per retry | -| any other exception | ERROR | no, the run raises | failed call and failed turn | -| `return {"error": "..."}` | UNSET | yes | successful call | - -For an upstream error the model should work around, raise `ToolFailed` (Pydantic AI 2.16 or newer). It marks the span failed, records the message as the tool result, and doesn't use up the tool's retry budget: - -```py -from pydantic_ai import ToolFailed - - -@support.tool_plain -def fetch_transport_data(city: str) -> dict: - """Fetch live public-transport data for a city.""" - raise ToolFailed("transport data service unavailable (503)") -``` - -Returning an error payload keeps the span green, and Maple counts the call as a success. - -Approval-gated tools (`requires_approval=True`) pause the run without a tool span. The call only gets an `execute_tool` span when the resumed run executes it, so pass the same `conversation_id=` to the run that sends `DeferredToolResults`. Maple then shows the paused run and the resumed run as two turns of the same session, both labeled with the original request. - -### Sub-agents: pass the conversation id down - -The common multi-agent pattern in Pydantic AI is delegation: a tool on the orchestrator calls `worker.run()`. The worker's run is nested in the orchestrator's trace, under the tool span, but it resolves its own conversation id, and with no history and no explicit id, that is a new UUID7. One trace then carries several ids. Maple uses one of them for the whole trace, and not necessarily yours, and it can split the turn into one turn per id. - -Pass the caller's id and usage to every delegate: +When a tool calls another agent's `run()`, that run generates its own id unless you pass the caller's. Pass `conversation_id` and `usage` from the tool's context. Give every agent a `name=`, which Maple uses to draw one lane per agent. ```py from pydantic_ai import Agent, RunContext @@ -230,25 +135,9 @@ async def research_weather(ctx: RunContext[None], city: str) -> str: return result.output ``` -Give every agent an explicit `name=`. It becomes `gen_ai.agent.name` on all of its spans, and Maple draws a lane per agent name. An `execute_tool research_weather` span whose only child is `invoke_agent weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. When several delegation tools are called in one model reply, Pydantic AI runs them concurrently and the lanes overlap in time. - -A pipeline of separate top-level runs (orchestrator, then a summary agent) produces one trace per run. With the same `conversation_id=` on each, they land in one session as consecutive turns. - -## Tokens and cost - -Each `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. Maple sums them per model call. +## Flush short-lived processes -One exception: with Anthropic models on Pydantic AI's `anthropic` provider and prompt caching, Maple currently counts cached input tokens twice. Pydantic AI reports input tokens including the cache for every provider, and Maple applies Anthropic's own convention, where they're separate. OpenAI, OpenRouter and Gemini calls aren't affected. - -The `invoke_agent` span reports the run's own total under `gen_ai.aggregated_usage.*`, which Maple doesn't add to the session total, so nothing is counted twice. A delegate's tokens stay on the delegate's spans. - -Streaming needs nothing extra. Pydantic AI requests usage on OpenAI-compatible streams (`stream_options.include_usage`), so streamed calls have token counts too. The streamed `chat` span also records time to first chunk, under a key Maple doesn't read yet. - -Cost shows as unpriced. Pydantic AI prices each call and writes the result to `operation.cost` on the `chat` span, but Maple reads cost only from `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or `llm.cost.total`, and never prices tokens itself. Tokens, models and call counts are complete. - -## Short-lived processes - -`BatchSpanProcessor` exports every few seconds, and the SDK flushes on a normal interpreter exit. That doesn't cover a Lambda that freezes after the handler returns, a worker killed by its supervisor, `os._exit()`, or a notebook kernel that never exits. Flush explicitly in those cases: +The SDK flushes on a normal interpreter exit. A Lambda that freezes after returning, a killed worker or a notebook kernel doesn't exit normally, so flush explicitly: ```py import asyncio @@ -263,35 +152,21 @@ def handler(event, context): trace.get_tracer_provider().force_flush() ``` -In a script or CLI, call `provider.shutdown()` at the end. With Logfire, use `logfire.force_flush()` and `logfire.shutdown()`. +In a script or CLI, call `provider.shutdown()` at the end. With Logfire, use `logfire.force_flush()` or `logfire.shutdown()`. ## Check that it works -Before you look in Maple, check the console. Pydantic AI prints an `observability: off` banner on the first run when no instrumentation is set up. If it's still there, `tracing.py` didn't run before your first `agent.run()`. - -Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: +Pydantic AI prints an `observability: off` banner on the first run when no instrumentation is set up. If you still see it, `tracing.py` didn't run before your first `agent.run()`. -- one session per conversation, with the id you passed as `conversation_id`, and framework **Pydantic AI**; -- one turn per `run()`, each labeled with the user's message, and a transcript with the prompts, replies and tool calls; -- `chat ` spans for model calls, `execute_tool ` spans for tool calls, and `invoke_agent ` spans for runs; -- a lane per sub-agent, named after its `name=`; -- token counts on every model call, including streamed ones; -- failed tool calls marked as failed, with the `ToolFailed` message as the result; -- cost shown as unpriced. +Run a conversation with two messages and a tool call, then open **Agent Sessions**. You should see one session with your conversation id and framework **Pydantic AI**, one turn per `run()`, a transcript with prompts, replies and tool calls, and token counts on every model call. Cost shows as unpriced, because Maple doesn't read the cost attribute Pydantic AI writes. ## Troubleshooting -- **Every message is its own session.** No `conversation_id=` was passed and the history didn't carry one. Pass it on every `run()`, `run_stream()` and `iter()`. -- **A multi-agent run shows several turns, or lands in another session.** Delegates minted their own ids. Pass `conversation_id=ctx.conversation_id` to every nested `run()`. -- **All sub-agents share one lane called `agent`.** The agents have no `name=`. Set one on each. -- **Nothing arrives from a script or Lambda.** The process ended before the batch was exported. Call `force_flush()` or `shutdown()` in a `finally`. -- **The transcript has messages but no text.** `include_content=False` is set, in `instrument_all()` or in that agent's `Instrumentation` capability. -- **Tool arguments or results read `[Scrubbed due to ...]`.** Logfire's scrubbing matched a word in them. Add a scrubbing callback or set `scrubbing=False`. -- **A failed tool shows as successful.** The tool returned an error value instead of raising. Raise `ToolFailed`. -- **`chat` spans are huge or exports fail with 413.** Images or documents are being recorded as base64. Set `include_binary_content=False`. -- **Spans show up twice.** Pydantic AI and a second instrumentor (Logfire's `instrument_openai()`, OpenInference, OpenLLMetry) both trace the same model calls. Keep Pydantic AI's instrumentation and remove the other one for the model client. -- **The install fails to resolve `logfire` and `opentelemetry-sdk`.** Logfire 5.1 pins the SDK below 1.45. On the Logfire path, don't add the OpenTelemetry packages yourself; Logfire installs them. -- **A `PydanticAIDeprecationWarning` about instrumentation versions 2, 3 and 4.** Remove `version=` from `InstrumentationSettings` to use the default. +- **Every message is its own session.** Pass `conversation_id=` on every `run()`, `run_stream()` and `iter()`. +- **A multi-agent run is split into several turns or sessions.** Pass `conversation_id=ctx.conversation_id` to every nested `run()`. +- **Tool arguments or results read `[Scrubbed due to ...]`.** Logfire's scrubbing matched a word like `session` or `auth`. Pass `scrubbing=logfire.ScrubbingOptions(callback=...)` that keeps `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, or `scrubbing=False`. +- **A failed tool shows as successful.** The tool returned an error value. Raise `ToolFailed("...")` (Pydantic AI 2.16+) so the call is marked failed and the model still sees the message. +- **Spans show up twice.** Another instrumentor (Logfire's `instrument_openai()`, OpenInference, OpenLLMetry) also traces the model client. Remove it and keep Pydantic AI's. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/smolagents.md b/apps/landing/src/content/docs/agent-tracing/smolagents.md index 0af22f2afe..c7d239a63c 100644 --- a/apps/landing/src/content/docs/agent-tracing/smolagents.md +++ b/apps/landing/src/content/docs/agent-tracing/smolagents.md @@ -1,17 +1,15 @@ --- title: "Trace Hugging Face smolagents with OpenTelemetry" -description: "Send smolagents runs to Maple as Agent Sessions, one per conversation, with the transcript, model and tool calls, tokens, failed tools and managed agents in their own lanes." +description: "Send smolagents runs to Maple through the OpenInference instrumentor so each conversation shows up as one Agent Session." group: "AI Agents" order: 25 navLabel: "smolagents" icon: "huggingface" --- -smolagents has no OpenTelemetry code of its own. Every span comes from OpenInference's `openinference-instrumentation-smolagents`, which patches `MultiStepAgent.run`, each agent step, every model's `generate` and `Tool.__call__`. By default those spans use OpenInference attribute names (`llm.input_messages.0.message.role`, `llm.token_count.prompt`), and every `agent.run()` starts a new trace with no conversation id. +smolagents has no tracing of its own. Its spans come from OpenInference's `openinference-instrumentation-smolagents`, which records every run, step, model call and tool call. Two defaults need changing for Maple: turn on the instrumentor's GenAI attributes, or the session page shows no transcript, and wrap each run in `using_session(...)`, or every message becomes its own session. -Two defaults have to change for Maple. Maple's session page reads the OpenTelemetry GenAI attributes (`gen_ai.*`) for smolagents, not OpenInference's own, so without the instrumentor's GenAI dual-write the session list shows token counts and the session page shows no transcript. And without `using_session(...)` around each run, a ten-message chat becomes ten one-turn sessions. - -This guide covers smolagents 1.26 with `openinference-instrumentation-smolagents` 0.1.40 on Python 3.10 or later, for both `ToolCallingAgent` and `CodeAgent`. +Tested with smolagents 1.26 and `openinference-instrumentation-smolagents` 0.1.40 on Python 3.10+, with `ToolCallingAgent` and `CodeAgent`. ## Quick setup with a coding agent @@ -25,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. ## Install the instrumentor and export to Maple @@ -34,9 +32,9 @@ pip install "smolagents[openai]>=1.26" "openinference-instrumentation-smolagents "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Skip the `smolagents[telemetry]` extra from the smolagents docs. It also installs the Arize Phoenix server, which you don't need to send traces to Maple. The `[openai]` extra is for `OpenAIServerModel`; use `[litellm]` if you run `LiteLLMModel`. +Use `[litellm]` instead of `[openai]` if you run `LiteLLMModel`. Skip the `smolagents[telemetry]` extra, which also installs the Arize Phoenix server. -Point the exporter at Maple with the standard OpenTelemetry variables: +Point the exporter at Maple. For an EU organization, use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. ```bash export OTEL_SERVICE_NAME=support-agent @@ -45,9 +43,7 @@ export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" ``` -EU organizations use `https://ingest.eu.maple.dev`. Set the base URL, not a signal path: `OTLPSpanExporter()` with no arguments appends `/v1/traces` to `OTEL_EXPORTER_OTLP_ENDPOINT`. If you pass `endpoint=` in code instead, it's used as is, so it has to end in `/v1/traces` or every export gets a 404. - -Then add a `tracing.py` and import it at the top of your entry point: +Add a `tracing.py` and import it at the top of your entry point, before the first `agent.run()`: ```py # tracing.py @@ -94,18 +90,13 @@ SmolagentsInstrumentor().instrument( ) ``` -What each part does: - -- **`enable_genai_semconv=True`** makes the instrumentor write `gen_ai.operation.name`, `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.usage.*` and `gen_ai.tool.*` next to its OpenInference attributes when each span ends. The environment variable `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` does the same, but only if it's set before `TraceConfig` is built; passing it in code avoids that ordering trap. -- **`SmolagentsForMaple`** fixes four things in the instrumentor's output: it names each agent, zeroes the token totals on run spans (see [Tokens and cost](#tokens-and-cost)), names tool spans after the tool, and records the call's arguments instead of the tool's schema. It runs in `on_start`, before the dual-write, and the dual-write never overwrites a key that's already set. - -The instrumentor patches smolagents' classes in place, so import order doesn't matter, as long as `instrument()` runs before the first `agent.run()`. It only patches the model classes smolagents exports. A `Model` subclass of your own that overrides `generate` produces no model spans. +`enable_genai_semconv=True` makes the instrumentor write the `gen_ai.*` attributes Maple reads for the transcript, tokens and tools. `SmolagentsForMaple` names each agent so managed agents get their own lane, stops run spans from counting tokens a second time, and fixes tool span names and arguments. Keep it as is. -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), don't create a second one. Add `SmolagentsForMaple()` and the OTLP exporter to the existing provider and pass that provider to `instrument()`. +If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add `SmolagentsForMaple()` and the exporter to that provider and pass it to `instrument()` instead of creating a second one. -## Group a conversation into one session +## Wrap each run in the conversation id -smolagents has no session, thread or conversation id. Memory lives on the agent object: `agent.run(task, reset=False)` continues the previous conversation, and every call is a new root span and a new trace. Maple groups traces into a session by the `session.id` attribute, and the instrumentor only sets it inside OpenInference's `using_session` context manager: +smolagents has no conversation id, and every `agent.run()` starts a new trace. Maple groups traces by `session.id`, which the instrumentor sets inside OpenInference's `using_session` block: ```py from openinference.instrumentation import using_session @@ -126,65 +117,15 @@ def handle_message(conversation_id: str, text: str) -> str: return str(agent.run(text, reset=False)) ``` -Use your app's own conversation id, the one it already stores the chat under. A new UUID per request gives you one session per message again, and a constant gives every user one shared session. - -`using_session` stores the id in a Python contextvar, not in OpenTelemetry baggage, and the instrumentor copies it onto every span it creates inside the block, including the model and tool spans of managed agents and of tool calls running in parallel threads. It does not reach spans you create with a plain OpenTelemetry tracer. If you add your own spans, pass `dict(get_attributes_from_context())` (also from `openinference.instrumentation`) as their attributes. `using_attributes(session_id=..., user_id=...)` works the same way and also sets `user.id`. - -If you skip this, every `agent.run()` shows up in **Agent Sessions** as its own one-turn session named after its trace id. Setting `gen_ai.conversation.id` yourself doesn't help: Maple reads `session.id` for smolagents. - -One agent object per conversation matters for more than tracing. A single module-level agent with `reset=False` shares its memory across every user who talks to it, and its traces would look like one long conversation. - -## Record prompts, responses and tool calls - -Content capture is on by default. Every model span carries the full message list sent to the model (system prompt, the task, earlier steps, tool results) and the model's reply, and every tool span carries the tool's arguments and result. With the GenAI dual-write on, Maple renders these as the session transcript. - -That full message list is large. smolagents' default system prompt is about 3,100 characters for `ToolCallingAgent` and 8,500 for `CodeAgent`, and with `reset=False` every model span repeats the whole conversation so far. Maple has no per-attribute limit, and ingest accepts requests up to 20 MiB. - -To keep prompts and outputs out of your traces: - -```py -config = TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True) -``` - -`hide_inputs` drops the input messages and replaces `input.value` with `__REDACTED__`; `hide_outputs` does the same for outputs. The session still shows its turns, model and tool calls, tokens and failures, with an empty transcript. Narrower switches exist: `hide_input_text` and `hide_output_text` keep the message structure but redact the text, and `hide_llm_invocation_parameters` drops temperature and max tokens. Each has an `OPENINFERENCE_HIDE_*` environment variable. - -These switches don't cover everything. The run span's `smolagents.task` attribute holds the previous `agent.run()` task in plain text, and the instrumentor doesn't mask it. If prompts can contain personal data, drop `smolagents.task` in an OpenTelemetry Collector with the `attributes` processor, or redact there with the `redaction` processor. +Use the id your app already stores the chat under. It must stay the same across the conversation and differ between conversations. Setting `gen_ai.conversation.id` yourself won't group anything, because Maple reads `session.id` for smolagents. -## Tools, errors and managed agents +Keep one agent object per conversation, as above. A single shared agent with `reset=False` mixes every user's memory into one conversation. -Each tool call is a span of OpenInference kind `TOOL`, with `gen_ai.operation.name` `execute_tool`, the tool's name in `gen_ai.tool.name` and its return value in `gen_ai.tool.call.result`. smolagents' own `final_answer` is a tool too, so every run that finishes normally ends with an `execute_tool final_answer` span. +Give every agent, including managed agents, a `name`. Unnamed agents share one lane. -A tool that raises is marked failed without extra code. The instrumentor ends the tool span with status `ERROR` and the exception as the status message, for example `RuntimeError: transport data service unavailable (503)`, and Maple counts it on the session and on the tool's page. The enclosing `Step N` span is also marked `ERROR`, with the wrapped `AgentToolExecutionError`, and carries two `exception` events for one failure. +## Flush before the process exits -smolagents feeds the error back to the model with "Now let's retry: take care not to repeat previous errors!", so a tool that's down stays down until `max_steps`. You'll see the same failed tool three or four times in one turn, and after `max_steps` the model is asked to answer without tools, which is where it tends to make data up. A `step_callbacks` hook that tells the model to stop after the first failure keeps both the trace and the answer honest. - -With `ToolCallingAgent`, a model reply that calls `final_answer` together with another tool raises `AgentExecutionError`, and that step shows as a failed `Step` span even though the run continues. That's the framework rejecting the model's output, not your code failing. - -Managed agents (`managed_agents=[...]`) show up as their own `.run` spans under the manager's step, with the managed agent's steps, model calls and tools inside. There's no tool span around the delegation. The `SmolagentsForMaple` processor turns the span name into `gen_ai.agent.name`, and Maple opens a lane for each agent whose name differs from its caller's. Give every agent a `name`; an unnamed agent's span is `ToolCallingAgent.run` or `CodeAgent.run`, and two unnamed agents share one lane. - -When a `ToolCallingAgent` model asks for several tools or managed agents in one reply, smolagents runs them in a thread pool (`max_tool_threads`) and copies the context into each thread, so parallel workers stay in the same trace and session. - -`CodeAgent` calls tools from the Python code the model writes, run by `LocalPythonExecutor` in your process. Those calls produce the same tool spans; positional arguments are recorded as a JSON array, like `["Berlin"]`. With a remote executor (`executor_type="e2b"`, `"docker"`, `"modal"` and others), the code and its tool calls run outside your process, so there are no tool spans, only the model and step spans. - -Two gaps remain in what Maple can show for smolagents tools. Tool spans have no `gen_ai.tool.call.id`, because smolagents doesn't pass the model's tool call id to the tool, so Maple can't link a tool span to the exact call in the model's reply. And the step's failure and the tool's failure are both on the trace, so a session with one broken tool has two failed spans. Maple counts one failed tool call, and reports the failed `Step` span separately as an "Other errors" warning with the `AgentToolExecutionError` message. - -## Tokens and cost - -Every model span carries input and output tokens from the provider's reply, as `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` plus the OpenInference `llm.token_count.*` originals. The instrumentor doesn't record cached or reasoning tokens, even when the provider reports them. The model is the id you passed, `gen_ai.request.model` `gpt-4o-mini` or `openai/gpt-4o-mini`; smolagents doesn't record the model name the provider returns. - -The provider comes from the model class, not the model: `OpenAIServerModel` is always `openai`, even for an Anthropic model behind OpenRouter. The model span is named after the class too: `OpenAIServerModel` is an alias of `OpenAIModel`, so its spans are `OpenAIModel.generate`. - -Streaming (`stream_outputs=True` on the agent) produces `OpenAIModel.generate_stream` spans. smolagents requests `stream_options={"include_usage": True}` for `OpenAIServerModel`, `LiteLLMModel` and `InferenceClientModel`, so streamed calls from those keep their token counts. - -The instrumentor also copies the agent monitor's token totals onto each `.run` span. Those totals repeat the model spans below it, and with `reset=False` the monitor is never reset, so turn four's run span carries the tokens of turns one to four. Maple can't net them against the model calls, because a `Step N` span sits in between. In our test, a four-turn conversation with 7,675 input tokens of model calls had another 15,626 on its run spans. `SmolagentsForMaple` sets `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens` to 0 on run spans. Maple reads those before OpenInference's `llm.token_count.*`, so each model call is counted once, and the OpenInference totals stay on the span for other tools. - -Maple shows cost only when a span carries one, and smolagents never records cost. Sessions show as **unpriced**, with token counts. - -Don't add `openinference-instrumentation-openai` or `openinference-instrumentation-litellm` next to the smolagents instrumentor. Neither checks for an existing model span, so every `OpenAIModel.generate` or `LiteLLMModel.generate` span gets a second model span under it for the same request. - -## Flush spans before the process exits - -`BatchSpanProcessor` exports every 5 seconds. The `TracerProvider` registers an `atexit` handler that flushes on a normal interpreter exit, which covers most scripts and CLIs. It doesn't run when the process is killed, calls `os._exit`, or is frozen between serverless invocations, and a notebook never exits. Flush yourself in those cases: +The SDK flushes on a normal interpreter exit. Serverless functions, killed workers and notebooks don't exit normally, so flush after each run: ```py from tracing import provider @@ -195,36 +136,19 @@ finally: provider.force_flush() # serverless: before returning; notebooks: after each run ``` -Call `provider.shutdown()` instead of `force_flush()` when the process is about to exit and won't trace anything else. - ## Check that it works -Run one conversation of two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions** in Maple. You should see: - -- **One session** for the conversation, framework **smolagents**, with one turn per `agent.run()`. Turn traces start at `assistant.run` (your agent's name). Every turn, and the session title, is labeled `New task:`, not with your message: Maple labels a turn with the first line of its user message, and smolagents puts `New task:` on that line. -- **The transcript**: your messages and the model's replies. smolagents sends each task to the model as `New task:` followed by your text, and tool results come back as `tool-response` messages. -- **Model calls** named `OpenAIModel.generate` (or `LiteLLMModel.generate`, `InferenceClientModel.generate`), each with a model, input and output tokens. -- **Tool calls** named `execute_tool get_weather` and `execute_tool final_answer`, with arguments and results. The tool-call count includes one `final_answer` per agent run, managed agents included. -- **Agents**: `assistant`, plus one lane per managed agent if you use them. -- **Cost**: unpriced. +Run two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions**. You should see one session with framework **smolagents**, one turn per `agent.run()`, a transcript, `OpenAIModel.generate` model calls with token counts, and `execute_tool ` tool calls. Each run also ends with an `execute_tool final_answer` call. -A second conversation with a different id is a second session. If a turn is missing, check that the process flushed. +Turns and the session title read `New task:`, because smolagents prefixes every task with that line. Cost shows as unpriced, since smolagents records none. ## Troubleshooting -- **No spans at all.** `instrument()` never ran, or ran after the agent was used. Import `tracing` first in the entry point and look for an `OTLPSpanExporter` error in the logs. -- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use `OTEL_EXPORTER_OTLP_ENDPOINT` with the base URL, or pass the full path. -- **Tokens in the list, empty session page.** The GenAI dual-write is off. Pass `TraceConfig(enable_genai_semconv=True)`, or set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` before `instrument()` runs. -- **One session per message.** The run isn't inside `using_session(...)`, or the id changes per request. Wrap every `agent.run()` and use the stored conversation id. -- **Two users in one session.** A shared agent object with `reset=False`, or a constant session id. Keep one agent and one id per conversation. -- **Every tool span is named `SimpleTool`.** `@tool` functions all become instances of a class called `SimpleTool`, and the instrumentor names tool spans after the class. `SmolagentsForMaple` renames them; `gen_ai.tool.name` has the real name either way. -- **Tool arguments show the tool's input schema.** The dual-write copies `tool.parameters`, the schema, into `gen_ai.tool.call.arguments`. `SmolagentsForMaple` replaces it with the call's arguments. -- **No lanes for managed agents.** The instrumentor puts the agent's name only in the span name. Add `SmolagentsForMaple` and give each agent a `name`. -- **A failing tool appears three or four times.** smolagents tells the model to retry after every error, up to `max_steps`. Stop it with a `step_callbacks` hook or a lower `max_steps`. -- **Session tokens are about double the model calls, or grow faster every turn.** The run spans' token totals are being counted. Keep the two zero-token lines for `.run` spans in `SmolagentsForMaple`. -- **Every model call appears twice.** A provider instrumentor (`openinference-instrumentation-openai` or `-litellm`) is also installed. Remove it. -- **No tool spans with `CodeAgent`.** A remote executor runs the generated code outside your process. Only `LocalPythonExecutor` produces tool spans. -- **A custom model class has no model spans.** The instrumentor only patches the model classes smolagents exports. Subclass one of them without overriding `generate`. +- **Tokens in the list, but the session page is empty.** Pass `TraceConfig(enable_genai_semconv=True)` to `instrument()`. +- **One session per message.** Wrap every `agent.run()` in `using_session(...)` with the stored conversation id. +- **Exports fail with 404.** `OTLPSpanExporter(endpoint=...)` doesn't append `/v1/traces`. Use the environment variable with the base URL, or pass the full path. +- **Every model call appears twice.** Remove `openinference-instrumentation-openai` or `openinference-instrumentation-litellm`; the smolagents instrumentor already covers model calls. +- **No tool spans with `CodeAgent`.** A remote executor (`e2b`, `docker`, `modal`) runs tools outside your process. Only the local executor produces tool spans. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/spring-ai.md b/apps/landing/src/content/docs/agent-tracing/spring-ai.md index 9d78aba8ff..b5016e7895 100644 --- a/apps/landing/src/content/docs/agent-tracing/spring-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/spring-ai.md @@ -1,17 +1,15 @@ --- title: "Trace Spring AI agents with OpenTelemetry" -description: "Send Spring AI ChatClient, model and tool spans to Maple with every turn sampled, one session per chat memory conversation, the transcript on the spans and failed tools marked as failed." +description: "Send Spring AI ChatClient, model and tool spans to Maple, with one session per chat memory conversation." group: "AI Agents" order: 31 navLabel: "Spring AI" icon: "spring" --- -Spring AI instruments itself with Micrometer Observations. Add Spring Boot's OpenTelemetry starter and every `ChatClient` call becomes a `spring_ai chat_client` span, with a `chat ` span per model call and an `execute_tool ` span per tool call. The model spans follow the OpenTelemetry GenAI conventions (model, token counts, cache tokens, finish reasons, response id), and Maple recognizes all of it as Spring AI without a separate instrumentation library. +Spring AI already emits a `spring_ai chat_client` span per `ChatClient` call, a `chat ` span per model call and an `execute_tool ` span per tool call, with model and token counts. You add Spring Boot's OpenTelemetry starter, set sampling to 100%, and add one configuration class that writes the transcript, tool names and failed tools where Maple reads them. The thing to get right is passing the conversation id on every call. -Four defaults work against you. Spring Boot samples 10% of traces, so nine turns out of ten never arrive. Prompts and replies are never written to spans: `log-prompt` and `log-completion` send them to the application log. A tool that throws ends its span as a success, because Spring AI hands the error message back to the model. And the advisor spans (`tool _calling `, `message_chat_memory`) have names Maple reads as extra tool and model calls. This guide fixes all four with a handful of properties and one configuration class. - -This guide covers Spring AI 2.0 (tested on 2.0.1) on Spring Boot 4.1 (4.1.1) and Java 21, with notes for Spring AI 1.1 on Boot 3.5. +Tested with Spring AI 2.0.1 on Spring Boot 4.1.1 and Java 21. Spring AI 1.1 on Boot 3.5 works too, with different dependencies and property names listed in the [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai). ## Quick setup with a coding agent @@ -25,40 +23,24 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. -## Install the OpenTelemetry starter and export to Maple +## Install the OpenTelemetry starter -Spring AI 2.0 requires Spring Boot 4. Import the Spring AI BOM and add your model starter and Boot's OpenTelemetry starter: +Next to the `spring-ai-bom` import (2.0.1) and your model starter, such as `spring-ai-starter-model-openai`, add Boot's OpenTelemetry starter: ```xml - - - - org.springframework.ai - spring-ai-bom - 2.0.1 - pom - import - - - - - - - org.springframework.ai - spring-ai-starter-model-openai - - - org.springframework.boot - spring-boot-starter-opentelemetry - - + + org.springframework.boot + spring-boot-starter-opentelemetry + ``` -`spring-boot-starter-opentelemetry` brings the Micrometer-to-OpenTelemetry tracing bridge, the OpenTelemetry SDK and the OTLP exporter. On Boot 4 you don't need Actuator for tracing. With Gradle, use the same two artifacts and `platform("org.springframework.ai:spring-ai-bom:2.0.1")`. +It brings the Micrometer tracing bridge, the OpenTelemetry SDK and the OTLP exporter. Boot 4 doesn't need Actuator for tracing. + +## Export to Maple -Then point the exporter at Maple in `application.properties`: +Add to `application.properties`: ```properties spring.application.name=support-agent @@ -76,15 +58,13 @@ management.otlp.metrics.export.headers.Authorization=Bearer ${MAPLE_INGEST_KEY} maple.ai.capture-content=true ``` -EU organizations use `https://ingest.eu.maple.dev`. Unlike the `OTEL_EXPORTER_OTLP_ENDPOINT` variable, this property takes the full URL, so keep `/v1/traces` on the end. The transport defaults to OTLP over HTTP with protobuf, which is what Maple ingest expects. `spring.application.name` becomes `service.name`. - -`management.tracing.sampling.probability=1.0` is the line that matters most. Boot's default is `0.1`, and a sampled-out turn leaves a hole in the session with no error anywhere. If you'd rather not send metrics, replace the two metrics lines with `management.otlp.metrics.export.enabled=false`. +For an EU organization, use `https://ingest.eu.maple.dev`. This property takes the full URL, so keep `/v1/traces` on the end. To skip metrics, replace the two metrics lines with `management.otlp.metrics.export.enabled=false`. -On Boot 4.1 you can configure the exporter with the standard variables instead: `OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev` and `OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer YOUR_INGEST_KEY` cover traces, metrics and logs, and Boot appends the signal path itself. Keep the sampling property either way. +Keep `management.tracing.sampling.probability=1.0`. Boot's default is `0.1`, which drops nine turns out of ten without any error. ## Add the attributes Maple reads -Spring AI's spans get you model calls and tokens. The transcript, tool names, failed tools and agent names need one configuration class. It uses three standard Micrometer and Spring AI extension points, so no Spring AI class is patched or replaced: +Add this class in a package your application scans. Boot registers both beans by itself: ```java package com.example.agent; @@ -194,33 +174,15 @@ public class MapleAiObservationConfig { } ``` -What each bean does: - -- **`ObservationFilter`** runs when each observation stops, after Spring AI's own conventions. It relabels the `chat_client` span as `invoke_agent` (Spring AI calls it `framework`, which Maple would otherwise count as a model call because the span name contains "chat"), copies the tool name and call id to the `gen_ai.tool.*` keys, and writes the conversation as `gen_ai.input.messages` and `gen_ai.output.messages`. It also renames the advisor spans to `spring_ai advisor`. Maple classifies spans without a known operation by name, so `tool _calling ` would count as a tool call and `message_chat_memory` as a model call on every turn. -- **`ToolExecutionExceptionProcessor`** marks the running tool span as failed before Spring AI turns the exception into a message for the model. See [Tools, errors and sub-agents](#tools-errors-and-sub-agents). - -Boot applies `ObservationFilter` beans to the observation registry by itself; there is nothing else to register. The class works the same in Kotlin. +The `ObservationFilter` marks each `ChatClient` call as an agent turn, copies tool names and call ids to the `gen_ai.tool.*` keys, and, with `maple.ai.capture-content=true`, writes the messages and tool arguments and results to the spans. Spring AI's own `log-prompt` and `log-completion` settings only write to the application log. The filter also renames the advisor spans, which Maple would otherwise count as extra model and tool calls. -Renaming the advisor spans is deliberate. Dropping them with an `ObservationPredicate` looks cleaner and works for `.call()`, but on `.stream()` Spring AI reads the model span's parent from the Reactor context, where it finds the skipped advisor. The streamed `chat` span then starts a trace of its own, outside the session. +The `ToolExecutionExceptionProcessor` marks a tool span as failed when the tool throws, then hands the error to the model as before. If your app already defines one, add the `error()` call to it instead of adding a second bean. -### Spring AI 1.1 on Spring Boot 3.5 - -The same approach works, with these differences: - -| | Spring AI 2.0, Boot 4 | Spring AI 1.1, Boot 3.5 | -|---|---|---| -| Tracing dependencies | `spring-boot-starter-opentelemetry` | `io.micrometer:micrometer-tracing-bridge-otel`, `io.opentelemetry:opentelemetry-exporter-otlp` and `spring-boot-starter-actuator` | -| Endpoint property | `management.opentelemetry.tracing.export.otlp.endpoint` | `management.otlp.tracing.endpoint` | -| Header property | `management.opentelemetry.tracing.export.otlp.headers.*` | `management.otlp.tracing.headers.*` | -| JSON in the filter | Jackson 3 `JsonMapper.shared()` | Jackson 2 `ObjectMapper` (checked exception) | -| Tool call id | `getToolCallId()` | not available, drop those lines | -| OpenAI model property | `spring.ai.openai.chat.model` | `spring.ai.openai.chat.options.model` | - -Boot 4 still accepts the Boot 3 property names but marks them deprecated. +Set `maple.ai.capture-content=false` to keep message and tool content out of Maple. Sessions, tokens, tool names and failures still show up, with an empty transcript. ## Group turns into one session -A chat backend handles one request per user message, and each `ChatClient` call is its own trace. Maple joins those traces into one session by the `spring.ai.chat.client.conversation.id` attribute on the `chat_client` span. Spring AI sets it from the `ChatMemory.CONVERSATION_ID` advisor parameter, the same parameter that selects the chat memory history: +Maple joins the traces of one conversation by the `spring.ai.chat.client.conversation.id` attribute. Spring AI sets it from the `ChatMemory.CONVERSATION_ID` advisor parameter, so pass that parameter on every request: ```java import org.springframework.ai.chat.client.ChatClient; @@ -253,104 +215,11 @@ public class ChatService { } ``` -Without the parameter, the attribute is missing and Maple shows every message as its own one-turn session, named after its trace id. Things to get right: - -- **Pass the id on every call.** It belongs on the request (`.advisors(...)` on `prompt()`), not on the builder. A constant set with `defaultAdvisors` puts every user into one session. -- **Use the conversation id your app already stores**, not a fresh UUID per request. -- **The id works without chat memory.** If your app sends the history itself, pass the parameter anyway; Spring AI records it from the request whether or not a memory advisor reads it. -- **Sub-agents don't need it.** Maple groups a whole trace by any span in it that carries the id, so the top-level `ChatClient` call is enough. - -Maple reads only `spring.ai.chat.client.conversation.id` for Spring AI spans. Adding `gen_ai.conversation.id` or `session.id` does nothing here. - -## Record prompts, responses and tool calls - -Spring AI has content switches, but none of them put the conversation where Maple reads it: - -- `spring.ai.chat.observations.log-prompt` and `log-completion` (and the `spring.ai.chat.client.observations.*` pair) write the prompt and reply to SLF4J at INFO level, with the trace id for correlation. They never add span attributes, and Maple's session views read span attributes only. -- `spring.ai.tools.observations.include-content` puts tool arguments and results on the span, but under `spring.ai.tool.call.arguments` and `spring.ai.tool.call.result`, which Maple doesn't read. - -With `maple.ai.capture-content=true`, the filter above writes the attributes Maple does read: - -| Span | Attributes | Contents | -|---|---|---| -| `chat ` | `gen_ai.input.messages` | everything sent to the model: system prompt, the history chat memory added, tool calls and tool results | -| `chat ` | `gen_ai.output.messages` | the reply, including tool calls the model asked for | -| `execute_tool ` | `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result` | the JSON arguments and the string returned to the model | - -Maple builds the transcript from these and labels each turn with the first line of the user's message. You can leave the Spring AI switches off; they only add log lines. - -Every `chat` span carries the whole conversation so far, so spans grow with long chats. Don't set `management.opentelemetry.tracing.limits.max-attribute-value-length`: a cut JSON value no longer parses, and Maple drops it. - -### Privacy: turn content off or redact it - -Set `maple.ai.capture-content=false` (the default in the class) and no message or tool content leaves the process. Sessions, turns, model names, tokens, tool names and failures still show up; the transcript is empty. To redact instead, change `message(...)` to mask what you don't want sent, for example email addresses in text parts. Keep personal data out of the conversation id and agent names; they are sent regardless of the setting. - -## Tools, errors and sub-agents - -Every tool call gets an `execute_tool ` span. Spring AI puts the name in `spring.ai.tool.definition.name`; the filter copies it to `gen_ai.tool.name`, which Maple reads for the tool pages. - -### Failed tools - -When a `@Tool` method throws, Spring AI's default `ToolExecutionExceptionProcessor` returns the exception message to the model as the tool result, and the `execute_tool` span ends with status OK. The model usually recovers gracefully, which is good for users and bad for debugging: Maple would count the call as a success. - -The processor bean in the configuration class calls `error()` on the running tool observation first. The span gets status `Error` with the exception message and an `exception` event, Maple counts it as a failed tool call, and the model still gets the message. Successful calls are untouched. - -Two things to know: - -- Defining the bean replaces Spring AI's, so `spring.ai.tools.throw-exception-on-error` no longer applies. To fail the whole call instead, build the fallback with `.alwaysThrow(true)`. -- If your app already defines a `ToolExecutionExceptionProcessor`, add the `error()` call to it instead of adding a second bean. - -A tool that returns an error string instead of throwing is a success as far as any tracer can tell. Throw if you want the failure counted. - -### Sub-agents as tools - -Spring AI has no agent class. The idiomatic multi-agent setup is one `ChatClient` per role, with the orchestrator calling the others through `@Tool` methods. Give each client its own agent name: - -```java -import org.springframework.ai.chat.client.ChatClient; -import org.springframework.ai.tool.annotation.Tool; -import org.springframework.stereotype.Component; - -@Component -public class Workers { - - private final ChatClient weather; - - public Workers(ChatClient.Builder builder, WeatherTools weatherTools) { - this.weather = builder.clone() - .defaultSystem("You answer weather questions using your tools.") - .defaultTools(weatherTools) - .defaultAdvisors(a -> a.param(MapleAiObservationConfig.AGENT_NAME, "weather_worker")) - .build(); - } - - @Tool(name = "weather_worker", description = "Ask the weather specialist about a city") - public String weatherWorker(String task) { - return weather.prompt().user(task).call().content(); - } - -} -``` - -In Maple, `execute_tool weather_worker` with the worker's `chat_client` span under it shows up as a delegation into a `weather_worker` lane, with the task as its input and the worker's answer as its output. A client without an agent name gets no lane; its model and tool calls are drawn in the caller's lane. +Use the conversation id your app already stores. Set it on the request, as above, and never with `defaultAdvisors` on the builder, which puts every user in one session. The parameter works without a chat memory advisor. Sub-agent calls inside tools don't need it, because they run in the same trace. -Spring AI runs the tool calls of one model response one after another, on the calling thread, so context flows without help. If you fan work out to your own executor, propagate the current observation to the worker threads with Micrometer's `context-propagation` library (`ContextSnapshot`, or a wrapped executor). Otherwise each worker starts a new trace, and since it carries no conversation id, Maple shows it as a separate session. +## Exit command-line apps explicitly -## Tokens and cost - -Every `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports them. Maple reads all four. The `chat_client` spans carry no usage, so nothing is counted twice. - -Streaming has one catch. OpenAI (and OpenAI-compatible gateways such as OpenRouter) only return usage on a stream when the request asks for it. Spring AI 2.0's OpenAI model asks by default, so streamed turns report tokens out of the box. That default disappears as soon as you set any `spring.ai.openai.chat.stream-options.*` property: then add `spring.ai.openai.chat.stream-options.include-usage=true` too, or streamed turns show zero tokens. On Spring AI 1.1, set `spring.ai.openai.chat.options.stream-usage=true`. - -Spring AI records the provider as `gen_ai.system`, derived from the client class rather than the model. Behind OpenRouter, a Claude model called through the OpenAI starter is labeled `openai`. Maple uses the provider only to decide whether cached tokens are included in the input count, so this matters only for cache figures. - -Spring AI emits no cost attribute, and Maple never prices tokens itself, so sessions show as unpriced. - -## Short-lived processes - -Spring Boot owns the `SdkTracerProvider` and shuts it down when the application context closes, which flushes the batch of pending spans. A web app needs nothing extra. For other shapes: - -- **`CommandLineRunner` apps and batch jobs.** Exit explicitly. The OpenAI starter's HTTP client keeps non-daemon threads alive for about 60 seconds after the last call, so a `main` that simply returns leaves the JVM, and the unflushed spans, waiting that long. `SpringApplication.exit` closes the context, which flushes: +A web app needs nothing extra: Boot flushes pending spans on shutdown. In a `CommandLineRunner` app, exit explicitly, or the OpenAI client's threads keep the JVM (and the unsent spans) waiting about 60 seconds: ```java public static void main(String[] args) { @@ -358,81 +227,25 @@ public static void main(String[] args) { } ``` -`kill -9` and `Runtime.halt()` lose the last batch. -- **Serverless (Spring Cloud Function on AWS Lambda and similar).** The runtime freezes the process between invocations, so flush before returning from each one. Don't shut the provider down. - -```java -import java.util.concurrent.TimeUnit; - -import io.opentelemetry.sdk.trace.SdkTracerProvider; - -// inject SdkTracerProvider, then at the end of each invocation: -tracerProvider.forceFlush().join(10, TimeUnit.SECONDS); -``` - -The exporter sends a batch every 5 seconds by default, so allow a few seconds after a conversation before looking for it in Maple. - -## Running the OpenTelemetry Java agent too - -Use one tracing setup per service. The OpenTelemetry Java agent does not turn Micrometer Observations into spans, so with the agent alone you get HTTP spans but no `chat_client`, `chat` or `execute_tool` span. You still need the starter and the configuration class from this guide. - -Adding the starter next to the agent is not enough, though. Boot then runs its own OpenTelemetry SDK, and the agent doesn't share context with it. In our test with agent 2.31.1, every Spring AI span arrived as its own trace: the model and tool calls lost their `chat_client` parent, so they fell out of their sessions. - -If the agent has to stay (it also instruments JDBC, Kafka and other libraries), give Micrometer the agent's `OpenTelemetry` instead of Boot's: - -```java -import io.opentelemetry.api.GlobalOpenTelemetry; -import io.opentelemetry.api.OpenTelemetry; - -@Bean -OpenTelemetry openTelemetry() { - return GlobalOpenTelemetry.get(); -} -``` - -Add this bean only where the agent is attached; without the agent, `GlobalOpenTelemetry.get()` returns a no-op and nothing is traced. With the bean, the agent exports every span, so configure the agent rather than Boot: - -- Set `OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev`, `OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer YOUR_INGEST_KEY`, `OTEL_SERVICE_NAME` and `OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production`. Boot's `management.opentelemetry.*` export properties and the sampling property no longer apply; the agent samples everything by default. -- Add `-Dotel.instrumentation.openai-java.enabled=false`. The agent ships instrumentation for the OpenAI Java SDK that Spring AI's OpenAI starter uses. It added no spans in our test with Spring AI 2.0.1, but it would duplicate every model call if it matched. -- HTTP client calls show up twice, once from Boot and once from the agent. They aren't AI spans, so sessions, counts and tokens are unaffected. +On serverless platforms, inject `SdkTracerProvider` and call `tracerProvider.forceFlush().join(10, TimeUnit.SECONDS)` at the end of each invocation. ## Check that it works -Run one conversation of two or three messages with the same conversation id, including one that calls a tool, plus a message with a second id. Wait about a minute, then open **Agent Sessions** in Maple. You should see: - -- **One session per conversation id**, with the framework shown as Spring AI and one turn per `ChatClient` call. A list of one-turn sessions named after trace ids means the `ChatMemory.CONVERSATION_ID` parameter is missing. -- **Spans named** `spring_ai chat_client` (as an agent span named after your `gen_ai.agent.name`), `chat openai/gpt-4o-mini` (your model id) and `execute_tool get_weather`, with `spring_ai advisor` spans in between. HTTP `POST` spans to the model provider appear muted next to them. -- **No spans named** `tool _calling `, `call`, `stream` or `message_chat_memory`. If they show up, the `ObservationFilter` bean isn't loaded. -- **The transcript**: each user message, the assistant's replies, and the tool calls with their arguments and results. -- **Tokens** on every model call, including streamed ones. -- **Sub-agents** as lanes named after each client's agent name, and a tool that threw counted as a failed tool call under its own name. -- **Cost** shown as unpriced. +Send two or three messages with the same conversation id, including one that calls a tool. After about a minute, **Agent Sessions** in Maple shows one session for that id with framework **Spring AI**, one turn per `ChatClient` call, the transcript, and tokens on every model call. Cost shows as unpriced, because Spring AI doesn't emit one. ## Troubleshooting -- **Only some turns arrive, or sessions have gaps.** Sampling is at Boot's default of 10%. Set `management.tracing.sampling.probability=1.0`. -- **Every message is its own session.** The `chat_client` span has no `spring.ai.chat.client.conversation.id`. Pass `.advisors(a -> a.param(ChatMemory.CONVERSATION_ID, id))` on every `prompt()` call. -- **All users land in one session.** The conversation id is a constant, usually set once with `defaultAdvisors` on the builder. Pass it per request. -- **Transcript is empty, but tokens and tools show up.** `maple.ai.capture-content` isn't `true`, or `MapleAiObservationConfig` isn't in a package Spring scans. `log-prompt` and `log-completion` don't help; they write to the log. -- **Twice as many model calls as expected, and a tool called `tool _calling ` in the tool list.** The advisor and `chat_client` spans are being counted. Make sure the `ObservationFilter` bean is loaded. -- **The streamed turn's model call is missing from its session.** Something drops Spring AI's advisor observations, usually an `ObservationPredicate`. Rename advisor spans as the filter does instead of skipping them. -- **Every span is its own trace.** The OpenTelemetry Java agent is attached next to the starter. See [Running the OpenTelemetry Java agent too](#running-the-opentelemetry-java-agent-too). -- **A command-line app hangs for a minute after its last reply.** The OpenAI client's threads keep the JVM alive. Exit with `System.exit(SpringApplication.exit(...))`. -- **A tool that threw shows as successful.** The custom `ToolExecutionExceptionProcessor` isn't active, or the app defines its own. Add the `registry.getCurrentObservation().error(exception)` call to the one that runs. -- **Streamed turns show zero tokens.** You set a `spring.ai.openai.chat.stream-options.*` property without `include-usage=true`. Add it. -- **A sub-agent shows up as its own session.** It ran on a thread without the caller's trace context. Propagate the observation to the executor, or run the sub-agent on the calling thread. -- **Tool names are missing on the tool pages.** The `ObservationFilter` isn't running; Spring AI alone emits only `spring.ai.tool.definition.name`. -- **Framework shows as Unidentified on some model spans.** With starters other than OpenAI, a `chat` span for a call without tools has no Spring AI marker, so Maple files it as a generic GenAI span. The session, transcript and tokens are unaffected. -- **Nothing arrives from tests.** `@SpringBootTest` turns tracing export off. Add `@AutoConfigureTracing` to the test class if you want spans from tests. -- **`Failed to publish metrics` warnings every minute.** The starter's metrics exporter is still pointed at `localhost:4318`. Set the two `management.otlp.metrics.export.*` lines or disable metrics export. -- **`401` from ingest.** The key is wrong or from the other region. The property must read `...headers.Authorization=Bearer YOUR_INGEST_KEY`. -- **Using LangChain4j instead of Spring AI.** This guide doesn't apply. Follow the [OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry). +- **Only some turns arrive.** Sampling is at Boot's default of 10%. Set `management.tracing.sampling.probability=1.0`. +- **Every message is its own session.** Pass `.advisors(a -> a.param(ChatMemory.CONVERSATION_ID, id))` on every `prompt()` call. +- **The transcript is empty.** `maple.ai.capture-content` isn't `true`, or `MapleAiObservationConfig` isn't in a scanned package. +- **Model calls are doubled and a tool named `tool _calling ` appears.** The `ObservationFilter` bean isn't loaded. +- **Every span is its own trace.** The OpenTelemetry Java agent is attached next to the starter. Remove it, or follow the Java agent steps in the [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai). ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) - [Agent tracing guides](/docs/agent-tracing) -- [Any language: the OpenTelemetry GenAI conventions](/docs/agent-tracing/opentelemetry) +- [Any language: the OpenTelemetry GenAI conventions](/docs/agent-tracing/opentelemetry), also for LangChain4j - [Spring AI observability reference](https://docs.spring.io/spring-ai/reference/observability/index.html) - [Spring AI tool calling](https://docs.spring.io/spring-ai/reference/api/tools.html) - [Spring Boot tracing reference](https://docs.spring.io/spring-boot/reference/actuator/tracing.html) diff --git a/apps/landing/src/content/docs/agent-tracing/strands.md b/apps/landing/src/content/docs/agent-tracing/strands.md index 3ea4589be9..83c81b3c18 100644 --- a/apps/landing/src/content/docs/agent-tracing/strands.md +++ b/apps/landing/src/content/docs/agent-tracing/strands.md @@ -1,17 +1,17 @@ --- title: "Trace Strands Agents with OpenTelemetry" -description: "Send Strands Agents traces to Maple with the prompt, reply and tool calls on every span, one session per conversation, and token counts that don't double." +description: "Send Strands Agents traces to Maple with the full transcript and one Agent Session per conversation." group: "AI Agents" order: 24 navLabel: "Strands Agents" icon: "strands" --- -Strands Agents ships its own OpenTelemetry tracer. Every `agent(...)` call becomes one trace with an `invoke_agent` span, an `execute_event_loop_cycle` span per reasoning step, a `chat` span per model call and an `execute_tool` span per tool call. Maple recognizes these spans as Strands without any extra instrumentation library. +Strands Agents ships its own OpenTelemetry tracer: every `agent(...)` call becomes one trace with `invoke_agent`, `chat` and `execute_tool` spans. Maple recognizes them without an extra instrumentation library. -The catch is where Strands puts the conversation. By default, prompts, replies and tool results are written as span events, and Maple reads span attributes only, so the transcript comes out empty even though tokens and tool calls show up. One environment variable fixes it. +By default Strands writes prompts and replies as span events, which Maple doesn't read, so you set one environment variable to move them onto span attributes. You also pass your conversation id as `session.id` on each agent. -This guide covers the Python SDK (`strands-agents` 1.54 or newer, tested on 1.57.1) and notes where the TypeScript SDK (`@strands-agents/sdk` 1.19) differs. +Tested with `strands-agents` 1.57.1 (1.54 or newer required) and the TypeScript SDK `@strands-agents/sdk` 1.19. ## Quick setup with a coding agent @@ -25,9 +25,9 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. -## Install Strands telemetry and export to Maple +## Install and configure the exporter The `otel` extra adds the OTLP/HTTP exporter. Add the extra for your model provider too (`openai`, `anthropic`, `litellm`; Bedrock needs none). @@ -35,7 +35,7 @@ The `otel` extra adds the OTLP/HTTP exporter. Add the extra for your model provi pip install 'strands-agents[otel,openai]>=1.57' ``` -Configure the exporter and the content settings with environment variables: +Set these environment variables: ```bash export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" @@ -46,15 +46,11 @@ export OTEL_RESOURCE_ATTRIBUTES="deployment.environment.name=production" export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only,gen_ai_use_latest_invocation_tokens" ``` -EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself, so set the base URL only. +For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. -The last line matters most. Each token does one job: +`OTEL_SEMCONV_STABILITY_OPT_IN` is the important one. Its three values switch to the current GenAI message format, write messages as span attributes (without this the transcript is empty), and make each `invoke_agent` span report only that call's tokens. Strands reads it once, when the first `Agent` is created, so set it in the environment rather than in code. -- `gen_ai_latest_experimental` switches Strands to the current GenAI conventions: messages as `gen_ai.input.messages` / `gen_ai.output.messages` in `{role, parts}` form, `gen_ai.system_instructions`, and tool arguments and results on `execute_tool` spans. -- `gen_ai_span_attributes_only` writes those messages as span attributes instead of span events. Without it, Maple shows tokens and tools but no transcript. It requires `strands-agents` 1.48 or newer. -- `gen_ai_use_latest_invocation_tokens` makes the `invoke_agent` span report this call's tokens instead of the agent's running total. See [Tokens and cost](#tokens-cost-and-the-agent-level-roll-up). - -Strands reads this variable once, when the first `Agent` is created. Set it in the environment (shell, `.env`, container config), not in code that runs after an agent exists. +To redact message content, append `gen_ai_unredacted_attributes=` with a `;`-separated allowlist of attributes to keep; everything else becomes `[REDACTED]`. Then start the tracer once, before your first agent runs: @@ -65,15 +61,11 @@ from strands.telemetry import StrandsTelemetry telemetry = StrandsTelemetry().setup_otlp_exporter() ``` -`StrandsTelemetry()` creates a `TracerProvider`, registers it globally and adds a `BatchSpanProcessor` with the stock OTLP/HTTP exporter, so all the standard `OTEL_EXPORTER_OTLP_*` variables apply. - -If your app already sets up OpenTelemetry (a global `TracerProvider` from your web framework, `opentelemetry-instrument`, or the ADOT distro on AgentCore), skip `StrandsTelemetry()` entirely. Strands uses whatever global provider exists; add a `BatchSpanProcessor(OTLPSpanExporter(...))` pointing at Maple to that provider instead of creating a second one. +If your app already sets up a global `TracerProvider` (your web framework, `opentelemetry-instrument`, or the ADOT distro on AgentCore), skip `StrandsTelemetry()`. Add a `BatchSpanProcessor(OTLPSpanExporter(...))` pointing at Maple to that provider instead. -## Group turns into one session +## Pass the conversation id as session.id -Each `agent(...)` call is its own trace, and Strands has no built-in conversation id on its spans. Without one, Maple shows every message as a separate one-turn session named after its trace id. - -Strands copies `trace_attributes` onto every span it creates for that agent. Maple reads `session.id` for Strands, so pass your conversation id there: +Strands copies `trace_attributes` onto every span of the agent, and Maple groups Strands traces by `session.id`: ```py from strands import Agent @@ -96,122 +88,15 @@ def handle_message(conversation_id: str, text: str) -> str: return str(agent(text)) ``` -This is the usual shape of a chat backend: one request per user message, history restored by the session manager (`FileSessionManager` here, `S3SessionManager` in production), and the same id used for history and for tracing. The session manager's `session_id` never reaches the spans on its own, so both arguments are needed. - -A few things to get right: - -- **Use the conversation id, not a per-request UUID.** A fresh id per call gives you one session per message again. -- **Don't share one module-level `Agent` across users.** Its `trace_attributes` would carry one user's id into everyone's traces. Create the agent per request, or per conversation. -- **Name every agent.** `name` becomes `gen_ai.agent.name`. Unnamed agents are all called `Strands Agents`, which merges sub-agents into one lane. - -Only `session.id` groups sessions for Strands. `gen_ai.conversation.id` is ignored on Python Strands spans, so there is no need to add it. - -## Record prompts, responses and tool calls +The session manager restores history but doesn't put its `session_id` on the spans, so you need both arguments. Use the conversation id your app already has, never a fresh UUID per request. -With the three opt-in tokens set, every span carries the conversation as JSON attributes: +Create the agent per request, as above. A shared module-level `Agent` would carry one user's id into everyone's traces. Give every agent a `name`: unnamed agents are all called `Strands Agents`, which merges sub-agents into one lane. -| Span | What Maple reads | -|---|---| -| `invoke_agent ` | this turn's user message, the final reply, `gen_ai.system_instructions` | -| `chat` | the full message history sent to the model, the model's reply with `finish_reason`, the system prompt | -| `execute_tool ` | `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result` (successful calls only) | +With `agent.as_tool()` sub-agents, only the orchestrator needs `session.id`, because the sub-agents run inside its trace. For a `Swarm`, pass `trace_attributes={"session.id": conversation_id}` to the `Swarm`. For a `Graph`, set `graph.trace_attributes = {"session.id": conversation_id}` after `builder.build()`, which doesn't forward them. -Maple builds the transcript from these, and labels each turn with the first line of the user's message. +## Flush in scripts and jobs -Leaving out `gen_ai_span_attributes_only` is the most common reason for an empty transcript. The spans look complete in a trace viewer that renders events, and Maple still shows nothing. - -### Privacy: redact or drop content - -Content capture is on by default in Strands; there is no single off switch. Redaction is controlled by one more token in the same variable, `gen_ai_unredacted_attributes=`, followed by a `;`-separated allowlist. Anything not listed is replaced with `[REDACTED]`: - -```bash -# Keep replies and tool results, redact user input and system prompts -export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only,gen_ai_use_latest_invocation_tokens,gen_ai_unredacted_attributes=gen_ai.output.*;gen_ai.tool.call.result" - -# Redact every message (empty allowlist) -export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only,gen_ai_use_latest_invocation_tokens,gen_ai_unredacted_attributes=" -``` - -The attributes covered are `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. Only a single trailing `*` works as a wildcard. Redacted values aren't JSON, so Maple leaves those parts of the transcript blank; tokens, tools and errors are unaffected. - -Tool descriptions and schemas (`gen_ai.tool.description`, `gen_ai.tool.json_schema`) and anything you put in `trace_attributes` are never redacted. Keep emails and customer names out of `trace_attributes`. - -## Tools, errors and sub-agents - -Every tool call gets an `execute_tool ` span with `gen_ai.tool.name`, `gen_ai.tool.call.id` and `gen_ai.tool.description`. - -When a tool raises, Strands catches the exception, feeds the error back to the model, and marks the span failed: status `ERROR` with the exception message (for example `transport data service unavailable (503)`) and `gen_ai.tool.status=error`. Maple counts it as a failed tool call and groups it with other failures of the same message on the tool pages. A tool that returns `{"status": "error", ...}` itself is marked the same way. You don't need to add anything. - -### Agents as tools - -`agent.as_tool()` wraps a sub-agent so an orchestrator can call it. The trace nests the way Maple expects: `execute_tool weather_worker` with the sub-agent's `invoke_agent weather_worker` span inside it, which Maple shows as a delegation with its own lane. - -```py -weather_worker = Agent(name="weather_worker", model=model, tools=[get_weather], callback_handler=None) -budget_worker = Agent(name="budget_worker", model=model, tools=[calculate], callback_handler=None) - -orchestrator = Agent( - name="orchestrator", - model=model, - tools=[ - weather_worker.as_tool(description="Weather for a city"), - budget_worker.as_tool(description="Travel budget arithmetic"), - ], - trace_attributes={"session.id": conversation_id}, - callback_handler=None, -) -orchestrator("Produce a mini briefing about Amsterdam: weather and a 3-day budget.") -``` - -Only the orchestrator needs `session.id`. Maple groups sessions per trace, and the sub-agents run inside the orchestrator's trace. Strands runs tool calls from one model response concurrently by default, so parallel sub-agents appear as overlapping sibling spans. - -### Graph and Swarm - -`Graph` and `Swarm` open their own root span (`invoke_graph` / `invoke_swarm`) with each node's `invoke_agent` span beneath it. The root carries the multi-agent object's `trace_attributes`, not the agents'. - -```py -from strands.multiagent import GraphBuilder, Swarm - -swarm = Swarm([researcher, writer], trace_attributes={"session.id": conversation_id}) - -builder = GraphBuilder() -builder.add_node(researcher, "research") -builder.add_node(writer, "write") -builder.add_edge("research", "write") -graph = builder.build() -graph.trace_attributes = {"session.id": conversation_id} # GraphBuilder has no setter for it -``` - -`GraphBuilder.build()` doesn't forward trace attributes, which is why the last line exists. Giving each node agent `trace_attributes={"session.id": ...}` works too, since one span per trace is enough. - -The graph node id is not exported. Maple identifies each node by its agent's `name`, so name node agents after what they do. - -### Human approval (interrupts) - -An interrupt raised from a `BeforeToolCallEvent` hook ends the current trace, and the resumed call starts a new one. Both carry the agent's `session.id`, so they stay in one session. The interrupted tool appears twice, once ended early (status OK, no result) and once with the real outcome; both spans share the same `gen_ai.tool.call.id`. The tool body runs once, but Maple counts both spans as tool calls, so an approved call shows as two calls, one of them without a result. - -## Tokens, cost and the agent-level roll-up - -Each `chat` span reports `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`. Input includes cached tokens, which matches how Maple reads them. Since 1.54, prompt-cache hits and writes use the names Maple reads (`gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_creation.input_tokens`). Older versions only emit `cache_read_input_tokens` / `cache_write_input_tokens`, which Maple ignores. - -Strands streams every model call internally and reads usage from the stream's final event, so streamed turns report tokens too. The OpenAI model provider requests `stream_options.include_usage` automatically. - -The `invoke_agent` span also carries usage, and this is where counts go wrong: - -- **Without `gen_ai_use_latest_invocation_tokens`,** it reports the agent instance's lifetime total. On a long-lived agent, turn 8 reports the sum of turns 1 to 8, and any backend that sums spans over-counts by several times. -- **With it,** the span reports only this call's tokens, which Maple's detail page recognizes as a roll-up of the `chat` spans below and doesn't count twice. - -A per-request agent (as in the session example above) never accumulates across turns, which is another reason to create agents per request. - -One Maple issue remains even with a correct setup: the **Agent Sessions** list currently shows about twice the real tokens for Strands sessions, in Python and TypeScript. The list only recognizes a roll-up whose model calls sit directly below it, and Strands puts an `execute_event_loop_cycle` span in between. The session's own page shows the right totals, the sum of the `chat` spans. - -Strands doesn't emit cost, so Maple shows sessions as unpriced. Maple never prices tokens itself. - -More gaps: Strands records time to first token as `gen_ai.server.time_to_first_token` in milliseconds, which Maple doesn't read, and `chat` spans carry no `gen_ai.response.id`. The finish reason is only inside the output messages, so Maple's reply-length check shows as skipped. The missing response id means Maple can't recognize the same call reported twice, so don't also enable a gateway export such as OpenRouter Broadcast for the same traffic. - -## Flush before short-lived processes exit - -Spans go through a `BatchSpanProcessor`, which exports every few seconds. A long-running server needs nothing extra. Scripts, CLIs, notebooks and jobs lose their last spans unless they flush: +Spans are exported in batches every few seconds. A long-running server needs nothing extra. Scripts, CLIs, notebooks and jobs lose their last spans unless they flush: ```py from telemetry import telemetry @@ -225,17 +110,15 @@ finally: On AWS Lambda, call `telemetry.tracer_provider.force_flush()` at the end of each invocation and don't call `shutdown()`, because the warm container reuses the provider. -A call cut off by `asyncio.wait_for` or task cancellation may export incomplete spans ([strands-agents/harness-sdk#3609](https://github.com/strands-agents/harness-sdk/issues/3609)). Ended spans still flush normally. +## TypeScript SDK -## TypeScript SDK differences - -The TypeScript SDK emits the same span tree and honours the same `OTEL_SEMCONV_STABILITY_OPT_IN` tokens, except `gen_ai_use_latest_invocation_tokens`, which it doesn't support. Its OpenTelemetry packages are optional peer dependencies, so install them explicitly: +The TypeScript SDK emits the same spans. Its OpenTelemetry packages are optional peer dependencies, so install them explicitly: ```bash npm install @strands-agents/sdk @opentelemetry/api @opentelemetry/sdk-trace-base @opentelemetry/sdk-trace-node @opentelemetry/resources @opentelemetry/exporter-trace-otlp-http @opentelemetry/sdk-metrics @opentelemetry/exporter-metrics-otlp-http ``` -Set `OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only"` and the same `OTEL_EXPORTER_OTLP_*` variables as above, then: +Use the same `OTEL_EXPORTER_OTLP_*` variables, with `OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_attributes_only"` (the SDK doesn't support the third value). Then: ```ts import { Agent, FileStorage, SessionManager } from "@strands-agents/sdk" @@ -246,7 +129,6 @@ const provider = setupTracer({ exporters: { otlp: true } }) // reads OTEL_EXPORT const model = new OpenAIModel({ modelId: "gpt-4o-mini", params: { max_tokens: 600 } }) -// One request per user message: build the agent, restore history, answer. async function handleMessage(conversationId: string, text: string): Promise { const agent = new Agent({ name: "support_agent", @@ -271,38 +153,21 @@ try { } ``` -Set both keys in `traceAttributes`. The TypeScript SDK names its tracer and `gen_ai.provider.name` after `OTEL_SERVICE_NAME`, so with a custom service name Maple can't tell the spans come from Strands. It files them as generic GenAI spans, which group by `gen_ai.conversation.id`, and shows the framework as "Unidentified". Transcript, tools, failures and tokens still work. The TypeScript SDK writes `traceAttributes` on the `invoke_agent` span only, which is enough, since Maple groups sessions per trace. - -Create the agent per request, as above. The TypeScript `invoke_agent` span always reports the agent instance's running total, so a long-lived agent reports turns 1 to N again on turn N and Maple's totals grow with every turn. A fresh agent per request, with history restored by `SessionManager`, reports only this turn. - -The exporter sends OTLP/HTTP JSON, which Maple ingest accepts. `setupTracer()` only flushes on Node's `beforeExit`, which never fires after `process.exit()`, so flush explicitly as above. Failed `Graph` nodes currently end with status OK ([harness-sdk#4166](https://github.com/strands-agents/harness-sdk/issues/4166)); tool failures are marked correctly. +Set both keys in `traceAttributes`. With a custom service name, Maple can't tell the spans come from Strands and groups them by `gen_ai.conversation.id`, showing the framework as "Unidentified". Create the agent per request here too: a reused TypeScript agent reports its running token total on every turn. ## Check that it works -Run one conversation of two or three messages, including one that calls a tool, then open **Agent Sessions** in Maple. Within a minute you should see: +Run a conversation of two or three messages, including one that calls a tool, then open **Agent Sessions** in Maple. You should see one session per conversation id with framework **Strands Agents**, one turn per `agent(...)` call, and a transcript with the user messages, replies and tool calls. -- **One session per conversation id**, with the framework shown as Strands Agents and one turn per `agent(...)` call. A list of one-turn sessions means `session.id` is missing. -- **The transcript**: each user message, the assistant's replies and the tool calls with their arguments and results. -- **Spans named** `invoke_agent support_agent` (your agent's `name`), `execute_event_loop_cycle`, `chat` and `execute_tool get_weather`. -- **Tokens** on every model call, including streamed ones (the session page has the right total; the list currently shows about twice that), and the model id you passed (for example `gpt-4o-mini`, or a Bedrock id such as `us.anthropic.claude-sonnet-4-5-20250929-v1:0`). -- **Sub-agents** as separate lanes named after each agent, with failed tool calls counted under the tool's name. -- **Cost** shown as unpriced. +Cost shows as unpriced, because Strands doesn't report it. The sessions list currently shows about twice the real token count for Strands; the session's own page has the correct total. ## Troubleshooting -- **Transcript is empty, but tokens and tools show up.** Content is still in span events. Add `gen_ai_span_attributes_only` (and `gen_ai_latest_experimental`) to `OTEL_SEMCONV_STABILITY_OPT_IN`, make sure it's set before the first `Agent` is created, and upgrade to 1.48 or newer. -- **Every message is its own session.** No `session.id` on the trace. Pass `trace_attributes={"session.id": conversation_id}` to the agent, or to the `Swarm` / `Graph` that runs it. Setting only the session manager's `session_id` isn't enough. -- **Two users' messages land in one session.** A shared `Agent` instance carries one `trace_attributes` dict for everyone. Create the agent per request. -- **Token totals look several times too high.** The `invoke_agent` span reports the agent's lifetime usage. Add `gen_ai_use_latest_invocation_tokens`, or create the agent per request. If only the sessions list shows about twice the session page's total, that's the known list issue from [Tokens, cost and the agent-level roll-up](#tokens-cost-and-the-agent-level-roll-up); trust the session page. -- **Cache tokens are zero with prompt caching on.** Versions before 1.54 use `cache_read_input_tokens`, which Maple doesn't read. Upgrade. -- **Every sub-agent is called "Strands Agents".** The agents have no `name`. Set `Agent(name=...)` on each one. -- **Graph session splits from the rest of the conversation.** `GraphBuilder.build()` drops trace attributes. Set `graph.trace_attributes` after building. -- **No model name on spans.** A custom `Model` subclass that only implements `get_config()` gets no `gen_ai.request.model` ([harness-sdk#4205](https://github.com/strands-agents/harness-sdk/issues/4205)). Give it a `config` dict with `model_id`. -- **Every model call appears twice.** Another instrumentation (OpenLIT, OpenLLMetry, OpenInference, an OpenAI or Bedrock GenAI instrumentor) is wrapping the same calls. Strands' own spans are enough; remove the other one. -- **Nothing arrives from a script.** The process exited before the batch exported. Call `force_flush()` and `shutdown()` in a `finally` block. -- **`401` from ingest.** The key is wrong or from the other region. The header must be written `Authorization=Bearer YOUR_INGEST_KEY` in `OTEL_EXPORTER_OTLP_HEADERS`. -- **TypeScript: framework shows "Unidentified" and sessions split.** Add `gen_ai.conversation.id` next to `session.id` in `traceAttributes`. -- **TypeScript: tokens grow with every turn.** A reused `Agent` reports its running total on `invoke_agent`. Create the agent per request and restore history with `SessionManager`. +- **Transcript is empty, but tokens and tools show up.** Add `gen_ai_span_attributes_only` and `gen_ai_latest_experimental` to `OTEL_SEMCONV_STABILITY_OPT_IN`, set before the first `Agent` is created. +- **Every message is its own session.** Pass `trace_attributes={"session.id": conversation_id}` to the agent, or to the `Swarm` or `Graph` that runs it. +- **Two users' messages land in one session.** A shared `Agent` carries one `trace_attributes` dict for everyone. Create the agent per request. +- **Every model call appears twice.** Another instrumentation (OpenLIT, OpenLLMetry, OpenInference, an OpenAI or Bedrock instrumentor) wraps the same calls. Remove it. +- **Nothing arrives from a script.** The process exited before the batch was exported. Call `force_flush()` and `shutdown()` in a `finally` block. ## Related diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md index f444b110c1..ee17288e01 100644 --- a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -1,17 +1,17 @@ --- title: "Trace Vercel AI SDK agents with OpenTelemetry" -description: "Send the AI SDK's OpenTelemetry spans to Maple so each chat conversation is one Agent Session with its transcript, tool calls, sub-agents and tokens, in Node.js or Next.js." +description: "Send the Vercel AI SDK's OpenTelemetry spans to Maple and group each chat into one Agent Session." group: "AI Agents" order: 10 navLabel: "Vercel AI SDK" icon: "vercel" --- -The Vercel AI SDK traces itself. Every `generateText`, `streamText` and `ToolLoopAgent` call emits an `invoke_agent` span, a `chat` span per model request with the prompt, the reply and the token counts, and an `execute_tool` span per tool call with its arguments and result. The spans follow the OpenTelemetry GenAI semantic conventions, prompts and replies are recorded by default, and Maple reads all of it without an extra instrumentation package. +The Vercel AI SDK emits OpenTelemetry GenAI spans for every `generateText`, `streamText` and `ToolLoopAgent` call, with the prompts, replies, tool calls and token counts. Maple reads them without an extra instrumentation package. -Two things go wrong by default. In AI SDK 7, nothing is traced until you call `registerTelemetry()` once at startup: without it the SDK emits zero spans, with no warning. And the AI SDK has no conversation id. Each call is its own trace with nothing linking it to the previous message, so a chat backend that handles one message per request shows up in Maple as one session per message. This guide fixes both. +You have to get two things right. In AI SDK 7 nothing is traced until you call `registerTelemetry()` at startup, and the SDK has no conversation id, so you pass one on every call or each message becomes its own session. -It covers AI SDK 7 (`ai` 7.0.106 or newer) on Node.js 22 or newer, in a plain Node.js service and in Next.js. It was tested with `ai` 7.0.118, `@ai-sdk/otel` 1.0.118, OpenTelemetry JS 0.222.0 and `@openrouter/ai-sdk-provider` 3.1.0 on Node.js 26. The Next.js setup uses `@vercel/otel` 2.1.3. AI SDK 5 and 6 work too, with less detail in the transcript; see [AI SDK 5 and 6](#ai-sdk-5-and-6). +Tested with `ai` 7.0.118, `@ai-sdk/otel` 1.0.118, OpenTelemetry JS 0.222.0 and `@vercel/otel` 2.1.3 on Node.js 26. You need `ai` 7.0.106 or newer and Node.js 22 or newer. On AI SDK 5 or 6, run `npx @ai-sdk/codemod v7` first; the skill has a fallback setup if you can't upgrade. ## Quick setup with a coding agent @@ -25,18 +25,18 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Use your key from **Settings → Ingestion**. Without one, the agent uses a placeholder you can replace later. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. -## Export AI SDK spans to Maple +## Install the packages -In AI SDK 7, OpenTelemetry support moved out of the `ai` package into `@ai-sdk/otel`. It creates spans; the OpenTelemetry SDK exports them. Install both: +In AI SDK 7, OpenTelemetry support lives in `@ai-sdk/otel`. It creates the spans, and the OpenTelemetry SDK exports them: ```bash npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/api @opentelemetry/sdk-node \ @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources ``` -Point the exporter at Maple with the standard OpenTelemetry variables: +## Point the exporter at Maple ```bash export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" @@ -46,7 +46,9 @@ export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. -Then set up tracing once, when your process starts: +## Register the AI SDK integration + +Create an `instrumentation.ts` and import it as the first line of your entry point (`import "./instrumentation"`): ```ts // instrumentation.ts @@ -82,15 +84,13 @@ registerTelemetry( ) ``` -Import it as the first line of your entry point (`import "./instrumentation"`). The AI SDK doesn't patch any modules, so what matters is that `registerTelemetry()` runs before the first AI SDK call. Call it exactly once: it appends to a global list, and every registered `OpenTelemetry` instance emits its own copy of every span. - -`usage: true` and `runtimeContext: true` add a few AI SDK-specific `ai.*` attributes next to the GenAI ones: the uncached and text token counts, and the conversation id you pass in. Maple identifies AI SDK spans by those `ai.*` keys on the `gen_ai` tracer. Without them, the `invoke_agent` and `chat` spans read as generic GenAI spans, the session's framework shows as **Unidentified**, and time to first token isn't picked up. +Call `registerTelemetry()` exactly once. Each call adds another integration, and each one emits its own copy of every span. Keep `usage: true` and `runtimeContext: true`: Maple uses the `ai.*` attributes they add to recognize AI SDK spans. -If your app already starts an OpenTelemetry SDK (auto-instrumentation, Sentry, your own `NodeTracerProvider`), don't start a second one. Add the `BatchSpanProcessor` above to the existing provider and keep the `registerTelemetry()` call. `@ai-sdk/otel` uses the global tracer provider. Don't pass it a `tracer` from another tracer name either: Maple expects the default `gen_ai` scope. +If your app already starts an OpenTelemetry SDK (auto-instrumentation, Sentry, your own `NodeTracerProvider`), don't start a second one. Add the `BatchSpanProcessor` to the existing provider and keep the `registerTelemetry()` call. ### Next.js -Next.js calls `register()` in `instrumentation.ts` once per server runtime. Use `@vercel/otel` there, as in the [Next.js guide](/docs/guides/instrumentation-nextjs), and register the AI SDK integration next to it: +In Next.js, register both in the `register()` function of `instrumentation.ts`, using `@vercel/otel` as in the [Next.js guide](/docs/guides/instrumentation-nextjs): ```ts // instrumentation.ts (project root, or src/instrumentation.ts) @@ -120,15 +120,11 @@ export function register() { } ``` -The AI SDK spans then nest under the request span Next.js creates for the route handler, like `POST /api/chat`. - -`@vercel/otel` ends every span of a trace that is still open when the trace's root span ends. AI SDK work that keeps running after the request span has ended, such as a stream read inside `after()`, loses the end of its spans: the reply and the token counts. Read the stream inside the request, for example by returning it as the response with `toUIMessageStreamResponse()` or `createAgentUIStreamResponse()`. +Return AI SDK streams as the response (`toUIMessageStreamResponse()` or `createAgentUIStreamResponse()`). `@vercel/otel` ends all open spans when the request ends, so a stream read later, for example in `after()`, loses its reply and token counts. -## Group every turn of a conversation into one session +## Pass the conversation id on every call -Maple groups traces into sessions by `gen_ai.conversation.id`. The AI SDK never sets it: there is no session or thread concept in `generateText`, and `ToolLoopAgent` doesn't keep one either. What it does have is `runtimeContext`, a per-call object your code passes in. The `enrichSpan` callback above copies `runtimeContext.conversationId` onto every span as `gen_ai.conversation.id`. - -Runtime context stays out of telemetry unless you list the key in `telemetry.includeRuntimeContext`, so each call needs both: +Maple groups traces into sessions by `gen_ai.conversation.id`. The `enrichSpan` callback above copies it from `runtimeContext.conversationId`, which only reaches telemetry if you also list it in `includeRuntimeContext`: ```ts import { streamText } from "ai" @@ -137,50 +133,33 @@ const result = streamText({ model, messages, runtimeContext: { conversationId: chatId }, - telemetry: { - functionId: "support_agent", - includeRuntimeContext: { conversationId: true }, - }, + telemetry: { functionId: "support_agent", includeRuntimeContext: { conversationId: true } }, }) ``` -Use the chat or thread id your app already stores. It must be stable for the whole conversation and different between conversations. A per-process constant merges every user into one session. - -If you skip this, each request shows up in **Agent Sessions** as its own one-turn session named `trace:`. +Use the chat or thread id your app already stores. It must stay the same for the whole conversation and differ between conversations. `functionId` becomes the agent name in Maple, so give each agent a distinct one. -### ToolLoopAgent - -`agent.generate()` and `agent.stream()` don't take `runtimeContext` per call. Declare the id as a call option and turn it into runtime context in `prepareCall`: +`ToolLoopAgent` doesn't take `runtimeContext` per call. Declare the id as a call option and turn it into runtime context in `prepareCall`: ```ts // agent.ts -import { openai } from "@ai-sdk/openai" import { ToolLoopAgent } from "ai" import { z } from "zod" export const assistant = new ToolLoopAgent({ - model: openai("gpt-4o-mini"), - instructions: "You are a helpful assistant.", - tools: { get_weather: getWeather, fetch_transport_data: fetchTransportData }, + // ...model, instructions, tools callOptionsSchema: z.object({ conversationId: z.string() }), prepareCall: ({ options, ...rest }) => ({ ...rest, runtimeContext: { conversationId: options.conversationId }, }), - telemetry: { - functionId: "support_agent", - includeRuntimeContext: { conversationId: true }, - }, + telemetry: { functionId: "support_agent", includeRuntimeContext: { conversationId: true } }, }) const result = await assistant.generate({ messages, options: { conversationId: chatId } }) ``` -`functionId` becomes `gen_ai.agent.name`. The agent's `id` isn't exported to telemetry, so set `functionId` on every agent you want to see by name. - -### Chat routes with useChat - -`useChat` already sends a stable chat id with every request, as `id` in the JSON body. Pass it through: +With `useChat`, the request body already carries a stable chat id as `id`. Pass it through: ```ts // app/api/chat/route.ts @@ -198,209 +177,29 @@ export async function POST(req: Request) { } ``` -With `streamText` directly, pass `runtimeContext: { conversationId: id }` as in the first example and return `result.toUIMessageStreamResponse()`. - -Tool approvals work the same way. In AI SDK 7 you mark a tool with `toolApproval: { delete_file: "user-approval" }` on the call or the agent (`needsApproval` on the tool is deprecated). The call then ends with a `tool-approval-request`, and resuming after the user answers is a new `generate()` or `stream()` call, so a new trace. Pass the same conversation id to both and they land in the same session as consecutive turns: one approval shows as two turns, the one that asked and the one that ran the tool. The approved `execute_tool` span sits directly under `invoke_agent`, not under a `step`, because the tool runs before the resumed call's first step. - -## Record prompts, responses and tool calls - -Content capture is on by default. With the setup above: - -- `invoke_agent` spans carry the call's `gen_ai.system_instructions`, `gen_ai.input.messages` and `gen_ai.output.messages`; -- each `chat` span carries the messages sent to the model and its reply, plus `gen_ai.tool.definitions`; -- each `execute_tool` span carries `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`. - -All of it is JSON in the GenAI message format, which Maple renders as the transcript. - -To keep content out of Maple, turn it off per call or per agent: - -```ts -telemetry: { - functionId: "billing_agent", - recordInputs: false, // no prompts, system instructions, tool definitions or tool arguments - recordOutputs: false, // no replies or tool results -}, -``` - -Sessions keep their turns, models, tool names, tokens and errors, but the transcript is empty. `telemetry: { isEnabled: false }` drops the call's spans entirely. - -Two more things to know: - -- **Files and images are recorded inline.** An image or PDF in the messages is written as base64 into the `chat` span, and every later step of the loop repeats the history. Keep attachments out of traced calls or set `recordInputs: false` on them. -- **Content is not redacted.** Anything a user types reaches Maple. For pattern-based redaction, run an OpenTelemetry Collector between your app and Maple, or turn recording off for the calls that handle sensitive data. - -`runtimeContext` values are only exported when listed in `includeRuntimeContext`, so user ids, tokens or tenant data you keep there for your tools stay out of the spans. - -## Tools, errors and sub-agents - -Every tool call is an `execute_tool ` span with `gen_ai.tool.name`, the model's `gen_ai.tool.call.id`, and the arguments and result. Maple matches each call to the model reply that requested it by that id. - -When a tool's `execute` throws, the AI SDK marks the span ERROR with the error message as the status message, records an `exception` event, and passes the error back to the model as a `tool-error` result. The loop continues, and Maple counts the call as failed: - -```ts -import { tool } from "ai" -import { z } from "zod" - -const fetchTransportData = tool({ - description: "Fetch live public-transport data for a city", - inputSchema: z.object({ city: z.string() }), - execute: async ({ city }): Promise => { - throw new Error(`transport data service unavailable (503) for ${city}`) - }, -}) -``` - -A tool that returns an error value (`return { error: "..." }`) keeps the span green, and Maple counts the call as a success. Throw instead. - -The explicit `Promise` return type matters in TypeScript: without it, an `execute` that always throws infers `never` and the tool fails to typecheck. - -### Sub-agents - -The common multi-agent pattern is agents as tools: the orchestrator has a tool whose `execute` runs another agent. The worker's spans nest under that tool span in the same trace: - -```ts -const weatherWorker = new ToolLoopAgent({ - model: openai("gpt-4o-mini"), - tools: { get_weather: getWeather }, - telemetry: { functionId: "weather_worker" }, -}) - -export const orchestrator = new ToolLoopAgent({ - model: openai("gpt-4o-mini"), - tools: { - delegate_weather: tool({ - description: "Ask the weather worker", - inputSchema: z.object({ task: z.string() }), - execute: async ({ task }) => (await weatherWorker.generate({ prompt: task })).text, - }), - }, - callOptionsSchema: z.object({ conversationId: z.string() }), - prepareCall: ({ options, ...rest }) => ({ - ...rest, - runtimeContext: { conversationId: options.conversationId }, - }), - telemetry: { functionId: "orchestrator", includeRuntimeContext: { conversationId: true } }, -}) -``` - -The worker doesn't need the conversation id: Maple needs it on one span per trace, and the orchestrator's spans have it. It does need its own `functionId`. Maple draws a lane per `gen_ai.agent.name`, and an `execute_tool delegate_weather` span whose only child is `invoke_agent` for `weather_worker` shows as a delegation, with the tool's arguments and result as the lane's input and output. When the model asks for several delegate tools in one reply, the AI SDK runs them concurrently and the lanes overlap in time. - -A pipeline of separate top-level calls (an orchestrator, then a summary agent) produces one trace per call. Pass the same conversation id to each, and they land in one session as consecutive turns. - -## Tokens and cost - -Each `chat` span carries `gen_ai.usage.input_tokens` and `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.cache_creation.input_tokens` when the provider reports caching. With `usage: true`, reasoning tokens are added as `ai.usage.outputTokenDetails.reasoningTokens`. Maple reads all of them. - -The `invoke_agent` span repeats the total of its own `chat` spans, two levels up (`invoke_agent` → `step` → `chat`). Maple doesn't net that repeat out yet, so a session's token total, in the session list and on the session page, is twice what the model calls used. The token counts on each `chat` span and the per-model breakdown on the session page are correct. Use those until this is fixed. - -A sub-agent's tokens stay on the sub-agent's spans: the orchestrator's `invoke_agent` repeats only its own model calls, not the workers'. - -Streaming needs nothing extra. The AI SDK normalizes provider usage, so `streamText` and `agent.stream()` calls have token counts too, and the streamed `chat` span records time to first chunk, which Maple shows as TTFT. - -Cost shows as unpriced. The AI SDK doesn't price calls, and Maple never prices tokens itself. Tokens, models and call counts are complete. If your model calls go through OpenRouter, its [Broadcast traces](/docs/agent-tracing/openrouter) carry the cost of each call, and Maple matches them to the AI SDK's `chat` spans by response id. - -## Short-lived processes - -`BatchSpanProcessor` exports every few seconds. A script, a CLI or a serverless function can end before that. Flush before exiting: - -```ts -import { sdk } from "./instrumentation" - -try { - await runConversation() -} finally { - await sdk.shutdown() // flushes, then stops the SDK -} -``` - -In a serverless handler that is reused between invocations, flush without stopping the SDK: - -```ts -import { spanProcessor } from "./instrumentation" - -export async function handler(event: { chatId: string; text: string }) { - try { - return await turn(event.chatId, event.text) - } finally { - await spanProcessor.forceFlush() - } -} -``` - -Streaming matters here. The `invoke_agent` span ends when the stream is fully read, not when `streamText` returns. In a script, read the stream to the end (`for await (const chunk of result.textStream)` or `await result.consumeStream()`) before flushing. A stream nobody reads never ends its spans, and they are never exported. - -On Vercel, `@vercel/otel` flushes at the end of each request through `waitUntil`. Work that runs outside a request, such as a queue consumer, a cron job or a workflow step, doesn't get that flush. Call `forceFlush()` on your span processor at the end of each unit of work there. - -## AI SDK 5 and 6 - -Before AI SDK 7, OpenTelemetry was built into the `ai` package and off by default. Turn it on per call with `experimental_telemetry`, and pass the conversation id as metadata: +Sub-agents called from a tool's `execute` run inside the caller's trace, so they don't need the id. They do need their own `functionId` to get their own lane. -```ts -const result = await generateText({ - model, - messages, - experimental_telemetry: { - isEnabled: true, - functionId: "support_agent", - metadata: { conversationId: chatId }, - }, -}) -``` +Prompts and replies are recorded by default. To keep them out of Maple for a call or agent, set `recordInputs: false` and `recordOutputs: false` in its `telemetry`. -Those versions emit the older span format: `ai.generateText` or `ai.streamText` for the call, `ai.generateText.doGenerate` or `ai.streamText.doStream` for each model request, and `ai.toolCall` for each tool call. The metadata lands as `ai.telemetry.metadata.conversationId`, which Maple doesn't read, so copy it to `gen_ai.conversation.id` with a span processor: +## Flush in scripts and serverless functions -```ts -import type { Context } from "@opentelemetry/api" -import type { ReadableSpan, Span, SpanProcessor } from "@opentelemetry/sdk-trace-base" - -export class ConversationIdProcessor implements SpanProcessor { - onStart(span: Span, _parentContext: Context) { - const id = span.attributes["ai.telemetry.metadata.conversationId"] - if (typeof id === "string") span.setAttribute("gen_ai.conversation.id", id) - } - onEnd(_span: ReadableSpan) {} - forceFlush() { - return Promise.resolve() - } - shutdown() { - return Promise.resolve() - } -} -``` +`BatchSpanProcessor` exports every few seconds, so a short-lived process can exit first. In a script, call `await sdk.shutdown()` in a `finally` block before exiting. In a serverless handler that is reused between invocations, call `await spanProcessor.forceFlush()` in a `finally` instead, so the SDK keeps running. -Add it before the exporting processor: `spanProcessors: [new ConversationIdProcessor(), spanProcessor]`. - -Maple reads the old format's models, tokens, prompts and tool calls. The assistant's reply is only recorded as plain text (`ai.response.text`), which Maple doesn't render, so transcripts show the user and tool messages without the final answer. Upgrading to AI SDK 7 fixes that. The `npx @ai-sdk/codemod v7` migration renames `experimental_telemetry` to `telemetry`, but you still need to install `@ai-sdk/otel`, call `registerTelemetry()`, and move the id from `metadata` (gone in AI SDK 7) to `runtimeContext` yourself. Drop the span processor once you have. +Read streams to the end (`await result.consumeStream()`) before flushing: a stream's spans only end when it has been read. On Vercel, `@vercel/otel` flushes after each request, but queue consumers and cron jobs need their own `forceFlush()`. ## Check that it works -Run one conversation with at least two messages and a tool call, then open **Agent Sessions** in Maple. Spans take a few seconds to arrive. You should see: - -- one session per conversation, with the id you passed as `conversationId`, and framework **Vercel AI SDK**; -- one turn per `generate()` or `stream()` call, each labeled with the user's message, and a transcript with the prompts, replies and tool calls; -- per turn, an `invoke_agent ` span, a `step ` span per loop iteration, a `chat ` span per model request and an `execute_tool ` span per tool call; -- the agent name from `functionId` in the agent filter, and a lane per sub-agent; -- token counts on every model call, including streamed ones, and TTFT on streamed calls (the session's token total reads double; see [Tokens and cost](#tokens-and-cost)); -- failed tool calls marked as failed, with the error message; -- cost shown as unpriced. +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session named after your conversation id with framework **Vercel AI SDK**, one turn per call, and a transcript with the prompts, replies and tool calls. -Span names carry the model id, not the agent name, so two agents on the same model have identically named `invoke_agent` spans. The agent name is on the span as `gen_ai.agent.name`. +Two things look off but are expected. Cost shows as unpriced, because the AI SDK doesn't report it. The session's token total is currently twice what the model calls used; the per-model breakdown on the session page is correct. ## Troubleshooting -- **No AI spans at all.** `registerTelemetry()` was never called, or ran in a module that isn't loaded. AI SDK 7 emits nothing without it, and `experimental_telemetry: { isEnabled: true }` alone does nothing. Import `@ai-sdk/otel` and register `new OpenTelemetry()` at startup. -- **Every message is its own session.** No `gen_ai.conversation.id` on the spans. Check all three parts: `enrichSpan` on the integration, `runtimeContext: { conversationId }` on the call (or `prepareCall` for an agent), and `includeRuntimeContext: { conversationId: true }`. Without the last one, `enrichSpan` receives an empty object. -- **Sessions show framework "Unidentified".** The spans carry no `ai.*` attributes. Set `usage: true` and `runtimeContext: true` on `new OpenTelemetry()`, and don't pass a custom `tracer` with another name. -- **The session's token total is twice the sum of its model calls.** Expected for now: Maple counts the total on each `invoke_agent` span on top of its `chat` spans. The per-call counts and the per-model breakdown are correct. If the spans themselves are duplicated, see the next item. -- **Every span shows up twice.** `registerTelemetry()` ran twice, both `OpenTelemetry` and `LegacyOpenTelemetry` are registered, or a second OpenTelemetry SDK exports the same spans (Sentry without `skipOpenTelemetrySetup: true` next to `@vercel/otel`, for example). Register one integration and one exporter. -- **One call has no spans while others do.** It passes `telemetry.integrations`, which replaces the globally registered integrations for that call. Add the `OpenTelemetry` instance to that list, or remove the option. -- **Nothing arrives from a script or function.** The process ended before the batch was exported. Call `sdk.shutdown()` or `spanProcessor.forceFlush()` in a `finally`. -- **A streamed turn is missing, or its spans have no tokens.** The stream was never read to the end, so its spans never ended. Consume the stream before flushing. Before `ai` 7.0.106, a provider error in the middle of a stream also left the spans open; upgrade. -- **Sub-agents share one lane, or have none.** The worker agents have no `functionId`. Set a distinct one on each. -- **A failed tool shows as successful.** The tool returned an error value. Throw from `execute`. -- **Spans arrive with no prompts or replies.** `recordInputs: false` or `recordOutputs: false` is set on that call or agent. -- **Exports fail with 413.** Images or documents in the messages are recorded as base64 on every step. Keep them out of traced calls or set `recordInputs: false` there. -- **Transcripts end at the tool results, with no final answer.** You are on AI SDK 5 or 6, which records the reply as plain text. Upgrade to AI SDK 7. +- **No AI spans at all.** `registerTelemetry()` never ran. `experimental_telemetry: { isEnabled: true }` alone does nothing in AI SDK 7. +- **Every message is its own session.** Check all three parts: `enrichSpan`, `runtimeContext: { conversationId }` on the call (or `prepareCall`), and `includeRuntimeContext: { conversationId: true }`. +- **Framework shows "Unidentified".** Set `usage: true` and `runtimeContext: true` on `new OpenTelemetry()`. +- **Every span shows up twice.** `registerTelemetry()` ran twice, or a second OpenTelemetry SDK (such as Sentry without `skipOpenTelemetrySetup: true`) exports the same spans. +- **A failed tool shows as successful.** The tool returned an error value. Throw from `execute` instead. ## Related diff --git a/skills/maple-agent-tracing-agno/SKILL.md b/skills/maple-agent-tracing-agno/SKILL.md index eb0b9f622d..83cceeb728 100644 --- a/skills/maple-agent-tracing-agno/SKILL.md +++ b/skills/maple-agent-tracing-agno/SKILL.md @@ -169,6 +169,20 @@ Run one real conversation: 2-3 turns with the same `session_id` including one to Raw span check (optional, e.g. with a console exporter in a scratch run): run spans `.run` have `session.id` + `gen_ai.operation.name=invoke_agent` + `gen_ai.agent.name`; model spans have `gen_ai.operation.name=chat`, `gen_ai.input.messages`, `gen_ai.usage.input_tokens`; tool spans have `gen_ai.operation.name=execute_tool`. +## Known behaviours (expected; explain if the user asks) + +- If `setup_tracing()`/`AgentOS(tracing=True)` ran first, `set_tracer_provider` in `tracing.py` logs "Overriding of current TracerProvider is not allowed" and spans go only to the AgentOS database. A second `instrument()` call logs "Attempting to instrument while already instrumented" and its `config` is ignored. +- The instrumentor patches Agno's run functions and every model class in `agno.models`, so agents created after `tracing.py` runs are traced with no further changes. +- Without `enable_genai_semconv`, spans carry only OpenInference attributes (`llm.input_messages.0.message.content`, `llm.token_count.prompt`); the sessions list still shows tokens/models, the detail page doesn't decode them for Agno. +- With `enable_genai_semconv=True` the root span also carries `gen_ai.conversation.id` = `session.id`; Maple ignores it for Agno. +- Maple reads span attributes only; the instrumentor emits no span events or OTLP logs, so nothing else needs enabling. Masked values are replaced with `__REDACTED__` in-process before export. +- Team trace shape (one trace per team run): member runs sit directly under the leader's run, next to (not inside) the `delegate_task_to_member` tool spans; those show as ordinary tool calls on the leader with member id and task as arguments. With `team.arun()` in `coordinate` mode, members called in one step run concurrently and their spans overlap; sync `team.run()` runs them sequentially. +- Team context leak is upstream issue agno#5573. "Failed to detach context" in the logs after a streamed team run is agno#5208: log noise, spans still export. +- Human-in-the-loop: the paused run (`.run`) and the resumed run (`.continue_run`) are two traces and two turns in one session; the approved tool call appears once, in the second. Instrumentor <1.0.8 doesn't wrap `continue_run()`, so resumed runs appear as loose model/tool calls with no session. +- Tokens: every model span has input/output tokens plus cache read/write when the provider reports them; streamed runs record usage from the final chunk. The run span has no tokens, so nothing is double counted. Reasoning tokens aren't broken out. Maple links tool results to calls through the message history since tool spans lack `gen_ai.tool.call.id`, and can't show streaming latency (no TTFT). +- Cost: OpenRouter's price is recorded as `llm.cost.total` (USD, instrumentor >=1.0.10). Providers called directly (OpenAI, Anthropic) return no price, so sessions are unpriced. +- A tool that raises: Agno catches it and hands the message to the model; the tool span is `ERROR`, the run span stays OK. Maple groups repeated failures by the exception message. + ## Do not - Do not rely on `setup_tracing()` or `AgentOS(tracing=True)` to reach Maple; they only write to the database. diff --git a/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md b/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md index 450a40ddde..0d08d1feaf 100644 --- a/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md +++ b/skills/maple-agent-tracing-claude-agent-sdk/SKILL.md @@ -205,6 +205,21 @@ Then in Maple → Agent Sessions (`https://app.maple.dev/agent-sessions`, EU `ap If spans exist but the turn nests under an unrelated trace, an inherited `TRACEPARENT` survived: fix the env stripping. +## Reference notes (for explaining results to the user) + +- Span map: `claude_code.interaction` = a turn titled with the prompt; `claude_code.llm_request` = model call (model, `input_tokens`, `output_tokens`, `cache_read_tokens`, `cache_creation_tokens`, `ttft_ms`, stop reason, failure); `claude_code.tool` = tool call; `claude_code.tool.blocked_on_user` / `claude_code.tool.execution` = phases shown in the trace, not counted as calls. +- Anthropic's `input_tokens` excludes both cache buckets; Maple counts it that way (total input = sum of the three). The CLI always streams and still records usage. +- Model id is shown as Claude Code sent it (e.g. `anthropic/claude-haiku-4.5` through OpenRouter). +- A failed model request (`success=false` on `claude_code.llm_request`, with `status_code` and `error`) counts as a failed LLM call. Without `OTEL_LOG_TOOL_DETAILS=1` a failed tool's error is only the `error_class` (`McpToolCallError` for SDK tools). +- A call rejected by the user or `canUseTool`: `blocked_on_user` span with `decision=reject` and no execution; Maple shows a call without a result, not a failure. +- An active app span is passed to the CLI as `TRACEPARENT` by both SDKs (the turn nests under the request that triggered it; that's desired). Interactive `claude` ignores inbound `TRACEPARENT`; only SDK and `claude -p` runs read it. +- Content truncation: 60 KB per attribute (`CLAUDE_CODE_OTEL_CONTENT_MAX_LENGTH`), `[TRUNCATED ...]` marker. +- Terminal sessions signed in with a Claude account carry `user.email` on every span/event. To drop or mask attributes, route through an OpenTelemetry Collector with a `redaction` or `attributes` processor. +- Content never in session views: assistant replies (`assistant_response` log event), the system prompt, arguments of non-Bash/non-file tools (MCP and SDK tool args are only in the `tool_result` log event). +- Cost: only on the `claude_code.api_request` log event (`cost_usd`, with `session.id`) and the `claude_code.cost.usage` metric; both are Claude Code's client-side estimate at list price unless managed settings set `modelPricing`. In SDK apps, the result message's `total_cost_usd` is the same estimate per `query()`. With logs on, Logs has one `claude_code.user_prompt`, one `claude_code.api_request` per model call and one `claude_code.tool_result` per tool run. +- `/status` in `claude` lists telemetry variables it ignored (repo settings). 401s or data going elsewhere: managed settings or `~/.claude/remote-settings.json` set endpoint/headers and win; the user must ask whoever manages Claude Code. +- For `claude -p` in CI, set the same two export intervals. + ## Do not - Do not omit `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`: zero spans without it (metrics/logs still flow, which hides the problem). diff --git a/skills/maple-agent-tracing-crewai/SKILL.md b/skills/maple-agent-tracing-crewai/SKILL.md index 55831219f5..176aa93391 100644 --- a/skills/maple-agent-tracing-crewai/SKILL.md +++ b/skills/maple-agent-tracing-crewai/SKILL.md @@ -203,6 +203,15 @@ Run one real conversation (2-3 messages, same conversation id, at least one tool - Cost: unpriced unless models go through LiteLLM (which records `llm.cost.total`). Expected. - No spans with scope `crewai.telemetry`, no `coding_agent` attribute (telemetry is off). +More details (from the human guide, for edge cases): + +- Tokens: model spans carry `gen_ai.usage.input_tokens`/`output_tokens` (plus cached and reasoning tokens when reported); crew and agent spans carry none, so nothing double-counts. CrewAI's OpenAI provider always requests `stream_options={"include_usage": True}` when streaming, so streamed calls keep tokens. +- Model = the one the provider returned (e.g. `anthropic/claude-haiku-4.5` behind OpenRouter); provider = the SDK used, so every OpenRouter model shows `openai`. +- `memory=True` / `planning=True` add real, billed model calls (memory analysis, embeddings, planning agent); they appear in the session. Expected. +- The Arize Phoenix CrewAI page still recommends the LiteLLM instrumentor; ignore it for native providers (it records nothing there). +- Flow span layout: `.kickoff` root, one `.` span per `@start`/`@listen`/`@router` method, crews and `Agent.kickoff()` nested inside. Conversational flow turns show `.route_conversation` and `.converse_turn` under the kickoff. +- Why the streaming wrapper works: `gen_ai.operation.name=invoke_agent` makes Maple treat it as the turn's agent span, so the two crew kickoff spans under it are one turn. + If sessions are split per message: `using_session` missing or id changing. No model spans/tokens: wrong or missing SDK instrumentor. Every call its own trace: `akickoff`. Empty session page details: `enable_genai_semconv` not applied to that instrumentor. Nothing arrives: exporter endpoint/header wrong, `OTEL_SDK_DISABLED=true`, or process exited without flushing. ## Do not diff --git a/skills/maple-agent-tracing-dspy/SKILL.md b/skills/maple-agent-tracing-dspy/SKILL.md index 324197408c..1aa1bed823 100644 --- a/skills/maple-agent-tracing-dspy/SKILL.md +++ b/skills/maple-agent-tracing-dspy/SKILL.md @@ -249,6 +249,21 @@ Run one conversation of 2-3 messages with the same conversation id (one using a If a check fails, see the troubleshooting list in the human guide. +## Known behaviours (expected, nothing to fix; explain them if the user asks) + +- `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` set before `instrument()` runs is equivalent to `TraceConfig(enable_genai_semconv=True)`. The GenAI dual-write never overwrites a key already set, so the callback's values win. +- Each `LM.__call__` sits under `Predict.forward`, `Predict(StringSignature).forward` and `ChatAdapter.__call__`; those are DSPy steps, not extra model calls (the callback marks the adapter span `invoke_workflow` so its name isn't read as a model call). +- A `dspy.History` input is expanded into earlier user/assistant messages, so each model call repeats the conversation so far. +- `dspy.ReAct` catches tool exceptions and hands `Execution error in : ...` back to the model. The tool span is `ERROR` and counted as failed; the `ReAct.forward` span and the user's module span stay `OK`, and the program's return value doesn't reveal the failure. +- Tool spans have no `gen_ai.tool.call.id`: ReAct asks the model for the next tool as text fields (`next_tool_name`, `next_tool_args`), not through the provider's tool-calling API. +- When `ChatAdapter` can't parse a reply, DSPy retries with `JSONAdapter`: one `Predict` span holds two adapter spans, each with its own `LM.__call__`. Both calls happened and both are billed. +- Cost is DSPy's estimate from each history entry's `cost`: on the `lm15` engine from DSPy's bundled model metadata, on the LiteLLM engine LiteLLM's `response_cost`. Maple never prices tokens; a model DSPy can't price shows as **unpriced**. +- An `LM` with a custom `engine=` reports whatever usage that engine puts on its response. +- Anthropic models run on the LiteLLM engine by default (as does `dspy.LM(..., engine="litellm")`); that is where an extra LiteLLM/OpenAI instrumentor would double-count. +- `dspy.Parallel` copies only DSPy's settings into its worker threads, not the OpenTelemetry context; `ThreadingInstrumentor` is what carries it. Without it, a test fan-out became four traces with most spans orphaned. +- Narrower content switches (`OPENINFERENCE_HIDE_INPUT_TEXT`, `OPENINFERENCE_HIDE_OUTPUT_TEXT`, `OPENINFERENCE_HIDE_LLM_INVOCATION_PARAMETERS`) redact parts of each model message; the callback's agent messages follow only `OPENINFERENCE_HIDE_INPUTS` / `OPENINFERENCE_HIDE_OUTPUTS`. +- With content hidden the session still shows turns, model and tool calls, tokens and failures, with an empty transcript and no tool arguments or results. + ## Do not - Do not add `openinference-instrumentation-litellm` or `-openai`. DSPy 3.4 runs most models on its own `lm15` engine, where they record nothing; on the LiteLLM engine they add a duplicate model span per call. diff --git a/skills/maple-agent-tracing-google-adk/SKILL.md b/skills/maple-agent-tracing-google-adk/SKILL.md index a4bd4e6840..8000984536 100644 --- a/skills/maple-agent-tracing-google-adk/SKILL.md +++ b/skills/maple-agent-tracing-google-adk/SKILL.md @@ -159,6 +159,28 @@ Without a real key (`MAPLE_TEST`), verify locally: temporarily add `SimpleSpanPr Tell the user about the known gaps: cost is unpriced; the model is the requested id, not the served one, and there is no `gen_ai.response.id`; with `StreamingMode.SSE` the transcript shows the streamed chunks and then the full reply (ADK records each chunk); session check headlines (provider errors, prompt cache) count `call_llm` and `generate_content` separately, so they show twice the LLM call count (tokens and LLM calls are netted correctly). +## Reference notes + +- Scope: ADK for Python. ADK for Go and Kotlin emit the same span names; their provider setup is not covered. +- Expected trace per turn: + + ```text + invocation + └─ invoke_agent assistant + ├─ call_llm + │ └─ generate_content openrouter/openai/gpt-4o-mini + ├─ execute_tool get_weather + └─ call_llm + └─ generate_content openrouter/openai/gpt-4o-mini + ``` + +- Tokens: `generate_content` carries `gen_ai.usage.input_tokens` / `output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` and `gen_ai.usage.reasoning.output_tokens` when reported (cached inside input, thinking inside output). `call_llm` repeats the same usage; Maple nets parent usage against children, so totals and LLM call count are correct. Streamed turns (`StreamingMode.SSE`) report usage; `LiteLlm` requests it via `stream_options.include_usage`. +- Cost: LiteLLM computes a cost but it never reaches ADK's spans; sessions are unpriced. +- `AgentTool` detail: Maple keeps one `gen_ai.conversation.id` per trace and picks the larger of the two, which is why the turn can move to another session. ADK's API docs also prefer `mode="single_turn"`. +- The approval `run_async()` is its own turn, labeled with the original request. +- With the settings in this skill, `generate_content` spans carry no provider attribute; doesn't affect grouping, tokens or transcript. +- 401 from the exporter: the header must be `Authorization=Bearer%20` with a key for the right region. + ## Do not - Do not rely on `OTEL_EXPORTER_OTLP_*` env alone with a plain `Runner`: nothing is exported. diff --git a/skills/maple-agent-tracing-haystack/SKILL.md b/skills/maple-agent-tracing-haystack/SKILL.md index c8018fa336..7ca620c6ba 100644 --- a/skills/maple-agent-tracing-haystack/SKILL.md +++ b/skills/maple-agent-tracing-haystack/SKILL.md @@ -248,6 +248,38 @@ Run one real conversation (≥2 turns, one tool call). If you can export to a lo Tell the user: known gaps are no `gen_ai.provider.name`, no `gen_ai.response.id`, no tool call id on tool spans, and cost only on OpenRouter. +## Reference: span mapping and expected trace + +| Haystack span | What `MapleHaystackTracer` adds | +| --- | --- | +| `haystack.agent.run` | `invoke_agent`, agent name from the pipeline component or `AgentTool` that runs it | +| `haystack.agent.step.llm`, and any `*ChatGenerator` component | `chat`: model, finish reason, input/output/cache/reasoning tokens, cost, messages | +| `haystack.agent.step.tool` | `execute_tool`: tool name, arguments, result, and `Error` status when the tool failed | +| every span | `gen_ai.conversation.id` inside a `conversation()` block | + +```text +haystack.pipeline.run + haystack.component.run assistant + haystack.agent.run invoke_agent, agent "assistant" + haystack.agent.step + haystack.agent.step.llm chat, openai/gpt-4o-mini + haystack.agent.step.tool execute_tool, get_weather + haystack.agent.step + haystack.agent.step.llm chat +``` + +## Known behaviours (expected; explain if the user asks) + +- Why not the ready-made options: plain `OpenTelemetryTracer` gives structure only (model, tokens, transcript, tool failures stay in `haystack.*` blobs; no session id). `openinference-instrumentation-haystack` gives model and tokens on generator spans but no tool spans (the Agent is one opaque chain), writes `session.id` (ignored for this dialect) and is labelled "Unidentified". OpenLLMetry's `opentelemetry-instrumentation-haystack` covers only `Pipeline.run`, `OpenAIGenerator` and `OpenAIChatGenerator`, no Agent steps, tool spans or tokens, and writes content as indexed `gen_ai.prompt.N.*` keys Maple doesn't read. +- Maple also recognizes Haystack's span names, but the `"haystack"` scope covers spans added in newer Haystack releases too. +- Haystack looks up the active tracer on every span, so there is no import-order trap for `enable_tracing()`. Without this tracer, `HAYSTACK_CONTENT_TRACING_ENABLED` is read once at the first `import haystack`; setting it later silently records nothing. +- Tools of one step run in parallel threads; their spans sit side by side under the step. +- Haystack wraps a raised tool exception in `ToolInvocationError`, feeds the text back to the model, writes `{"error": "..."}` as the tool output and leaves the span `Unset`. The tracer sets `Error` regardless of `raise_on_tool_invocation_failure`. +- An Agent inside a `PipelineTool` is named after its component in the inner pipeline. +- The Agent itself reports no usage, so there is no double counting: each model call counts once on its `haystack.agent.step.llm` span. +- Approval gates (`ConfirmationHook` at `before_tool`) stay in the same run and session. A rejected call never reaches the tool, so it has no tool span; the model sees the rejection as a tool result in the transcript. An Agent with hooks also gets a `haystack.agent.hook` span before its tool calls; like `haystack.agent.step`, it carries no model or tool attributes and isn't counted. +- `haystack.pipeline.input_data` on the root span is a plain tag Haystack's own content switch never gated; `content=False` drops it. To redact rather than drop message content, filter values in `_messages()`. + ## Do not - Don't rely on `OpenTelemetryTracer`/`OpenTelemetryConnector` alone: Maple gets no model, tokens, transcript or tool failures. diff --git a/skills/maple-agent-tracing-langchain/SKILL.md b/skills/maple-agent-tracing-langchain/SKILL.md index b86fe07c72..788db099a6 100644 --- a/skills/maple-agent-tracing-langchain/SKILL.md +++ b/skills/maple-agent-tracing-langchain/SKILL.md @@ -187,6 +187,18 @@ Run one real conversation: 2+ messages with the same id, at least one tool call, Known gaps (not setup bugs, don't try to fix): tool spans have no `gen_ai.tool.call.id` or arguments (arguments are in the model's tool call in the transcript); tool results are LangChain's serialized `ToolMessage` JSON (output under `data.content`); chat model spans have no `gen_ai.response.id`; with a checkpointer every turn's label is the thread's first user message (Maple takes it from the `model` node span, where the instrumentor records only the first message; the transcript inside each turn is right). +More details (from the human guide, for edge cases): + +- `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` is equivalent to `enable_genai_semconv=True`, only if set before `TraceConfig` is built. The dual-write happens at span end and never overwrites a key already set (so `AgentSpans` values win). +- Why `AgentSpans`: the instrumentor marks a span AGENT only when its name contains "agent" (`name="support_agent"` yes, `name="assistant"` no) and never sets `gen_ai.agent.name`. Spans with no operation are classified by name, so the `tools` node counts as a tool call and `ChatPromptTemplate` as a model call unless marked `invoke_workflow`. +- Provider comes from the LangChain integration: `ChatOpenAI` reports `openai` even for an Anthropic model behind OpenRouter or another OpenAI-compatible gateway. +- Tool spans lack `gen_ai.tool.call.id`, so Maple matches them to the model's tool calls by name; a reply that calls the same tool twice can mismatch. +- With a checkpointer every model span repeats the whole history. Maple has no per-attribute limit; ingest accepts requests up to 20 MiB. +- HITL: the pause ends the turn's trace; the resume is a new trace, so an approved action shows as two turns in one session (shared `thread_id` is what joins them). +- LangGraph Server verified with `langgraph dev` (langgraph-api 0.10.3). +- LangSmith OTel export (`LANGSMITH_OTEL_ENABLED` + `LANGSMITH_OTEL_ONLY`, tested langsmith 0.14.1): Maple labels it "LangChain" and reads `langsmith.metadata.thread_id`, but prompts/completions arrive as byte attributes (hex, unreadable transcript, no turn labels), interrupts are marked ERROR, middleware wrappers and the `tools` node count as extra tool calls, agent names only in `langsmith.metadata.lc_agent_name` (unread; LangSmith sets `gen_ai.operation.name` after start so a start-time processor can't fix it), and flush needs `wait_for_all_tracers()` (`langchain_core.tracers.langchain`) then `provider.force_flush()`. Don't recommend it. +- LangChain.js/LangGraph.js: LangSmith JS OTel mode is experimental (`initializeOTEL()` deprecated) and the JS OpenInference instrumentor has no GenAI dual-write, so Maple ignores its `session.id` (one session per trace). + Local check without Maple: add `SimpleSpanProcessor(ConsoleSpanExporter())` temporarily and confirm `gen_ai.conversation.id` is identical on every span of every turn of one conversation. ## Do not diff --git a/skills/maple-agent-tracing-litellm/SKILL.md b/skills/maple-agent-tracing-litellm/SKILL.md index fc589b4fa8..0945bc58a8 100644 --- a/skills/maple-agent-tracing-litellm/SKILL.md +++ b/skills/maple-agent-tracing-litellm/SKILL.md @@ -124,7 +124,7 @@ async def call_model(conversation_id: str, messages: list, tools: list | None): ) ``` -- Alternative session carrier: body `metadata: {"session_id": ...}` (`extra_body={"metadata": {...}}`), verified. +- Alternative session carrier: body `metadata: {"session_id": ...}` (`extra_body={"metadata": {...}}`), verified. A W3C `baggage` header with `session.id=...` is used only when neither the header nor metadata is present. - Resulting tree per request: `invoke_agent` → `POST /chat/completions` (proxy FastAPI server span) → `chat ` + `auth /chat/completions`. Known Maple limitation: the `auth /chat/completions` span is counted as an extra LLM call (litellm scope + "chat" in the name), so LLM call counts double on the proxy path in the sessions list; the session detail page also counts the FastAPI `POST /chat/completions` span (it carries `gen_ai.request.model`), so it shows 3x. Tokens, cost, transcript and sessions are correct. Tell the user; nothing to fix app-side. - Never also instrument the app's OpenAI client when the proxy traces: double LLM calls and tokens. Pick gateway OR in-app. - Do not set `OTEL_IGNORE_CONTEXT_PROPAGATION=true` on the proxy. @@ -174,7 +174,7 @@ async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: - v2 default is `no_content`. `capture_message_content="span_only"` (or env `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only`) puts `gen_ai.input.messages` / `gen_ai.output.messages` JSON on `chat` spans. Maple's transcript needs them. - Never `event_only` / `span_and_event` for Maple: events are not read. - Messages are OpenAI chat format (`{role, content, tool_calls}`). Maple's transcript ignores `tool_calls` inside messages: a call that only requested tools shows an empty reply, and tool calls render only from `execute_tool` spans (Step 5). So the `execute_tool` spans are required for tool calls to appear at all. -- User wants content off → `no_content` + drop the tool args/result attributes in `run_tool`. `litellm.turn_off_message_logging = True` keeps structure but replaces text with `redacted-by-litellm`. +- User wants content off → `no_content` + drop the tool args/result attributes in `run_tool`. `litellm.turn_off_message_logging = True` keeps structure but replaces text with `redacted-by-litellm` (applies to every LiteLLM logging callback). Pattern-based redaction → an OTel Collector between the app and Maple. ## Step 5: Tools, errors, sub-agents @@ -271,6 +271,19 @@ Run one real conversation: 2+ messages with the same id, one tool call, one stre Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExporter())` to the provider and confirm `gen_ai.conversation.id` on every `chat` span and the parent ids. +## Troubleshooting + +- `ModuleNotFoundError: No module named 'opentelemetry._events'`, or proxy logs `Error initializing custom logger` and exports nothing → OTel >= 1.44 on LiteLLM 1.103; pin 1.43.0. +- App spans arrive, no `chat` spans → sync `litellm.completion()` under v2; switch to `acompletion()`. +- Spans named `litellm_request` / `raw_gen_ai_request`, each call its own session → v1 logger; use the `OpenTelemetryV2` instance. +- Model, tokens, prompts missing and the SDK logs `Setting attribute on ended span` → v1 writing onto an already-ended parent; use v2 (or Step 2c for sync-only). +- Framework shows Unidentified → own span carries `gen_ai.conversation.id` / `maple_ai.session.id`; remove it. +- No prompts/replies → content capture off (v2 default); `span_only`. +- Proxy spans in their own traces, apart from the app's agent span → no `traceparent` on the request; `propagate.inject(headers)` inside the agent span; no `OTEL_IGNORE_CONTEXT_PROPAGATION` on the proxy. +- Nothing arrives from the proxy → the OTLP env vars aren't in the proxy's own environment, so it stays on the default `console` exporter. +- Session cost lower than LiteLLM spend → `gen_ai.usage.cost` on nested agent spans; report on the outermost only (Step 6). +- Empty assistant reply in the transcript → that call only requested tools; tool rows come from `execute_tool` spans. + ## Do not - Do not use the v1 logger (`litellm.callbacks=["otel"]` alone) when async is possible: no session key, op `acompletion`, attributes lost under parent spans. diff --git a/skills/maple-agent-tracing-llamaindex/SKILL.md b/skills/maple-agent-tracing-llamaindex/SKILL.md index 6abdffa8d6..dded63a7b3 100644 --- a/skills/maple-agent-tracing-llamaindex/SKILL.md +++ b/skills/maple-agent-tracing-llamaindex/SKILL.md @@ -216,6 +216,18 @@ Run one real conversation: 2+ messages with the same id, at least one tool call, - [ ] Cost shows "unpriced". - [ ] No attribute contains an API key or `Bearer ` token. +More details (from the human guide, for edge cases): + +- Native `llama-index-observability-otel` (0.7.0, `LlamaIndexOpenTelemetry`): model settings/prompt only in `LLMChatStartEvent` span events (Maple ignores events); the end event with reply+usage is dropped on streamed calls; two nested `.astream_chat` spans per call; ignores `OTEL_EXPORTER_OTLP_*` and defaults to `ConsoleSpanExporter` without `span_exporter=`. If a user insists on keeping it and only wants sessions: `instrument_tags({"gen_ai.conversation.id": conversation_id})` around `agent.run()` groups traces (dotted tag keys become span attributes verbatim). +- Span layout the processor fixes: `BaseWorkflowAgent.run_agent_step` → `._prepare_chat_with_tools` (kind LLM, ~1 ms) + `.astream_chat` (OpenAILike override) → `.astream_chat` (inner, holds messages+usage). Classes implementing the call directly (`OpenAI`) produce two spans. Merge only applies to directly nested identical names, so `CondensePlusContextChatEngine.chat` → `OpenAI.chat` is untouched. Tokens were never doubled (only innermost has usage), only call counts. +- `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` is equivalent to the config flag only if set before `TraceConfig` is built. +- Tool spans: `FunctionTool.acall` (kind TOOL, `execute_tool`), result is LlamaIndex `ToolOutput` JSON with `raw_input`/`raw_output` (real args are in `raw_input`). On failure the `call_tool` step stays OK because `FunctionAgent` hands the error to the model. +- Agents-as-tools: a tool span with one `FunctionAgent.run` child shows as a delegation; tool args/result become the lane's input/output. Maple opens a lane for every agent span whose name differs from its caller's. +- Provider: `OpenRouter` and every `OpenAILike` report `openai`, even for Anthropic models. No response model name or `gen_ai.response.id` recorded. +- OpenRouter Broadcast for cost: because model spans lack `gen_ai.response.id`, nest Broadcast spans under them (https://maple.dev/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces) or each call counts twice. +- With a persistent `Context` every model span repeats the whole chat. Maple has no per-attribute limit; ingest accepts requests up to 20 MiB. +- Streaming query engines on llama-index-core 0.14.25: the model call of a `StreamingResponse` runs after the query span ended, so it lands in a separate trace. Fix pending in OpenInference PR #3841 (https://github.com/Arize-ai/openinference/pull/3841); until released, pin `llama-index-core<0.14.25` if the app traces streaming query engines. Agents unaffected. + Local check without Maple: temporarily pass `SimpleSpanProcessor(ConsoleSpanExporter())` into `LlamaIndexForMaple` and confirm `gen_ai.conversation.id` is identical on every span of a conversation and `gen_ai.agent.name` is set. ## Do not diff --git a/skills/maple-agent-tracing-mastra/SKILL.md b/skills/maple-agent-tracing-mastra/SKILL.md index 8628c42456..39d9d19e44 100644 --- a/skills/maple-agent-tracing-mastra/SKILL.md +++ b/skills/maple-agent-tracing-mastra/SKILL.md @@ -152,18 +152,19 @@ HITL resumes (`approveToolCallGenerate` / `declineToolCallGenerate` / `approveTo - On by default: `gen_ai.output.messages` on `chat` spans, tool args/results on `execute_tool`. `gen_ai.input.messages` on `chat` spans (with the system prompt as the first message) only exists because of `mapleSpanProcessor`; without it the transcript has replies but no prompts. Tool calls inside earlier history are summarized by Mastra as `[tool: ]` text; the full calls are on the `execute_tool` spans. - `gen_ai.system_instructions` on `invoke_agent` is plain text, which Maple does not decode; the system prompt shows through the system message in the `chat` input instead. - Opt-out per request: `tracingOptions: { hideInput: true, hideOutput: true }` (whole trace). It hides span input/output only, not metadata; `mapleSpanProcessor` removes the one metadata copy of the reply (`mastra.metadata.body`). -- `SensitiveDataFilter` is auto-applied (redacts values under keys like password/token/apiKey/authorization/secret). Do not disable it (`sensitiveDataFilter: false`) unless the user asks. +- `SensitiveDataFilter` is auto-applied (redacts values under keys like password/token/apiKey/authorization/secret). Do not disable it (`sensitiveDataFilter: false`) unless the user asks. It matches key names, not free text: a password typed into the chat still reaches Maple. For pattern-based redaction of message text, suggest an OpenTelemetry Collector between the app and Maple. - Serialization caps: 128 KiB/string, 50 items/array, 50 keys/object, depth 8. If agents keep > ~40 messages of history (`lastMessages` > 40 or custom history), add `serializationOptions: { maxArrayLength: 200 }` to the config, or the newest messages are cut from the transcript. ## Step 5: Tools, errors, sub-agents -- Tools: `createTool({ id, description, inputSchema, execute })`. Span name `execute_tool `, `gen_ai.tool.name` = id. Give tools real ids. +- Tools: `createTool({ id, description, inputSchema, execute })`. Span name `execute_tool `, `gen_ai.tool.name` = id. Give tools real ids. The exporter leaves tool calls out of the `chat` span's output messages, so a model reply that only requests tools shows as an empty assistant message in the transcript; the calls show from the `execute_tool` spans. Expected. - Failures must throw (`throw new Error("...")`). Thrown → span status ERROR with the message as status message, `error.type=unknown`, plus an `exception` event. Returning `{ error }` = counted as success in Maple. If a tool swallows errors into a return value and the user wants failures visible, rethrow. - Every `new Agent({ id, name, ... })` needs a distinct `name`: it is `gen_ai.agent.name`, which Maple uses for lanes. - Supervisor agents (`agents: {...}` on an Agent): each delegation is `execute_tool agent-` → `invoke_agent ` inside the supervisor's trace, one lane per sub-agent. Mastra gives each delegation a thread id `-`, which sorts after the real id, so without `mapleSpanProcessor` Maple picks a sub-agent's id as the session and splits the turn. Same for `agent.network()` and workflow steps calling agents with their own `memory`. The processor fixes all of them; do not remove it. - Maple counts each `execute_tool agent-` delegation as a tool call, next to the sub-agents' own tools. A sub-agent's `chat` input starts with the supervisor's system prompt and the user's original message (Mastra forwards the supervisor's conversation); expected. - HITL (`requireApproval: true` tools): the model call that requests the tool is exported twice, once without tokens or output when the run suspends and once with its tokens when the run resumes. Maple shows one extra model call with 0 tokens per approval; token totals are right. Expected, not a setup error. +- Workflow runs: the root span is `invoke_workflow `; Maple treats it as the turn. Steps in `.parallel([...])` show as overlapping lanes. - Workflow steps that call an agent: pass `tracingContext` from the step's `execute` args into `agent.generate(prompt, { tracingContext })`, or the agent starts a separate trace. ## Step 6: Flush @@ -187,7 +188,18 @@ Run one conversation: 2+ user messages with the same thread id (one streamed), a - No span has `mastra.metadata.headers` or `mastra.metadata.body`. - `execute_tool ` spans have the real tool name, arguments and result; a throwing tool is marked failed with its message; successful tools are not. - Supervisor/workflow: one session, one turn per run, one lane per sub-agent `name`, all spans in one trace; tool calls include the `agent-` delegations. -- Cost shows as unpriced (expected: Mastra emits no cost attribute). +- Cost shows as unpriced (expected: Mastra emits no cost attribute). Mastra exports no `gen_ai.response.id`; that's fine, each call's usage is reported once. + +## Troubleshooting symptoms + +- `Custom configuration requires endpoint. Tracing will be disabled.` → no `endpoint` passed in code (env vars are not read). +- `Traces http/json exporter is not installed` / `http/protobuf exporter is not installed` → protocol package missing (optional deps skipped); set `protocol: "http/protobuf"` and install `@opentelemetry/exporter-trace-otlp-proto`. +- `Export FAILED` with 401/403 in debug output → missing `Authorization` header or wrong key/region. Value is `Bearer ` + ingest key. +- One model call per turn carrying the turn's summed tokens → `@mastra/core` / `@mastra/observability` / `@mastra/otel-exporter` from different releases; exporter fell back to `model_generation` as the model call. Update all three together. +- Dozens of `model_chunk` spans → add `excludeSpanTypes: [SpanType.MODEL_CHUNK]`. Hundreds of `workflow_step` spans → remove `includeInternalSpans: true`. +- A workflow's agents in separate traces → pass `tracingContext` from the step. +- Latest user message missing in long chats → raise `serializationOptions.maxArrayLength`. +- Spans twice → two exporters to Maple (OtelExporter + OTel bridge, or + OpenLLMetry/OpenInference on the AI SDK). Keep OtelExporter only. ## Do not diff --git a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md index f6e9e71f3f..7c61ab7941 100644 --- a/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md +++ b/skills/maple-agent-tracing-microsoft-agent-framework/SKILL.md @@ -84,7 +84,19 @@ configure_otel_providers( trace.get_tracer_provider().add_span_processor(ConversationIdProcessor()) ``` +Equivalent env-var config, with a bare `configure_otel_providers()` call: + +```bash +export OTEL_SERVICE_NAME="" +export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" +export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" +export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer " +export ENABLE_SENSITIVE_DATA="true" +export ENABLE_MESSAGE_EVENTS="false" +``` + - `otlp_protocol` is mandatory: MAF defaults to gRPC (spec says http/protobuf) and fails silently or with an ImportError. +- `configure_otel_providers()` appends `/v1/traces`, `/v1/metrics`, `/v1/logs` to the endpoint. `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`, if set, is used as-is and must include `/v1/traces`. - Existing provider instead: add `BatchSpanProcessor(OTLPSpanExporter(endpoint="https://ingest.maple.dev/v1/traces", headers={...}))` and `ConversationIdProcessor()` to it, then `from agent_framework.observability import enable_instrumentation; enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)`. - For OpenRouter or any Chat Completions endpoint use `OpenAIChatCompletionClient(model=..., api_key=..., base_url=...)`, not `OpenAIChatClient` (Responses API; OpenRouter rejects its `previous_response_id` on turn 2). @@ -92,6 +104,7 @@ trace.get_tracer_provider().add_span_processor(ConversationIdProcessor()) ```bash dotnet add package Microsoft.Agents.AI --version 1.22.0 +dotnet add package Microsoft.Agents.AI.OpenAI --version 1.22.0 dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol --version 1.19.1 ``` @@ -134,6 +147,8 @@ sealed class ConversationIdProcessor : BaseProcessor ## Step 3: Conversation id +MAF only sets `gen_ai.conversation.id` from the provider's own conversation id (`AgentSession.service_session_id`: Responses API with `store=True`, or a Foundry agent owning the thread). With Chat Completions, OpenRouter, Ollama or local `AgentSession` history it is never set, and `AgentSession.session_id` is not exported, so every `agent.run()` becomes its own `trace:` session. + Wrap every request/turn so all spans of that turn start inside `conversation()`: ```py @@ -154,12 +169,16 @@ async def handle_message(session: AgentSession, text: str) -> str: - Python: `enable_sensitive_data=True` (or `ENABLE_SENSITIVE_DATA=true`). .NET: `EnableSensitiveData = true` (or `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=true`). - If the process sets `OTEL_SEMCONV_STABILITY_OPT_IN`, it must include `gen_ai_latest_experimental` (e.g. `http,gen_ai_latest_experimental`); otherwise MAF moves content off the spans into log events Maple doesn't read. - `enable_message_events=False`: otherwise every message is also exported as OTLP log records (duplicate payload). +- `gen_ai.tool.definitions` (each tool's JSON schema) is sent on every `invoke_agent` span even with sensitive data off. - If the user wants no prompt content stored, leave sensitive data off and tell them the transcript and tool arguments/results will be empty. ## Step 5: Tools, errors, sub-agents - Give every `Agent` a distinct `name` (Maple lanes key on `gen_ai.agent.name`; unnamed agents get a UUID). - Tool failures: raising from the tool function is enough; MAF sets ERROR + `error.type` on `execute_tool`. Do not catch and return an error string from the tool body (that hides the failure). +- Python MAF also logs each tool failure as an ERROR and a WARN log record; `configure_otel_providers()` exports them as OTLP logs next to the span. +- Approval-gated tools (`@tool(approval_mode="always_require")`): the resume is a new `agent.run()` and trace; it joins the session only inside `conversation()`. A rejected call emits no `execute_tool` span; the rejection only appears in the next `chat` span's input messages. +- Workflows record fan-in as span links, so parallel executors are siblings under `workflow.run`. - Sub-agents: prefer `worker.as_tool()` in the orchestrator's `tools=[...]`, or MAF workflows/orchestrations. Do not also instrument the provider SDK (e.g. OpenInference/OpenLLMetry OpenAI instrumentors): that double-counts every call. ## Step 6: Flush @@ -217,6 +236,7 @@ trace.set_tracer_provider(provider) - Transcript comes only from `invoke_agent` spans (`ChatCompletionAgent`). Model-call content is Python logging, which Maple doesn't read; code that uses the kernel without an agent has no transcript. Tell the user. - Orchestrations on `InProcessRuntime`: wrap in `with conversation(id), tracer.start_as_current_span(""):` and call `runtime.start()` inside it, or the run splits into many traces. - Flush with `provider.shutdown()` in `finally`. +- Other SK differences to expect (not setup bugs): tool spans are `execute_tool -`; failing tools get ERROR + `error.type` but no result attribute; finish reasons are Python enum names (`FinishReason.STOP`), so Maple's reply-length and refusal checks can't read them; `temperature=0` is omitted from spans. - .NET SK: same env vars (or `AppContext` switches `Microsoft.SemanticKernel.Experimental.GenAI.EnableOTelDiagnostics[Sensitive]`), `AddSource("Microsoft.SemanticKernel*")`, same `ConversationIdProcessor`. Shows as "Unidentified" framework in Maple. ## Step 7: Verify @@ -232,6 +252,13 @@ Run one real conversation (2-3 turns, one tool call; a second conversation if ch - No attribute contains the model API key or `Bearer `. - Cost: none is emitted; Maple shows sessions as unpriced. Framework label: "Microsoft Agent Framework" / "Semantic Kernel" (Python); .NET shows "Unidentified". +## Tokens and cost notes + +- `chat` spans carry `gen_ai.usage.input_tokens` / `output_tokens` plus cache-read, cache-creation and reasoning buckets when the provider returns them. The OpenAI chat-completions client requests `include_usage` itself when streaming. +- `invoke_agent` repeats the sum of its own `chat` calls; Maple nets it out, so don't strip it. +- `gen_ai.provider.name` is the client type (`openai` even when pointed at OpenRouter or Ollama); `server.address` holds the real base URL. +- .NET streamed `chat` spans carry `gen_ai.response.time_to_first_chunk` (time to first token in Maple); Python MAF doesn't record it. + ## Do not - Do not leave the MAF protocol at its gRPC default. diff --git a/skills/maple-agent-tracing-openai-agents/SKILL.md b/skills/maple-agent-tracing-openai-agents/SKILL.md index f85747bec5..136846778e 100644 --- a/skills/maple-agent-tracing-openai-agents/SKILL.md +++ b/skills/maple-agent-tracing-openai-agents/SKILL.md @@ -142,6 +142,32 @@ Without it the SDK sends no `stream_options` for non-OpenAI clients and every `r Verified with `@openai/agents` 0.18.0, `@arizeai/openinference-instrumentation-openai-agents` 0.2.15, `@opentelemetry/sdk-trace-node` 2.11. Use a `NodeTracerProvider({ resource: detectResources({ detectors: [envDetector] }), spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())] })` (`@opentelemetry/resources`, `@opentelemetry/exporter-trace-otlp-proto`; the 2.x provider does NOT read `OTEL_SERVICE_NAME` without `envDetector`, you get `unknown_service:node`), `provider.register()`, then `new OpenAIAgentsInstrumentation({ tracerProvider: provider }).manuallyInstrument(agents)`. Session: wrap each run in `context.with(setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }), () => agents.run(agent, text))` (`setAttributes` from `@arizeai/openinference-core`). Tell the user the limits: framework shows as Unidentified, inputs render as messages but model replies and the model's tool-call requests are MISSING from each call (`output.value` is the raw `chat.completion` response, which Maple can't decode; earlier replies only show as history in the next call's input, so each turn's final reply is absent), no tool arguments/results fields, no agent lanes, the reply-length check is skipped (no finish reason read). `await provider.forceFlush()` before a script exits. Maple ignores `session.id`/`setSession` for this scope. +Reference setup (verified): + +```ts +import * as agents from "@openai/agents" +import { context } from "@opentelemetry/api" +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { detectResources, envDetector } from "@opentelemetry/resources" +import { BatchSpanProcessor, NodeTracerProvider } from "@opentelemetry/sdk-trace-node" +import { setAttributes } from "@arizeai/openinference-core" +import { OpenAIAgentsInstrumentation } from "@arizeai/openinference-instrumentation-openai-agents" + +export const provider = new NodeTracerProvider({ + resource: detectResources({ detectors: [envDetector] }), // OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES + spanProcessors: [new BatchSpanProcessor(new OTLPTraceExporter())], // OTEL_EXPORTER_OTLP_* variables +}) +provider.register() +new OpenAIAgentsInstrumentation({ tracerProvider: provider }).manuallyInstrument(agents) + +export function handleMessage(conversationId: string, text: string) { + const ctx = setAttributes(context.active(), { "gen_ai.conversation.id": conversationId }) + return context.with(ctx, () => agents.run(agent, text, { session })) +} +``` + +Models, tokens and tool names come through as in Python. If the user needs the full session page (replies, tool args/results, agent lanes) in TypeScript, point them to https://maple.dev/docs/agent-tracing/provider-sdks to emit the GenAI attributes themselves. + ## Step 3: Session id (one conversation = one session) Maple reads `session.id` on this framework's spans. Only OpenInference's `using_session` sets it. `RunConfig(group_id=...)`, `trace_metadata` and SDK `Session` ids are NOT exported. @@ -163,14 +189,29 @@ async def handle_message(conversation_id: str, text: str) -> str: - `run_streamed`: call it inside the `with` block (its background task inherits the context there); consuming `stream_events()` may continue inside or after. - Human-in-the-loop resumes (`Runner.run(agent, state)` after `state.approve(...)`): wrap in the same `using_session` id. The resume is its own trace (a second turn with the same user message as label) whose root is the workflow-named span, not an `invoke_agent` root. - Set `RunConfig(workflow_name=...)` to name the trace root; the default `Agent workflow` is the same for every run. The name must NOT contain `chat`, `completion` or `tool` (any case): the per-run CHAIN span carries it with no operation, and Maple's name fallback then counts every run as an extra LLM call (`chat`, `completion`) or tool call (`tool`). End it in `workflow` or `agent`. -- Multi-agent fan-out with several `Runner.run` calls: wrap them in one `with trace("")` (from `agents`) inside `using_session`, or each run becomes its own trace/turn. +- `using_session` stores the id in a contextvar; asyncio tasks inherit it, so concurrent tool calls and agents-as-tools inside the block get it too. +- Multi-agent fan-out with several `Runner.run` calls: wrap them in one `with trace("")` (from `agents`) inside `using_session`, or each run becomes its own trace/turn: + + ```py + import asyncio + + from agents import trace + + with using_session(conversation_id), trace("amsterdam briefing"): + weather, budget = await asyncio.gather( + Runner.run(weather_worker, "Weather in Amsterdam?"), + Runner.run(budget_worker, "3-day budget for Amsterdam?"), + ) + ``` - Do not set `gen_ai.conversation.id` or `maple_ai.session.id` by hand; the dual-write copies `session.id` to `gen_ai.conversation.id` already. ## Step 4: Content - Content is ON by default (SDK `trace_include_sensitive_data=True`, OpenInference copies it). With Step 2c, messages land in `gen_ai.input.messages` / `gen_ai.output.messages`. Nothing to enable. - Content is sent three times per model span (flattened `llm.input_messages.*`, `input.value`, `gen_ai.input.messages`). Expected; don't strip the OpenInference keys, the GenAI ones are derived from them. -- User wants no content / PII-sensitive: `OPENAI_AGENTS_TRACE_INCLUDE_SENSITIVE_DATA=false` (or `RunConfig(trace_include_sensitive_data=False)`). Tell them: empty transcript, tool error details redacted, tokens/tools/failures remain. Alternative at the bridge: `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)`. +- User wants no content / PII-sensitive: `OPENAI_AGENTS_TRACE_INCLUDE_SENSITIVE_DATA=false` (or `RunConfig(trace_include_sensitive_data=False)`). Tell them: empty transcript, tool error details redacted (`Tool execution failed. Error details are redacted.`), tokens/tools/failures remain. Alternative at the bridge: `TraceConfig(enable_genai_semconv=True, hide_inputs=True, hide_outputs=True)` (or `OPENINFERENCE_HIDE_INPUTS` / `OPENINFERENCE_HIDE_OUTPUTS`); also empties the transcript. +- Partial redaction (e.g. mask emails): drop or rewrite attributes in an OpenTelemetry Collector with the `transform` or `redaction` processor. +- Every model span repeats the conversation so far. Maple has no per-attribute limit and accepts requests up to 20 MiB, so this costs bandwidth, not data. ## Step 5: Tools, errors, sub-agents @@ -180,6 +221,8 @@ async def handle_message(conversation_id: str, text: str) -> str: - Handoffs: a `handoff to ` span (counted as a tool call named `transfer_to_` via `MapleSpanFixes`) and the target agent span as a sibling of the source agent's. Nothing to add. - `needs_approval=True` tools: the paused run records a tool span without a result, and the resumed run records the executed call again, so Maple shows the tool twice for one approved call. Expected; tell the user. - Known gaps, don't try to fix: no `gen_ai.tool.call.id` on tool spans; no `gen_ai.response.id` on Chat Completions model spans; no cost. +- Model spans: named `generation` (Chat Completions) or `response` (Responses API). Model on Chat Completions = the configured id (`openai/gpt-4o-mini`); on Responses = the name OpenAI returns, usually a dated snapshot (`gpt-4o-mini-2024-07-18`). Provider is always `openai` (even an Anthropic model behind OpenRouter); Maple applies the OpenAI token convention (cached tokens inside the input total). Responses API also records cached input tokens and reasoning tokens (`llm.token_count.completion_details.reasoning`). The bridge doesn't export the SDK's per-run/per-turn usage totals, so nothing is double-counted. +- Cost via OpenRouter: its Broadcast traces (https://maple.dev/docs/agent-tracing/openrouter) carry per-call cost. Because Chat Completions model spans have no `gen_ai.response.id`, nest the Broadcast spans under them (see "Join Broadcast to your own traces" in that guide) or each call is counted twice. ## Step 6: Flush @@ -222,6 +265,7 @@ Raw span check (optional, e.g. an `InMemorySpanExporter` in a scratch run): mode - Do not disable the SDK's tracing (`set_tracing_disabled(True)`, `OPENAI_AGENTS_DISABLE_TRACING`, `tracing_disabled=True`) to stop the OpenAI upload; `set_trace_processors` already removes it. - Do not omit `TraceConfig(enable_genai_semconv=True)`. - Do not register `MapleSpanFixes` with `add_trace_processor` or after `instrument()`; it must precede the OpenInference processor. +- Do not re-add `default_processor()` (or anything via `add_trace_processor`) without a valid OpenAI key: logs fill with `Tracing client error 401`. - Do not put `chat`, `completion` or `tool` in `workflow_name`. - Do not pass OpenRouter-style model strings (`Agent(model="openai/gpt-4o-mini")`): the SDK strips `openai/` and rejects other prefixes (`Unknown prefix: anthropic`). Use `OpenAIChatCompletionsModel(model=..., openai_client=...)`. - Do not rely on `group_id`, `trace_metadata` or SDK `Session` ids for the Maple session; use `using_session`. diff --git a/skills/maple-agent-tracing-openrouter/SKILL.md b/skills/maple-agent-tracing-openrouter/SKILL.md index b5bbaa7965..f7ae615bd7 100644 --- a/skills/maple-agent-tracing-openrouter/SKILL.md +++ b/skills/maple-agent-tracing-openrouter/SKILL.md @@ -153,6 +153,24 @@ Run one conversation (3+ turns, one with a tool call) with a fixed `session_id`, Tell the user: cost is shown (OpenRouter's charge); cache writes, TTFT, environment, tool calls and agent lanes are not available from Broadcast; transcript is raw JSON and Broadcast-only turns are unlabeled segments. Claude models (`gen_ai.provider.name=anthropic`): Maple applies Anthropic's input-excludes-cache rule to OpenRouter's cache-inclusive input, so cache reads count twice in token totals (cost unaffected). +## Known behavior (tell the user when relevant) + +- Transport: OTLP over HTTP with JSON encoding only; Maple ingest accepts JSON on `/v1/traces`, no collector needed. +- `session_id` also makes OpenRouter route a session's requests to the same provider (better prompt-cache hits). Body `session_id` wins over the `x-session-id` header if both are sent. +- No `session_id` → each call is its own session named `trace:`, one turn, one model call. +- Turns: Maple counts one turn per trace, so Broadcast-only a user message with three model calls = three turns (shown as unlabeled Segment 1, 2, ...). +- Dedup when nested: if `LLM Generation` is a descendant of the app's model-call span, usage is counted at the deepest span that reports it; siblings or separate traces are matched by `gen_ai.response.id`. When both spans carry cost, Maple keeps the larger (OpenRouter's). +- Span tree: `LLM Generation` root, `provider attempt N: ` children, sometimes `generation` / `moderation` children. Only `LLM Generation` counts as a model call, except on the all-attempts-failed path: that call has no usage, and Maple counts the root and each failed attempt as separate model calls (one failed request with one attempt = two LLM calls). +- Content format: `gen_ai.prompt` = `{"messages": [...]}`, `gen_ai.completion` = `{"completion": "...", "reasoning": "..."}`, both JSON strings; Maple shows them as raw JSON blocks. If the app also emits `gen_ai.input.messages`, that transcript renders normally. Values over 10,000,000 chars are shortened with a `.truncated` attribute; ingest body limit 20 MiB. +- Attributes on `LLM Generation`: `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, `gen_ai.usage.input_tokens.cached` (included in input), `gen_ai.usage.output_tokens.reasoning` (included in output), `gen_ai.usage.total_cost` (USD, actual charge), `gen_ai.request.model` / `gen_ai.response.model` (OpenRouter slug), `gen_ai.response.finish_reasons`. +- Not read by Maple: `gen_ai.usage.input_tokens.cache_write` (cache-write column empty, totals fine), `trace.metadata.openrouter.first_token_ms` (Maple reads TTFT only from `gen_ai.response.time_to_first_chunk`), the **Cost** generation-metadata option's `span.metadata.openrouter_generation.*`. +- `gen_ai.provider.name` = the model's author (`openai`, `anthropic`); the serving provider is `trace.metadata.openrouter.provider_name` (e.g. `Amazon Bedrock`). +- Streaming: tokens and cost present without `stream_options.include_usage` (server-side accounting). +- `service.name` on Broadcast spans is always `openrouter`, no environment attribute. Custom keys in the `trace` object arrive as `trace.metadata.` (searchable in Traces, don't set service/environment). +- OpenRouter's sample trace (`Test Trace - OpenRouter Observability`, an `openai/gpt-4-turbo` call with sample tokens/cost) can appear as a session; ignore it. +- Test Connection passes but nothing arrives: check API key filter, data regions, **Enable Broadcast** on the account/org the app key belongs to, and that the key isn't `MAPLE_TEST`. +- Destinations can be created via OpenRouter's observability API (`type: "otel-collector"`) with a management key (only if the user asks; see Do not). + ## Do not - Do not use a constant or per-client `session_id`; it merges every user into one session. diff --git a/skills/maple-agent-tracing-opentelemetry/SKILL.md b/skills/maple-agent-tracing-opentelemetry/SKILL.md index 62811c0970..216cbe81d2 100644 --- a/skills/maple-agent-tracing-opentelemetry/SKILL.md +++ b/skills/maple-agent-tracing-opentelemetry/SKILL.md @@ -65,10 +65,15 @@ Rules: Message JSON (`input.messages`/`output.messages`): array of `{role, parts}`; parts `{type:"text",content}`, `{type:"tool_call",id,name,arguments:}`, `{type:"tool_call_response",id,response}`, `{type:"reasoning",content}`. Output messages add `finish_reason`. `system_instructions` = array of parts, no role: `[{"type":"text","content":"..."}]`. Always a JSON **string** attribute; plain text and structured (non-string) attribute values don't render. +Also read: `reasoning` parts are rendered; a message may carry `content` (string or part array) instead of `parts`. `gen_ai.response.model` wins over `gen_ai.request.model` when both are set. `gen_ai.response.finish_reasons` feeds the refusal (`content_filter`) and truncation (`length`) checks. + +Legacy spellings are read as fallbacks (new key wins when both are set; use current names in new code): `gen_ai.system` (→ `gen_ai.provider.name`, renamed in semconv 1.37), `gen_ai.usage.prompt_tokens`/`completion_tokens`, whole-value `gen_ai.prompt`/`gen_ai.completion`, `gen_ai.usage.cache_creation.input_tokens` (→ `cache_write`), `gen_ai.usage.total_cost` (→ `cost`). + `provider.name` = the API actually called: `openai`, `anthropic`, `gcp.gemini`, `gcp.vertex_ai`, `aws.bedrock`, `azure.ai.openai`, `mistral_ai`, `groq`, `x_ai`, `deepseek`, or `openrouter` for OpenRouter. ## Step 4: Session id (required) +- The id is read only from spans that have `gen_ai.operation.name`; on an unclassified span it is ignored. - Set `gen_ai.conversation.id` on the turn's `invoke_agent` span from the app's conversation/chat/thread id. Same value for every message of a conversation; different across conversations. One classified span per trace is enough; every span in the trace joins. - Never: `uuid4()`/`randomUUID()` per request, the trace id, a module-level constant, a per-process default. No real id (single-shot script) → generate one per conversation, not per message, and reuse it. - Sub-agents in the same trace: no id (they inherit the trace's session) or the same id. Two different ids in one trace → the lexically larger silently wins. @@ -78,6 +83,28 @@ Escape hatch, only when a framework's spans carry a session key Maple ignores fo - Only on your own wrapper span, never on framework spans (it re-vendors the span to `maple`: framework decoding lost, usage read as inclusive of cache). - Use the same value the framework would use for the session. - Not needed for hand-written spans: use `gen_ai.conversation.id`. +- The session then shows framework **Maple**. + +```ts +// chatId comes from your request; frameworkAgent is the framework's agent +await tracer.startActiveSpan( + "invoke_agent support", + { + attributes: { + "gen_ai.operation.name": "invoke_agent", + "gen_ai.agent.name": "support", + "maple_ai.session.id": chatId, + }, + }, + async (span) => { + try { + return await frameworkAgent.run(message) // the framework's spans nest under this one + } finally { + span.end() + } + }, +) +``` ## Step 5: Content @@ -89,11 +116,13 @@ Escape hatch, only when a framework's spans carry a session key Maple ignores fo ## Step 6: Tools, errors, sub-agents -- Tool failure: set status ERROR with the error message as description, set `error.type` (exception class or error code), no `gen_ai.tool.call.result`, then return the error to the model as the tool result so the loop continues. Keep the message specific: Maple groups failures by it. +- Tool failure: set status ERROR with the error message as description, set `error.type` (exception class or error code), no `gen_ai.tool.call.result`, then return the error to the model as the tool result so the loop continues. Keep the message specific: Maple's tool pages group failures by it (ids and numbers masked). +- Maple counts any span as failed if it has status ERROR, a non-empty `error.type`, or `gen_ai.response.status`=`failed`. - Tools that return `{"error": ...}` instead of raising: mark the span failed the same way when you detect it. - Model call failure: status ERROR + `error.type` (HTTP status or exception class), rethrow. - Agent run failure: same on `invoke_agent`. - Sub-agent: call its loop inside the delegating tool's `execute_tool` span, so `execute_tool ask_x` → `invoke_agent x` → its `chat`/`execute_tool`. Distinct `gen_ai.agent.name` per agent (lanes need it). +- Delegation detection: an `execute_tool` span whose only child is an `invoke_agent` span is drawn as a delegation into a lane named after the child's `gen_ai.agent.name`; the tool's arguments/result become the lane's input/output. Two agents with the same name share one lane; an `invoke_agent` span without a name gets no lane. - Parallel tools: start each `execute_tool` span inside the turn's context (Node `Promise.all` keeps it; Python threads need `contextvars.copy_context().run`). ## Step 7: Tokens and cost diff --git a/skills/maple-agent-tracing-provider-sdks/SKILL.md b/skills/maple-agent-tracing-provider-sdks/SKILL.md index 85c42c7804..f6ccf7a52b 100644 --- a/skills/maple-agent-tracing-provider-sdks/SKILL.md +++ b/skills/maple-agent-tracing-provider-sdks/SKILL.md @@ -119,6 +119,20 @@ Local check without Maple: temporarily add `SimpleSpanProcessor(ConsoleSpanExpor - Anthropic instrumentation 1.2b0 records no time to first chunk for `messages.stream()`. - Gemini path (Python instrumentation, TS mapping) not yet run end to end against a live model; OpenAI and Anthropic paths are verified. +## Troubleshooting + +- Every model call its own session → call ran outside `agent_span`, or the span ended before the call (e.g. a stream consumed after it closed). +- One session per turn → conversation id changes per request. +- Empty transcript → capture unset, `EVENT_ONLY` or legacy `true`; set `SPAN_ONLY` in the process that makes the calls. +- No Python model spans → `.instrument()` ran after the first request, or the OpenLLMetry package (`opentelemetry-instrumentation-openai`) was installed instead of `-genai-openai`. +- Duplicate model spans → second instrumentation on the same SDK. `opentelemetry-instrument` loads every installed instrumentation package, so uninstall extras rather than just not calling them. +- Twin traces per call with OpenRouter → OpenRouter Broadcast also exports the calls; keep one source or nest Broadcast under the app's spans (https://maple.dev/docs/agent-tracing/openrouter#join-broadcast-to-your-own-traces). +- Streamed turn has 0 tokens → missing `stream_options.include_usage`. +- Failing tool shown as success → exception caught outside `run_tool` without status/`error.type`. +- Sub-agent calls in the orchestrator's lane → same `gen_ai.agent.name`, or never wrapped. +- Exporter logs 401 → wrong key or region; EU keys only work with `ingest.eu.maple.dev`. +- OpenInference instead of the GenAI packages (Python): OpenAI shows as "OpenInference · OpenAI" with tokens counted but the transcript is the raw request JSON as one message, no turn labels; the Anthropic/Gemini OpenInference packages show as Unidentified with the same raw transcript. Prefer the `-genai-` packages. + ## Do not - Do not install `opentelemetry-instrumentation-openai` or `opentelemetry-instrumentation-anthropic` (OpenLLMetry) or the deprecated `opentelemetry-instrumentation-openai-v2`; install the `-genai-` packages. diff --git a/skills/maple-agent-tracing-pydantic-ai/SKILL.md b/skills/maple-agent-tracing-pydantic-ai/SKILL.md index 7b0bd73a7a..e1e3e289f3 100644 --- a/skills/maple-agent-tracing-pydantic-ai/SKILL.md +++ b/skills/maple-agent-tracing-pydantic-ai/SKILL.md @@ -188,6 +188,22 @@ Quick local cue: Pydantic AI 2.51 prints an `observability: off` banner on the f Local check without Maple: add `ConsoleSpanExporter` via `SimpleSpanProcessor` temporarily and confirm `gen_ai.conversation.id` is identical across runs of one conversation and on delegate spans. +## Known behavior (tell the user when relevant) + +- Tool failure matrix: `ToolFailed` → span ERROR, model sees message, run continues; `ModelRetry` → ERROR, one failed call per retry; other exception → ERROR, run raises, failed turn; `return {"error": ...}` → UNSET, shows as successful call. +- Approval-gated tools (`requires_approval=True`) pause the run with no tool span; the `execute_tool` span appears only in the resumed run (the one sending `DeferredToolResults`). Same `conversation_id=` → paused and resumed runs are two turns of one session, both labeled with the original request. +- Delegation rendering: an `execute_tool ` span whose only child is `invoke_agent ` shows as a delegation; the tool's args/result become the lane's input/output. Several delegation tools in one model reply run concurrently, so lanes overlap in time. +- `include_content=False` also drops exception messages (only the exception type is kept). +- Instrumentation `version`: 5 default (tested); 2-4 deprecated with `PydanticAIDeprecationWarning` (version 2 used different span names); 6 is opt-in and sends tool results with `role: "tool"`. Fix the warning by removing `version=`. +- Tokens: `chat` spans carry `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens`, plus `gen_ai.usage.cache_read.input_tokens` / `gen_ai.usage.cache_creation.input_tokens` when the provider caches. `invoke_agent` carries run totals under `gen_ai.aggregated_usage.*`, which Maple does not add to the session total. Delegate tokens stay on the delegate's spans. +- Anthropic + prompt caching via Pydantic AI's `anthropic` provider: Maple currently counts cached input tokens twice (Pydantic AI reports input tokens including cache for every provider; Maple applies Anthropic's convention where they're separate). OpenAI, OpenRouter, Gemini unaffected. +- Streaming: Pydantic AI requests usage on OpenAI-compatible streams (`stream_options.include_usage`), so streamed calls have tokens. Time-to-first-chunk is recorded under a key Maple doesn't read yet. +- Cost: Pydantic AI writes `operation.cost` on `chat` spans; Maple reads cost only from `gen_ai.usage.cost`, `gen_ai.usage.total_cost` or `llm.cost.total` and never prices tokens itself, so cost shows as unpriced. +- Logfire with `send_to_logfire=True` (or a Logfire token in env) sends to both Logfire and Maple. +- Provider extras: swap `[openai]` for `anthropic`, `google`, `openrouter`, ...; the full `pydantic-ai` package also works. +- Duplicate spans: another instrumentor on the model client (Logfire `instrument_openai()`, OpenInference, OpenLLMetry) double-traces model calls; keep Pydantic AI's. +- Export 413 / huge `chat` spans: binary content recorded as base64; set `include_binary_content=False`. + ## Do not - Do not rely on the automatic `gen_ai.conversation.id`: it's a new UUID7 per run without history. diff --git a/skills/maple-agent-tracing-smolagents/SKILL.md b/skills/maple-agent-tracing-smolagents/SKILL.md index ba72fd8281..b189254975 100644 --- a/skills/maple-agent-tracing-smolagents/SKILL.md +++ b/skills/maple-agent-tracing-smolagents/SKILL.md @@ -143,6 +143,17 @@ def handle_message(conversation_id: str, text: str) -> str: - `CodeAgent` with `executor_type` other than local (`e2b`, `docker`, `modal`, ...) runs tools outside the process: no tool spans. Tell the user; nothing to fix in tracing. - Known, not fixable here: no `gen_ai.tool.call.id` on tool spans; a failed tool also marks its `Step N` span ERROR (Maple: toolErrorCount 1, plus an "Other errors" warning for the Step span). +## Known behavior (tell the user when relevant) + +- `ToolCallingAgent`: a model reply that calls `final_answer` together with another tool raises `AgentExecutionError`; that step shows as a failed `Step` span although the run continues. It's the framework rejecting the model output, not the user's code. +- A tool failure puts two `exception` events on the enclosing `Step N` span (wrapped `AgentToolExecutionError`). +- `CodeAgent` with `LocalPythonExecutor`: tool calls from generated code produce normal tool spans; positional args are recorded as a JSON array (`["Berlin"]`). +- Tokens: only input/output tokens (`gen_ai.usage.input_tokens`/`output_tokens` + `llm.token_count.*`); cached and reasoning tokens are not recorded. Model = the requested id (`gen_ai.request.model`), not the provider's response model. +- Provider comes from the model class: `OpenAIServerModel` is always `openai`, even for an Anthropic model behind OpenRouter. `OpenAIServerModel` is an alias of `OpenAIModel`, so spans are `OpenAIModel.generate`. +- Streaming (`stream_outputs=True`) → `.generate_stream` spans; smolagents requests `stream_options={"include_usage": True}` for `OpenAIServerModel`, `LiteLLMModel`, `InferenceClientModel`, so tokens are kept. +- Cost: smolagents records none; sessions show as unpriced. +- Turn labels and session title read `New task:` (smolagents prefixes each task; Maple uses the first line of the user message). + ## Step 6: Flush - `TracerProvider` flushes on normal interpreter exit (atexit). That covers servers and CLIs that exit normally. diff --git a/skills/maple-agent-tracing-spring-ai/SKILL.md b/skills/maple-agent-tracing-spring-ai/SKILL.md index 8ff680c90e..bffcca9ae5 100644 --- a/skills/maple-agent-tracing-spring-ai/SKILL.md +++ b/skills/maple-agent-tracing-spring-ai/SKILL.md @@ -32,7 +32,7 @@ Mechanism: Spring AI's Micrometer Observations → `micrometer-tracing-bridge-ot ### 2a. Dependencies (Boot 4, Spring AI 2.0) -Keep the existing `spring-ai-bom` import and model starter. Add: +Keep the existing `spring-ai-bom` import and model starter. If there is no BOM yet, import `org.springframework.ai:spring-ai-bom:2.0.1` (`pomimport` in `dependencyManagement`; Gradle `implementation(platform("org.springframework.ai:spring-ai-bom:2.0.1"))`). Add: ```xml @@ -88,6 +88,8 @@ Then the agent exports everything: set `OTEL_EXPORTER_OTLP_ENDPOINT=https://inge - Deps: `io.micrometer:micrometer-tracing-bridge-otel` + `io.opentelemetry:opentelemetry-exporter-otlp` + `spring-boot-starter-actuator` (required on Boot 3.5; no `spring-boot-starter-opentelemetry`). - Properties: `management.otlp.tracing.endpoint=https://ingest.maple.dev/v1/traces`, `management.otlp.tracing.headers.Authorization=Bearer ...`; sampling property unchanged. Stream usage: `spring.ai.openai.chat.options.stream-usage=true`. - Step 3 class: replace `tools.jackson.databind.json.JsonMapper.shared().writeValueAsString(...)` with a Jackson 2 `com.fasterxml.jackson.databind.ObjectMapper` (wrap the checked `JsonProcessingException`), and delete the two `getToolCallId()` lines (not in 1.1). +- OpenAI model property: `spring.ai.openai.chat.options.model` on 1.1 (2.0: `spring.ai.openai.chat.model`). +- Boot 4 still accepts the Boot 3 `management.otlp.tracing.*` names but marks them deprecated. ## Step 3: Add the configuration class @@ -201,6 +203,10 @@ public class MapleAiObservationConfig { } ``` +Boot applies `ObservationFilter` beans to the registry automatically; nothing else to register. The filter runs when each observation stops, after Spring AI's own conventions. Why each part exists: the `chat_client` span is labeled `framework` by Spring AI and its name contains "chat", so without `invoke_agent` Maple counts it as a model call; Maple classifies spans without a known operation by name, so unrenamed advisor spans count `tool _calling ` as a tool call and `message_chat_memory` as a model call on every turn. `spring.ai.tools.observations.include-content` writes `spring.ai.tool.call.arguments/result`, which Maple does not read; the filter's `gen_ai.tool.call.*` keys are the ones read. + +Content notes: every `chat` span carries the whole conversation so far, so spans grow with long chats (don't cap them; see 2b). With `maple.ai.capture-content=false` no message/tool content leaves the process; to redact instead, mask values inside `message(...)`. The conversation id and agent names are sent regardless: keep personal data out of them. + Kotlin project: translate one to one (e.g. `ObservationFilter { context -> ...; context }`), same beans, same keys. Do NOT drop advisor observations with an `ObservationPredicate` instead of renaming them: on `.stream()` calls Spring AI takes the model span's parent from the Reactor context, which then holds the skipped (no-op) advisor observation, so the streamed `chat` span becomes its own trace outside the session (verified). @@ -223,8 +229,45 @@ Maple reads ONLY `spring.ai.chat.client.conversation.id` for Spring AI, on the ` - Tools that return error strings instead of throwing count as success. Point it out; don't change behavior unasked. - Sub-agents: one `ChatClient` per role, exposed to the orchestrator as a `@Tool(name = "", description = ...)` method that calls it. Build each from `builder.clone()` with its own `AGENT_NAME` param. Maple shows `execute_tool ` → worker `chat_client` as a delegation lane. - Own thread pools / `CompletableFuture` fan-out: propagate the observation to worker threads (Micrometer `context-propagation`: `ContextSnapshot`, a wrapped executor, or build the worker observation with `.parentObservation(parent)`). Otherwise each worker starts a new trace and becomes its own `trace:` session. +- Spring AI runs the tool calls of one model response sequentially on the calling thread, so context flows without help unless the app fans out itself. +- A `ChatClient` without an agent name gets no lane; its model and tool calls are drawn in the caller's lane. - Spring AI 2.0 has no native tool-approval/HITL mechanism; don't invent one. +Sub-agent example: + +```java +import org.springframework.ai.chat.client.ChatClient; +import org.springframework.ai.tool.annotation.Tool; +import org.springframework.stereotype.Component; + +@Component +public class Workers { + + private final ChatClient weather; + + public Workers(ChatClient.Builder builder, WeatherTools weatherTools) { + this.weather = builder.clone() + .defaultSystem("You answer weather questions using your tools.") + .defaultTools(weatherTools) + .defaultAdvisors(a -> a.param(MapleAiObservationConfig.AGENT_NAME, "weather_worker")) + .build(); + } + + @Tool(name = "weather_worker", description = "Ask the weather specialist about a city") + public String weatherWorker(String task) { + return weather.prompt().user(task).call().content(); + } + +} +``` + +## Step 5b: Tokens and cost (no action needed, explain if asked) + +- `chat` spans carry `gen_ai.usage.input_tokens`, `output_tokens`, and `cache_read.input_tokens` / `cache_creation.input_tokens` when the provider reports them; Maple reads all four. `chat_client` spans carry no usage, so nothing is double counted. +- Spring AI records the provider as `gen_ai.system`, derived from the client class, not the model: a Claude model behind OpenRouter via the OpenAI starter is labeled `openai`. Maple uses the provider only to decide whether input includes cached tokens, so only cache figures can be affected. +- No cost attribute is emitted; sessions show as unpriced (Maple never prices tokens). +- With model starters other than OpenAI, a `chat` span for a call without tools may carry no Spring AI marker, so Maple files it as a generic GenAI span (framework "Unidentified" on that span). Session, transcript and tokens are unaffected. + ## Step 6: Flush - Web app: nothing. Boot shuts down the `SdkTracerProvider` on context close, which flushes. @@ -247,7 +290,7 @@ Run one real conversation: 2-3 turns with the same conversation id including one - [ ] A tool that threw is counted as failed, with its message; successful tools are not. - [ ] Sub-agents appear as lanes under their agent names. - [ ] Cost shows as unpriced (Spring AI emits no cost; expected). -- [ ] App logs have no `Failed to publish metrics` / OTLP export errors, no `401`. +- [ ] App logs have no `Failed to publish metrics` / OTLP export errors, no `401` (`401` = wrong key, key from the other region, or the header property not reading `...headers.Authorization=Bearer `). ## Do not diff --git a/skills/maple-agent-tracing-strands/SKILL.md b/skills/maple-agent-tracing-strands/SKILL.md index f7d5f37f93..35632ac925 100644 --- a/skills/maple-agent-tracing-strands/SKILL.md +++ b/skills/maple-agent-tracing-strands/SKILL.md @@ -128,6 +128,7 @@ const agent = new Agent({ - Content capture is ON by default; Step 2's tokens only move it onto attributes. Nothing else to enable. - Redaction, only if the user asks or the repo handles regulated data: append `gen_ai_unredacted_attributes=` to `OTEL_SEMCONV_STABILITY_OPT_IN`. `;`-separated, single trailing `*` only. Covered keys: `gen_ai.input.messages`, `gen_ai.output.messages`, `gen_ai.system_instructions`, `gen_ai.tool.call.arguments`, `gen_ai.tool.call.result`. Empty list (`gen_ai_unredacted_attributes=`) redacts all to `[REDACTED]`. Example keeping replies only: `...,gen_ai_unredacted_attributes=gen_ai.output.*;gen_ai.tool.call.result`. - Not redactable: `gen_ai.tool.description`, `gen_ai.tool.json_schema`, `trace_attributes` values. +- Redacted values aren't JSON, so Maple leaves those transcript parts blank; tokens, tools and errors are unaffected. ## Step 5: Tools, errors, sub-agents @@ -149,6 +150,7 @@ telemetry.tracer_provider.shutdown() - Lambda: `force_flush()` at the end of each invocation, no `shutdown()`. - Existing provider (Step 0): flush that provider instead. +- A call cut off by `asyncio.wait_for` or task cancellation may export incomplete spans (upstream harness-sdk#3609); ended spans still flush. - TS: `await provider.forceFlush(); await provider.shutdown()` before exit. `setupTracer`'s own `beforeExit` flush does not run after `process.exit()`. ## Step 7: Verify @@ -161,12 +163,19 @@ Run one short conversation (2-3 messages, one tool call; plus a failing tool if - Transcript shows user messages, replies, tool calls with args and results. Empty transcript → `gen_ai_span_attributes_only` missing or set after the first `Agent(`. - Spans: `invoke_agent ` → `execute_event_loop_cycle` → `chat` / `execute_tool `; model id on `chat` spans. - Input/output tokens on every `chat` span, including streamed turns. Session total on the session detail page ≈ sum of `chat` spans, not several times more. The Agent Sessions LIST currently shows ~2x for Strands (Python and TS) even when setup is correct (Maple nets roll-ups only one level deep; `execute_event_loop_cycle` sits between `invoke_agent` and `chat`). Tell the user; don't try to fix it in their code. +- No time to first token in Maple: Strands emits `gen_ai.server.time_to_first_token` (ms), which Maple doesn't read. Expected. - "Reply length" check shows skipped (finish reason only inside output messages). Expected. - Failed tool counted as failed; successful tools not. - Sub-agents in separate lanes with their own names. - Cost shows "unpriced" (Strands emits no cost). Expected. - No duplicate `chat` spans per model call. +Symptoms: +- No model name on spans: a custom `Model` subclass that only implements `get_config()` gets no `gen_ai.request.model` (harness-sdk#4205). Give it a `config` dict with `model_id`. +- `401` from ingest: wrong key or key from the other region; header must be `Authorization=Bearer ` in `OTEL_EXPORTER_OTLP_HEADERS`. +- Token totals several times too high on the session page: `invoke_agent` reports the agent's lifetime usage. Add `gen_ai_use_latest_invocation_tokens` (Python) or construct the agent per request. +- Cache tokens zero with prompt caching on: `strands-agents` < 1.54 emits `cache_read_input_tokens` / `cache_write_input_tokens`, which Maple ignores. Upgrade. + If you can't reach Maple, set `OTEL_EXPORTER_OTLP_ENDPOINT` to a local collector or use `telemetry.setup_console_exporter()` once and check a `chat` span has `gen_ai.input.messages` as an ATTRIBUTE (not in `events`) and `session.id`. ## Do not diff --git a/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md index 5668fc76f8..cecbfc005d 100644 --- a/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md +++ b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md @@ -13,13 +13,13 @@ One conversation = one Maple Agent Session, one turn per `generate()`/`stream()` How it works: AI SDK 7 emits GenAI-semconv spans through `@ai-sdk/otel` (`invoke_agent ` → `step ` → `chat ` + `execute_tool `) on tracer `gen_ai`, once `registerTelemetry(new OpenTelemetry())` has run. Content is on by default. Maple detects the AI SDK by `ai.*` attributes on the `gen_ai`/`ai` scope and groups sessions by `gen_ai.conversation.id`, which the AI SDK never sets. You add it with `enrichSpan` from `runtimeContext`. -Known gaps (tell the user, don't try to fix): cost shows as "unpriced" (AI SDK emits no cost; Maple never prices tokens); the session token total (list and detail page) is 2x the real usage, because Maple doesn't net the `invoke_agent` total against its `chat` spans two levels down (per-`chat` counts and the per-model breakdown are correct); on AI SDK 5/6 the final assistant reply is missing from transcripts. +Known gaps (tell the user, don't try to fix): cost shows as "unpriced" (AI SDK emits no cost; Maple never prices tokens); the session token total (list and detail page) is 2x the real usage, because Maple doesn't net the `invoke_agent` total against its `chat` spans two levels down (per-`chat` counts and the per-model breakdown are correct); on AI SDK 5/6 the final assistant reply is missing from transcripts. If model calls go through OpenRouter, its Broadcast traces carry per-call cost and Maple matches them to the AI SDK `chat` spans by response id (see the OpenRouter guide). ## Step 0: Detect - `ai` version in `package.json` / lockfile: - `>= 7`: this skill. Upgrade to `^7.0.106` or newer if older (earlier 7.x leave spans open when a stream errors mid-read). Node.js >= 22 required. - - `5.x` / `6.x`: ask the user whether to upgrade to 7 (recommended; `npx @ai-sdk/codemod v7`). If not, go to "AI SDK 5/6" at the end. + - `5.x` / `6.x`: ask the user whether to upgrade to 7 (recommended; `npx @ai-sdk/codemod v7`). The codemod only renames `experimental_telemetry` to `telemetry`; you still install `@ai-sdk/otel`, call `registerTelemetry()`, and move the id from `metadata` (removed in 7) to `runtimeContext`, then drop any `ConversationIdProcessor`. If not, go to "AI SDK 5/6" at the end. - Mastra (`@mastra/core`) or another framework built on `ai`: stop, use that framework's skill instead. - Existing OpenTelemetry: search for `NodeSDK`, `NodeTracerProvider`, `registerOTel`, `@vercel/otel`, `Sentry.init`, `@langfuse/otel`, `LangfuseSpanProcessor`, `braintrust`, `registerTelemetry(`, `experimental_telemetry`, `telemetry:`. - An SDK/provider already exists: reuse it. Add one Maple exporting span processor to it. Never start a second SDK. @@ -164,8 +164,9 @@ await assistant.generate({ messages, options: { conversationId: chatId } }) ## Step 4: Content - On by default. Do not set `recordInputs`/`recordOutputs` unless the user asks for privacy; if they do, set them per call/agent in `telemetry` and tell them the transcript will be empty for those calls. +- Content is not redacted. For pattern-based redaction, suggest an OpenTelemetry Collector between the app and Maple, or `recordInputs`/`recordOutputs: false` on the sensitive calls. `telemetry: { isEnabled: false }` drops a call's spans entirely. - Never set `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` / `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT`: truncated JSON is dropped by Maple. -- Calls that send images/PDFs: warn the user they are recorded as base64 on every step; suggest `recordInputs: false` on those calls if payloads are large. +- Calls that send images/PDFs: warn the user they are recorded as base64 on every step; suggest `recordInputs: false` on those calls if payloads are large (exports failing with 413 are the symptom). ## Step 5: Tools, errors, sub-agents @@ -191,7 +192,7 @@ Run one real conversation: 2+ turns with the same id, one streamed, one tool cal - Transcript shows user messages, assistant replies and tool calls; turn labels are the user's messages. - Every `chat` span has input and output tokens, including the streamed turn; the streamed `chat` span has TTFT. The session token total is 2x the sum of the `chat` spans (known Maple gap, not a setup error; don't try to fix it); the per-model breakdown matches the `chat` spans. - Tool calls have name, arguments, result; a throwing tool is counted as failed with its message; successful tools are not failed. -- Sub-agents show as their own lanes named after their `functionId`. +- Sub-agents show as their own lanes named after their `functionId`. Span names carry the model id, not the agent name (`invoke_agent `); two agents on one model have identically named spans, and the name is on `gen_ai.agent.name`. - No attribute contains the provider API key or `Bearer `. - Cost shows as unpriced (expected). - The process exited cleanly and no turn is missing (flush ran). From f0e6d178305032db9add1c40a765e106d0e64e43 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 14:29:07 +0200 Subject: [PATCH 06/39] docs(agent-tracing): trim the Vercel AI SDK setup to three packages, keep the span processor variant for serverless --- .../docs/agent-tracing/vercel-ai-sdk.md | 48 ++++++++++--------- .../SKILL.md | 41 +++++++++------- 2 files changed, 50 insertions(+), 39 deletions(-) diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md index ee17288e01..4dcf433c2b 100644 --- a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -11,7 +11,7 @@ The Vercel AI SDK emits OpenTelemetry GenAI spans for every `generateText`, `str You have to get two things right. In AI SDK 7 nothing is traced until you call `registerTelemetry()` at startup, and the SDK has no conversation id, so you pass one on every call or each message becomes its own session. -Tested with `ai` 7.0.118, `@ai-sdk/otel` 1.0.118, OpenTelemetry JS 0.222.0 and `@vercel/otel` 2.1.3 on Node.js 26. You need `ai` 7.0.106 or newer and Node.js 22 or newer. On AI SDK 5 or 6, run `npx @ai-sdk/codemod v7` first; the skill has a fallback setup if you can't upgrade. +Tested with `ai` 7.0.118, `@ai-sdk/otel` 1.0.118, OpenTelemetry JS 0.222.0 and `@vercel/otel` 2.1.3 on Node.js 26 and Bun 1.3. You need `ai` 7.0.106 or newer and Node.js 22 or newer. On AI SDK 5 or 6, run `npx @ai-sdk/codemod v7` first; the skill has a fallback setup if you can't upgrade. ## Quick setup with a coding agent @@ -29,19 +29,19 @@ Your ingest key is in **Settings → Ingestion**. ## Install the packages -In AI SDK 7, OpenTelemetry support lives in `@ai-sdk/otel`. It creates the spans, and the OpenTelemetry SDK exports them: - ```bash -npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/api @opentelemetry/sdk-node \ - @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources +npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/sdk-node ``` +`@ai-sdk/otel` creates the spans and the OpenTelemetry Node SDK exports them. `@opentelemetry/api` comes with it as a peer dependency. + ## Point the exporter at Maple ```bash +export OTEL_SERVICE_NAME="support-agent" +export OTEL_RESOURCE_ATTRIBUTES="deployment.environment.name=production" export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" -export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" ``` For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. @@ -53,29 +53,18 @@ Create an `instrumentation.ts` and import it as the first line of your entry poi ```ts // instrumentation.ts import { OpenTelemetry } from "@ai-sdk/otel" -import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" -import { resourceFromAttributes } from "@opentelemetry/resources" import { NodeSDK } from "@opentelemetry/sdk-node" -import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" import { registerTelemetry } from "ai" -// Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS -export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) - -export const sdk = new NodeSDK({ - resource: resourceFromAttributes({ - "service.name": "support-agent", - "deployment.environment.name": process.env.NODE_ENV ?? "development", - }), - spanProcessors: [spanProcessor], -}) +// Reads OTEL_SERVICE_NAME, OTEL_RESOURCE_ATTRIBUTES and OTEL_EXPORTER_OTLP_* +export const sdk = new NodeSDK() sdk.start() registerTelemetry( new OpenTelemetry({ usage: true, runtimeContext: true, - // The conversation id, explained in the next section + // The conversation id, explained below enrichSpan: ({ runtimeContext }) => typeof runtimeContext?.conversationId === "string" ? { "gen_ai.conversation.id": runtimeContext.conversationId } @@ -86,7 +75,22 @@ registerTelemetry( Call `registerTelemetry()` exactly once. Each call adds another integration, and each one emits its own copy of every span. Keep `usage: true` and `runtimeContext: true`: Maple uses the `ai.*` attributes they add to recognize AI SDK spans. -If your app already starts an OpenTelemetry SDK (auto-instrumentation, Sentry, your own `NodeTracerProvider`), don't start a second one. Add the `BatchSpanProcessor` to the existing provider and keep the `registerTelemetry()` call. +### Serverless, or an app that already uses OpenTelemetry + +Two setups need a handle on the span processor: a serverless handler that flushes after every invocation, and an app that already starts its own OpenTelemetry SDK (auto-instrumentation, Sentry, your own `NodeTracerProvider`), where you shouldn't start a second one. Install the exporter packages and create the processor yourself: + +```bash +npm install @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto +``` + +```ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" + +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) +``` + +Pass it as `new NodeSDK({ spanProcessors: [spanProcessor] })`, or add it to your existing provider instead of creating a `NodeSDK`. Keep the `registerTelemetry()` call either way. ### Next.js @@ -183,7 +187,7 @@ Prompts and replies are recorded by default. To keep them out of Maple for a cal ## Flush in scripts and serverless functions -`BatchSpanProcessor` exports every few seconds, so a short-lived process can exit first. In a script, call `await sdk.shutdown()` in a `finally` block before exiting. In a serverless handler that is reused between invocations, call `await spanProcessor.forceFlush()` in a `finally` instead, so the SDK keeps running. +The SDK exports spans in batches every few seconds, so a short-lived process can exit first. In a script, call `await sdk.shutdown()` in a `finally` block before exiting. In a serverless handler that is reused between invocations, call `await spanProcessor.forceFlush()` in a `finally` instead (see [the variant above](#serverless-or-an-app-that-already-uses-opentelemetry)), so the SDK keeps running. Read streams to the end (`await result.consumeStream()`) before flushing: a stream's spans only end when it has been read. On Vercel, `@vercel/otel` flushes after each request, but queue consumers and cron jobs need their own `forceFlush()`. diff --git a/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md index cecbfc005d..1e37b8aa2a 100644 --- a/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md +++ b/skills/maple-agent-tracing-vercel-ai-sdk/SKILL.md @@ -38,37 +38,26 @@ Known gaps (tell the user, don't try to fix): cost shows as "unpriced" (AI SDK e ## Step 2a: Install and init (Node.js) ```bash -npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/api @opentelemetry/sdk-node \ - @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto @opentelemetry/resources +npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/sdk-node ``` -Use the repo's package manager. Env (or the repo's equivalent): +Use the repo's package manager (`@opentelemetry/api` arrives as a peer of `sdk-node`; add it explicitly only if the package manager doesn't install peers). Works on Node.js 22+ and Bun. Env (or the repo's equivalent): ```bash +OTEL_SERVICE_NAME=support-agent +OTEL_RESOURCE_ATTRIBUTES=deployment.environment.name=production OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev OTEL_EXPORTER_OTLP_HEADERS=Authorization=Bearer -OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf ``` -Create `instrumentation.ts` (adapt service name/environment; to inline instead of env, pass `{ url: "https://ingest.maple.dev/v1/traces", headers: { authorization: "Bearer " } }` to `OTLPTraceExporter`): +`NodeSDK()` with no span processors builds a batched OTLP http/protobuf exporter from these variables. Create `instrumentation.ts`: ```ts import { OpenTelemetry } from "@ai-sdk/otel" -import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" -import { resourceFromAttributes } from "@opentelemetry/resources" import { NodeSDK } from "@opentelemetry/sdk-node" -import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" import { registerTelemetry } from "ai" -export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) - -export const sdk = new NodeSDK({ - resource: resourceFromAttributes({ - "service.name": "support-agent", - "deployment.environment.name": process.env.NODE_ENV ?? "development", - }), - spanProcessors: [spanProcessor], -}) +export const sdk = new NodeSDK() sdk.start() registerTelemetry( @@ -83,6 +72,24 @@ registerTelemetry( ) ``` +Use the explicit span processor variant instead when either applies: + +- Serverless handler (Lambda, Cloud Run jobs, queue consumers, cron, Vercel Workflow steps): needs `spanProcessor.forceFlush()` per invocation; `NodeSDK` only has `shutdown()`. +- The repo already starts an OpenTelemetry SDK/provider: add the processor to it, don't create a second `NodeSDK`. +- Inlining the key instead of env: pass `{ url: "https://ingest.maple.dev/v1/traces", headers: { authorization: "Bearer " } }` to `OTLPTraceExporter`. + +```bash +npm install @opentelemetry/sdk-trace-base @opentelemetry/exporter-trace-otlp-proto +``` + +```ts +import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" +import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base" + +export const spanProcessor = new BatchSpanProcessor(new OTLPTraceExporter()) +export const sdk = new NodeSDK({ spanProcessors: [spanProcessor] }) +``` + - `import "./instrumentation"` as the FIRST line of every entry point (server, worker, CLI). `registerTelemetry` must run before the first AI SDK call. - `usage: true` + `runtimeContext: true` are required, not optional: they put `ai.*` keys on `invoke_agent`/`chat` spans, which is how Maple labels the vendor "Vercel AI SDK", picks the AI SDK token convention and reads TTFT. - Do not pass `tracer:` to `OpenTelemetry` unless reusing a provider requires it; if you must, use `provider.getTracer("gen_ai")`. Other scope names break detection. From 811e642ba6b623c5276d1c6974ce3523a6f615ea Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 14:29:53 +0200 Subject: [PATCH 07/39] docs(agent-tracing): drop package trivia from the Vercel AI SDK install step --- apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md | 2 -- 1 file changed, 2 deletions(-) diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md index 4dcf433c2b..42e29e30aa 100644 --- a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -33,8 +33,6 @@ Your ingest key is in **Settings → Ingestion**. npm install ai@^7.0.106 @ai-sdk/otel @opentelemetry/sdk-node ``` -`@ai-sdk/otel` creates the spans and the OpenTelemetry Node SDK exports them. `@opentelemetry/api` comes with it as a peer dependency. - ## Point the exporter at Maple ```bash From 081f41339beaf501e6ab3e609d9b727ecae5fdf7 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 14:35:21 +0200 Subject: [PATCH 08/39] docs(agent-tracing): say when to use the OpenTelemetry guide instead of listing languages --- apps/landing/src/content/docs/agent-tracing/opentelemetry.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md index 8eb70beebf..1960db1524 100644 --- a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md +++ b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md @@ -7,7 +7,7 @@ navLabel: "Any language (OTel GenAI)" icon: "opentelemetry" --- -If your agent loop is hand-written, runs on a framework without OpenTelemetry support, or lives in Go, Rust, Ruby or Elixir, you write the agent spans yourself. Maple needs three kinds: an `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. The details that break most setups are the conversation id and sending messages as JSON strings. +Use this guide when no other guide covers your agent, for example a hand-written agent loop. You write the agent spans yourself with any OpenTelemetry SDK. Maple needs three kinds: an `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. The details that break most setups are the conversation id and sending messages as JSON strings. Tested with the OpenTelemetry JS SDK 2.11 on Node.js 26 and the Python SDK 1.45 on Python 3.14, calling OpenRouter, against the [GenAI conventions](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. From 403a6de2300102f671f75e4a5d71db7532b1b61d Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 14:39:56 +0200 Subject: [PATCH 09/39] docs(agent-tracing): cut claims a reader setting up tracing doesn't need --- .../content/docs/agent-sessions/overview.md | 14 +++--- .../src/content/docs/agent-tracing.mdx | 8 ++-- .../src/content/docs/agent-tracing/agno.md | 35 +++++--------- .../docs/agent-tracing/claude-agent-sdk.md | 30 +++++------- .../src/content/docs/agent-tracing/crewai.md | 47 ++++++++---------- .../src/content/docs/agent-tracing/dspy.md | 35 +++++--------- .../content/docs/agent-tracing/google-adk.md | 33 +++++-------- .../content/docs/agent-tracing/haystack.md | 40 ++++++---------- .../content/docs/agent-tracing/langchain.md | 39 +++++++-------- .../src/content/docs/agent-tracing/litellm.md | 40 +++++++--------- .../content/docs/agent-tracing/llamaindex.md | 48 +++++++------------ .../src/content/docs/agent-tracing/mastra.md | 29 +++++------ .../microsoft-agent-framework.md | 37 +++++++------- .../docs/agent-tracing/openai-agents.md | 38 ++++++--------- .../content/docs/agent-tracing/openrouter.md | 24 ++++------ .../docs/agent-tracing/opentelemetry.md | 39 +++++++-------- .../docs/agent-tracing/provider-sdks.md | 30 ++++-------- .../content/docs/agent-tracing/pydantic-ai.md | 25 ++++------ .../content/docs/agent-tracing/smolagents.md | 37 ++++++-------- .../content/docs/agent-tracing/spring-ai.md | 30 ++++-------- .../src/content/docs/agent-tracing/strands.md | 32 ++++++------- .../docs/agent-tracing/vercel-ai-sdk.md | 24 ++++------ 22 files changed, 282 insertions(+), 432 deletions(-) diff --git a/apps/landing/src/content/docs/agent-sessions/overview.md b/apps/landing/src/content/docs/agent-sessions/overview.md index 082f5320a1..46d8ac7d3c 100644 --- a/apps/landing/src/content/docs/agent-sessions/overview.md +++ b/apps/landing/src/content/docs/agent-sessions/overview.md @@ -6,7 +6,7 @@ order: 1 navLabel: "Overview" --- -A chat backend handles each user message as its own request, so a ten-message conversation is ten traces. **Agent Sessions** groups those traces back into one conversation and shows it turn by turn. To send your agent's traces, pick your framework in [Trace your AI agent](/docs/agent-tracing). +Each user message in a conversation is usually its own trace. **Agent Sessions** groups those traces into one conversation and shows it turn by turn. To send your agent's traces, pick your framework in [Trace your AI agent](/docs/agent-tracing). ## Sessions, turns and calls @@ -28,16 +28,16 @@ The example below is a support agent where the customer asks to change a deliver
The overview. The failed tool call leads the page: update_shipping_address returned unsupported_destination in turn 2.
-The **overview** splits the session's wall clock into model time, tool time and idle time, and totals cost and tokens per model. Here the agent was busy for 14 seconds of a 2 minute 25 second session; the rest was the customer typing. +The **overview** splits the session's wall clock into model time, tool time and idle time, and totals cost and tokens per model. -Below that is the verdict (completed, completed with warnings, or failed, with the span that ended it) and the checks behind it, failing ones first. The checks cover completion, context window, reply length, refusals, rate limits, provider errors, tool errors, tool arguments, tool timeouts, tool availability, repeated calls, stalls, prompt cache and structured output. A check that needs data your instrumentation doesn't record says it was skipped and what to capture. +Below that is the verdict (completed, completed with warnings, or failed, with the span that ended it) and the checks behind it, failing ones first. A check that needs data your instrumentation doesn't record says it was skipped and what to capture.
The transcript view of a session: system instructions, user and assistant messages in sequence, each model call annotated with its model, tokens, cost and finish reason, and a tool call row with its latency and payload sizes.
The transcript. Each model call carries its model, tokens, cost and finish reason; tool calls sit where the model made them.
-The **transcript** is the conversation as the model saw it: system instructions, user and assistant messages, and tool calls with arguments and results. It needs message content on your spans, which most instrumentations leave off by default. Each framework guide shows the switch. +The **transcript** is the conversation as the model saw it: system instructions, user and assistant messages, and tool calls with arguments and results. It needs message content on your spans, which most instrumentations leave off by default; each framework guide shows the switch.
The trace view of a session: three turns, each with an invoke_agent span, chat spans labeled with their model and token counts, and execute_tool spans, on a time axis with the idle gaps between turns removed. One tool span is marked with its error type. @@ -53,8 +53,6 @@ The **trace** view puts every span of every turn on one time axis, with the idle The **list** has one row per session with its model, duration, call counts, tokens, cost and errors. Sort by cost to find expensive conversations, or filter to sessions that called a given tool. -A session view loads up to 2,000 spans. Longer sessions show the first 2,000 and say so. - ## Find the tools that fail The **Tools** tab covers every tool call across all sessions. @@ -74,7 +72,7 @@ The **Tools** tab covers every tool call across all sessions.
What exactly happened? An error group opens on the failed calls themselves: arguments on the left, the result on the right, and a link to the trace.
-A tool call counts as failed when its span has an `ERROR` status or an `error.type` attribute. Failures with the same message are grouped, with ids and numbers masked, so a thousand `order 12345 not found` errors form one group. +A tool call counts as failed when its span has an `ERROR` status or an `error.type` attribute. Failures with the same message, ignoring ids and numbers, form one group. ## Query sessions from your coding agent @@ -88,4 +86,4 @@ The [MCP server](/docs/reference/mcp) exposes the same data. `list_agent_session - **Token totals look doubled.** Two instrumentations recorded the same model call, usually the framework's and a provider SDK instrumentor. Turn one off. - **Nothing appears at all.** Check **Explore → Traces** for the service first. No traces there means the exporter isn't reaching Maple, often a short-lived script that exits before flushing. Traces there but no session means the framework's tracing isn't on. -A framework shown as **Unidentified** still gets sessions, transcripts and tools. If yours isn't covered, send the framework and a sample trace to [support@maple.dev](mailto:support@maple.dev) or [Discord](https://discord.gg/BnXjKuwJqP). Until then, [the OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry) works for any agent in any language. +A framework shown as **Unidentified** still gets sessions, transcripts and tools. If yours has no guide, [the OpenTelemetry GenAI guide](/docs/agent-tracing/opentelemetry) works for any agent, and a sample trace sent to [support@maple.dev](mailto:support@maple.dev) or [Discord](https://discord.gg/BnXjKuwJqP) helps us add one. diff --git a/apps/landing/src/content/docs/agent-tracing.mdx b/apps/landing/src/content/docs/agent-tracing.mdx index d82b628b57..221da1ac6f 100644 --- a/apps/landing/src/content/docs/agent-tracing.mdx +++ b/apps/landing/src/content/docs/agent-tracing.mdx @@ -9,7 +9,7 @@ navLabel: "Overview" import GuideGrid from "../../components/docs/GuideGrid.astro" import { AGENT_GUIDE_SECTIONS } from "../../lib/agent-tracing-guides" -Each guide below sets up one framework so every conversation shows up in Maple as one [Agent Session](/docs/agent-sessions/overview), with its transcript, model and tool calls, tokens and failures. Maple ingests OTLP directly, so you turn on the framework's own OpenTelemetry instrumentation and point it at Maple. There is no Maple SDK. +Each guide sets up one framework so every conversation shows up in Maple as one [Agent Session](/docs/agent-sessions/overview). You turn on the framework's own OpenTelemetry instrumentation and point it at Maple; there is no Maple SDK. ## Quick setup with a coding agent @@ -34,7 +34,7 @@ Your ingest key is in **Settings → Ingestion**. EU organizations should say EU ))} -If you use a gateway like OpenRouter or LiteLLM together with a framework, instrument only the framework, or every model call is recorded twice. The [OpenRouter guide](/docs/agent-tracing/openrouter) covers the one setup where both work together. +If you use a gateway like OpenRouter or LiteLLM together with a framework, instrument only the framework, or every model call is recorded twice. The [OpenRouter guide](/docs/agent-tracing/openrouter) covers the one exception. ## Connection details @@ -47,10 +47,10 @@ export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" export OTEL_SERVICE_NAME="support-agent" ``` -Maple accepts OTLP over HTTP only, so set the protocol to `http/protobuf` if your exporter defaults to gRPC. If you pass a traces endpoint in code or through `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`, include `/v1/traces` yourself. +Maple doesn't accept gRPC, so set the protocol to `http/protobuf` if your exporter defaults to it. A traces endpoint passed in code or through `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` must end in `/v1/traces`. Keep agent traffic at 100% sampling, or sampled-out turns leave gaps in the session. See [Sampling and throughput](/docs/concepts/sampling-throughput). ## Not listed? -Anything that emits the OpenTelemetry GenAI conventions works, in any language. [Any language](/docs/agent-tracing/opentelemetry) lists what Maple reads. OpenInference and OpenLLMetry instrumentations also work, with the framework shown as **Unidentified**. Tell us what you run at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP) and we'll add a guide. +Anything that emits the OpenTelemetry GenAI conventions works; [Any language](/docs/agent-tracing/opentelemetry) lists what Maple reads. OpenInference and OpenLLMetry instrumentations also work, shown as **Unidentified**. Tell us what you run at [support@maple.dev](mailto:support@maple.dev) or on [Discord](https://discord.gg/BnXjKuwJqP) and we'll add a guide. diff --git a/apps/landing/src/content/docs/agent-tracing/agno.md b/apps/landing/src/content/docs/agent-tracing/agno.md index 42279d4c52..1f1f08662c 100644 --- a/apps/landing/src/content/docs/agent-tracing/agno.md +++ b/apps/landing/src/content/docs/agent-tracing/agno.md @@ -7,9 +7,9 @@ navLabel: "Agno" icon: "agno" --- -`openinference-instrumentation-agno` traces every Agno agent and team run, model call and tool call. Agno's own `setup_tracing()` uses the same instrumentor but writes spans to your AgentOS database, so to reach Maple you install it yourself with an OTLP exporter. +This guide sends Agno's OpenInference spans to Maple with an OTLP exporter. Agno's own `setup_tracing()` writes to your AgentOS database, not to Maple. -The one thing to get right is `session_id`. Without it, Agno generates an id once per `Agent` object, and a server with one shared agent puts every user's conversation into the same Maple session. +Pass `session_id` on every run. Without it, a server with one shared `Agent` puts every user's conversation into the same Maple session. Tested with Agno 3.0.11 and `openinference-instrumentation-agno` 1.0.10 on Python 3.10 to 3.14. @@ -43,7 +43,7 @@ export OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf export AGNO_TELEMETRY=false ``` -EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. `AGNO_TELEMETRY=false` only turns off Agno's anonymous usage pings. +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. `AGNO_TELEMETRY=false` turns off Agno's anonymous usage pings. ## Initialize tracing @@ -64,16 +64,15 @@ trace.set_tracer_provider(provider) AgnoInstrumentor().instrument( tracer_provider=provider, - # Also write gen_ai.* attributes, which Maple's session detail page reads config=TraceConfig(enable_genai_semconv=True), ) ``` -Keep `enable_genai_semconv=True`. Without it the sessions list still shows tokens, but the session page has no transcript. If you can't change the `instrument()` call, set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` instead. +Without `enable_genai_semconv=True` the session page has no transcript. If you can't change the `instrument()` call, set `OPENINFERENCE_ENABLE_GENAI_SEMCONV=true` instead. ### Keep the AgentOS traces view -`AgentOS(tracing=True)` and `setup_tracing(db=...)` skip their setup once `tracing.py` has registered a provider, so the AgentOS traces view stops filling. To keep it, add Agno's database exporter to your provider, with the same `db` you give AgentOS: +Once `tracing.py` runs, `AgentOS(tracing=True)` and `setup_tracing(db=...)` stop filling the AgentOS traces view. To keep it, add Agno's database exporter to your provider, with the same `db` you give AgentOS: ```py from agno.tracing.exporter import DatabaseSpanExporter @@ -83,7 +82,7 @@ provider.add_span_processor(BatchSpanProcessor(DatabaseSpanExporter(db=db))) ## Pass session_id on every run -Each `run()` or `arun()` starts a new trace. Maple joins them into a session by the `session.id` Agno puts on the run span, taken from the `session_id` argument: +Each `run()` or `arun()` is its own trace. Pass `session_id` to group them into one session: ```py from agno.agent import Agent @@ -113,19 +112,19 @@ async def chat_stream(conversation_id: str, user_id: str, message: str): yield event.content ``` -Use the conversation id your app already stores. Agno uses the same id to load history from the agent's `db`, so tracing and memory stay in sync. Pass it to `team.run()`, workflows and `continue_run()` as well. Team members inherit the team's id, and a resumed human-in-the-loop run shows up as a second turn in the same session. +Use the conversation id your app already stores, and pass it to `team.run()`, workflows and `continue_run()` as well. ## Teams, failed tools and content -Give every `Agent` and `Team` a `name=`. Maple opens a lane for each named team member, and unnamed ones show up as `Agent.run` or `Team.run` with no lane. +Give every `Agent` and `Team` a `name=`. Unnamed ones show up as `Agent.run` or `Team.run` with no lane of their own. -A tool that raises is marked failed, even though Agno hands the error back to the model and the run continues. A tool that returns an error string counts as a success. +A tool that raises is marked failed. A tool that returns an error string counts as a success. To keep content out of Maple, set `OPENINFERENCE_HIDE_INPUT_MESSAGES`, `OPENINFERENCE_HIDE_OUTPUT_MESSAGES`, `OPENINFERENCE_HIDE_INPUTS` and `OPENINFERENCE_HIDE_OUTPUTS` to `true` before the instrumentor starts. Tool arguments are still exported; removing them needs a `redaction` processor in an OpenTelemetry Collector. ## Flush before short-lived processes exit -`BatchSpanProcessor` exports every 5 seconds. Long-running servers flush on shutdown, but scripts, CLIs, notebooks, workers and serverless handlers need an explicit flush: +Scripts, notebooks, workers and serverless handlers need an explicit flush: ```py from tracing import provider @@ -137,27 +136,19 @@ finally: provider.shutdown() # scripts: flush and stop at the end of the process ``` -On AWS Lambda and similar platforms, call only `force_flush()`, since the next invocation reuses the process. - ## Check that it works -Run a conversation of two or three turns with one `session_id`, including a tool call, and open **Agent Sessions** filtered by your service name. You should see one session with your `session_id` as its id, framework **Agno**, one turn per `run()`, a transcript, and model calls such as `OpenRouter.invoke` with tokens. A session named `trace:...` means the run span had no `session.id`. - -Cost appears in the sessions list when your provider returns one (OpenRouter does). The session page shows it as not reported for Agno, and providers without prices show as **unpriced**. +Run a conversation of two or three turns with one `session_id`, including a tool call, and open **Agent Sessions** filtered by your service name. You should see one session with your `session_id` as its id, framework **Agno**, one turn per `run()`, a transcript, and model calls such as `OpenRouter.invoke` with tokens. A session named `trace:...` means the run had no `session_id`. Cost shows in the sessions list only when your provider returns one, as OpenRouter does. ## Troubleshooting - **Nothing arrives in Maple.** `setup_tracing()` or `AgentOS(tracing=True)` ran before `tracing.py` and the spans went to the AgentOS database. Import `tracing` first. - **Every user is in one giant session.** The shared `Agent` runs without `session_id=`. Pass it on every `run()`, `arun()` and `continue_run()`. - **Every turn is its own session.** You pass a new id per request, often a fresh `uuid4()`. Use the stored conversation id. -- **Tokens in the list, no transcript on the session page.** `enable_genai_semconv` is off, or other code called `instrument()` first (the second call is ignored). -- **An agent run lands inside the previous team run's trace.** Instrumentor 1.0.10 leaves the team span attached to the thread or task. In scripts and workers, run each team run in its own context with `asyncio.create_task(...)` or `contextvars.copy_context().run(...)`. +- **Tokens in the list, no transcript on the session page.** `enable_genai_semconv` is off, or other code called `instrument()` first. +- **An agent run lands inside the previous team run's trace.** In scripts and workers, run each team run in its own context with `asyncio.create_task(...)` or `contextvars.copy_context().run(...)`. - **Every model call appears twice.** Remove the second instrumentor, usually `openinference-instrumentation-openai`, OpenLIT or Phoenix's `register(auto_instrument=True)`. ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [Agent tracing guides](/docs/agent-tracing) -- [Agno tracing documentation](https://docs.agno.com/agent-os/tracing/overview) -- [openinference-instrumentation-agno on PyPI](https://pypi.org/project/openinference-instrumentation-agno/) -- [OpenInference configuration variables](https://arize.com/docs/phoenix/tracing/how-to-tracing/advanced/masking-span-attributes) diff --git a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md index fe846738a0..9478feefc3 100644 --- a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md @@ -7,9 +7,9 @@ navLabel: "Claude Agent SDK & Claude Code" icon: "claude" --- -Every Agent SDK `query()` starts the Claude Code CLI as a child process, and the CLI exports OpenTelemetry spans for each turn, model request and tool call. You configure it with environment variables, the same ones for the TypeScript SDK, the Python SDK and `claude` in your terminal. There is nothing else to install. +The Claude Agent SDK and Claude Code export OpenTelemetry spans for each turn, model request and tool call once you set a few environment variables. There is nothing to install. -Two things need care. Spans only exist with `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`, and in the SDK every `query()` starts a new session unless you resume the conversation's session id. +Spans only exist with `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`, and in the SDK every `query()` starts a new session unless you resume the conversation's session id. Tested with `@anthropic-ai/claude-agent-sdk` 0.3.283 (TypeScript), `claude-agent-sdk` 0.2.160 (Python) and Claude Code 2.1.283. @@ -29,11 +29,11 @@ Your ingest key is under **Settings → Ingestion**. EU organizations should say ## Pass the telemetry variables to the CLI -Without `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` the CLI exports metrics and logs but no spans, and Agent Sessions stays empty. `OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf` is also required, because Claude Code has no default protocol. EU organizations use `https://ingest.eu.maple.dev`. +`OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf` is required, because Claude Code has no default protocol. EU organizations use `https://ingest.eu.maple.dev`. -The snippets below also drop any inherited `TRACEPARENT`. Claude Code's Bash tool and most CI systems set one, and without this your agent's turns nest inside that outer trace. +The snippets below drop any inherited `TRACEPARENT`, which Claude Code's Bash tool and most CI systems set. Otherwise your agent's turns nest inside that outer trace. -Never set an exporter to `console` in an SDK app. The SDK reads the CLI's standard output as its message stream. +Never set an exporter to `console` in an SDK app. It breaks the SDK's message stream. ### TypeScript Agent SDK @@ -105,9 +105,9 @@ MAPLE_ENV = { } ``` -Pass the env on every `query()` call, as in the next section. You can also set the same variables in your Dockerfile or deployment manifest and skip `env`, as long as no `TRACEPARENT` is set there. +Pass the env on every `query()` call, as in the next section. Or set the same variables in your Dockerfile or deployment manifest and skip `env`, as long as no `TRACEPARENT` is set there. -An `env` block in `~/.claude/settings.json` or the project's `.claude/settings.json` overrides `options.env`. A server app that doesn't need settings files can pass `settingSources: []` (Python `setting_sources=[]`). +An `env` block in `~/.claude/settings.json` or the project's `.claude/settings.json` overrides `options.env`. Server apps can pass `settingSources: []` (Python `setting_sources=[]`) to skip settings files. ### Claude Code in your terminal, IDE or desktop app @@ -131,11 +131,11 @@ Put the variables under `env` in `~/.claude/settings.json`, then start a new `cl } ``` -Use your user settings, your shell or [managed settings](https://code.claude.com/docs/en/managed-settings). Since 2.1.282, Claude Code ignores these variables in a repository's `.claude/settings.json`. Terminal sessions report the service `claude-code`, and each `claude` session is one Maple session. +A repository's `.claude/settings.json` can't set these variables. Use your user settings, your shell or [managed settings](https://code.claude.com/docs/en/managed-settings). Terminal sessions report the service `claude-code`. ## Resume the session on every turn -Maple groups Claude Code spans by `session.id`, which is the Claude session id. A chat backend that calls `query()` once per message without resuming gets one Maple session per message, and the agent forgets the previous message. +A chat backend that calls `query()` once per message without resuming gets one Maple session per message, and the agent forgets the previous message. Store a UUID with each conversation. Pass it as `sessionId` on the first turn and as `resume` on every turn after: @@ -177,23 +177,21 @@ async def reply(conversation: dict, text: str) -> str | None: return None ``` -`resume` reads the transcript from `~/.claude/projects/` on the machine that ran the earlier turns. If the next message can land on another host, use the SDK's [`sessionStore`](https://code.claude.com/docs/en/agent-sdk/session-storage) option. A Python `ClaudeSDKClient`, or a TypeScript `query()` fed an async iterable, keeps one session for all its turns and needs none of this. +`resume` reads the earlier turns from `~/.claude/projects/` on the same machine. If messages can land on different hosts, use the SDK's [`sessionStore`](https://code.claude.com/docs/en/agent-sdk/session-storage) option. A Python `ClaudeSDKClient`, or a TypeScript `query()` fed an async iterable, keeps one session for all its turns and needs none of this. ## Choose what content to record -Claude Code redacts content by default. `OTEL_LOG_USER_PROMPTS=1` records prompts, which title each turn. `OTEL_LOG_TOOL_DETAILS=1` records Bash commands, file paths and full tool error messages. `OTEL_LOG_TOOL_CONTENT=1` records what each tool returned. - -`OTEL_LOG_TOOL_CONTENT=1` sends whatever files Claude reads and whatever commands print, secrets included. Turn on only what your Maple organization is allowed to store. +Claude Code redacts content by default. `OTEL_LOG_USER_PROMPTS=1` records prompts, which title each turn. `OTEL_LOG_TOOL_DETAILS=1` records Bash commands, file paths and tool error messages. `OTEL_LOG_TOOL_CONTENT=1` records tool results, including any secrets in files Claude reads or in command output. Turn on only what your Maple organization is allowed to store. ## Let each turn finish exporting -The CLI exits at the end of each turn, and its final flush has a short timeout. Let every `query()` loop reach its `result` message; breaking out early, calling `close()` or aborting kills the CLI before it flushes. In a script, keep the process alive about 5 seconds after the last `query()`. On serverless platforms, finish the loop before returning the response. +Let every `query()` loop reach its `result` message. Breaking out early, calling `close()` or aborting kills the CLI before it exports the turn. In a script, keep the process alive about 5 seconds after the last `query()`. On serverless platforms, finish the loop before returning the response. ## Check that it works Run a two-turn conversation through `reply()` with at least one tool call, then open **Agent Sessions**. You should see one session with the framework **Claude Agent SDK**, one turn per message titled with the prompt, model calls with their tokens, and the tool calls by name. -The transcript has no assistant replies and cost shows as unpriced. Claude Code puts both only on log events, which you can search under **Logs** when the logs exporter is on. +The transcript has no assistant replies and cost shows as unpriced. Both are on log events under **Logs** when the logs exporter is on. ## Troubleshooting @@ -207,7 +205,5 @@ The transcript has no assistant replies and cost shows as unpriced. Claude Code ## Related - [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. -- [Trace your AI agent](/docs/agent-tracing): guides for other frameworks. - [Provider SDKs](/docs/agent-tracing/provider-sdks): tracing direct calls with the Anthropic SDK instead of the Agent SDK. - Claude Code [Monitoring reference](https://code.claude.com/docs/en/monitoring-usage): every variable, span attribute and event. -- Agent SDK [Observability with OpenTelemetry](https://code.claude.com/docs/en/agent-sdk/observability). diff --git a/apps/landing/src/content/docs/agent-tracing/crewai.md b/apps/landing/src/content/docs/agent-tracing/crewai.md index 20f8e9023a..3a3383a852 100644 --- a/apps/landing/src/content/docs/agent-tracing/crewai.md +++ b/apps/landing/src/content/docs/agent-tracing/crewai.md @@ -7,13 +7,11 @@ navLabel: "CrewAI" icon: "crewai" --- -CrewAI doesn't export traces to your backend on its own. OpenInference's `openinference-instrumentation-crewai` records crews, agents and tools, and a second instrumentor for the SDK CrewAI calls records the model calls, prompts and tokens. Without that second instrumentor you get no transcript, and without a session id every `kickoff()` is its own session. - -Tested with CrewAI 1.15, `openinference-instrumentation-crewai` 1.1.18 and `openinference-instrumentation-openai` 0.1.61 on Python 3.10 to 3.13. +CrewAI needs two OpenInference instrumentors: `openinference-instrumentation-crewai` for crews, agents and tools, and one for your model provider, which records prompts and tokens. You also wrap every `kickoff()` in a conversation id so a chat becomes one session. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-crewai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-crewai) skill, which contains every step of this guide. +Paste this prompt into Claude Code, Codex, Cursor or another coding agent. It installs the [maple-agent-tracing-crewai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-crewai) skill and follows it. ```text Set up Maple agent tracing for CrewAI in this project. @@ -35,15 +33,15 @@ pip install "crewai>=1.15" "openinference-instrumentation-crewai>=1.1.18" \ Pick the model instrumentor by the model string you pass to `LLM(...)`: -| Model string | SDK CrewAI calls | Instrumentor | -| --- | --- | --- | -| `openai/…`, `openrouter/…`, `deepseek/…`, `ollama/…`, `custom_openai=True`, or a bare name like `gpt-4.1-mini` | `openai` | `openinference-instrumentation-openai` | -| `anthropic/…` or a bare `claude-…` | `anthropic` | `openinference-instrumentation-anthropic` | -| `gemini/…` or a bare `gemini-…` | `google-genai` | `openinference-instrumentation-google-genai` | -| `bedrock/…` | `boto3` | `openinference-instrumentation-bedrock` | -| Anything else (needs `crewai[litellm]`) | `litellm` | `openinference-instrumentation-litellm` | +| Model string | Instrumentor | +| --- | --- | +| `openai/…`, `openrouter/…`, `deepseek/…`, `ollama/…`, `custom_openai=True`, or a bare name like `gpt-4.1-mini` | `openinference-instrumentation-openai` | +| `anthropic/…` or a bare `claude-…` | `openinference-instrumentation-anthropic` | +| `gemini/…` or a bare `gemini-…` | `openinference-instrumentation-google-genai` | +| `bedrock/…` | `openinference-instrumentation-bedrock` | +| Anything else (needs `crewai[litellm]`) | `openinference-instrumentation-litellm` | -Install only the ones your crews use. The LiteLLM instrumentor records nothing for the native providers in the first four rows. +Install only the ones your crews use. The LiteLLM instrumentor records nothing for the first four rows. ## Point the exporter at Maple @@ -58,7 +56,7 @@ export CREWAI_TRACING_ENABLED=false EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, it has to end in `/v1/traces`. -The last two variables turn off CrewAI's anonymous analytics and its own trace uploader, whose first-run prompt waits for input at the end of a run. Don't use `OTEL_SDK_DISABLED=true` for this, because it disables your Maple traces too. +The last two turn off CrewAI's own telemetry and a first-run prompt that blocks the process at exit. Don't use `OTEL_SDK_DISABLED=true` instead, because it disables your Maple traces too. ## Initialize tracing @@ -76,11 +74,7 @@ from opentelemetry.sdk.trace.export import BatchSpanProcessor class CrewAIAgentNames(SpanProcessor): - """Copies each CrewAI agent's role to gen_ai.agent.name, which Maple uses for agent lanes.""" - def on_start(self, span, parent_context=None): - # The instrumentor records the role (graph.node.id) just after the agent span starts, - # so name the agent span when its first child starts, while it's still open. parent = trace.get_current_span(parent_context) attrs = getattr(parent, "attributes", None) or {} role = attrs.get("graph.node.id") @@ -88,7 +82,7 @@ class CrewAIAgentNames(SpanProcessor): parent.set_attribute("gen_ai.agent.name", role) -provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider = TracerProvider() provider.add_span_processor(CrewAIAgentNames()) provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) @@ -98,9 +92,9 @@ CrewAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_de OpenAIInstrumentor().instrument(tracer_provider=provider, config=config, skip_dep_check=True) ``` -Pass `config` with `enable_genai_semconv=True` to every instrumentor, or the session page has no transcript and no tool details. `CrewAIAgentNames` gives each agent role its own lane. `skip_dep_check=True` stops an instrumentor from silently skipping itself when its version check disagrees with your installed packages. +Pass `config` to every instrumentor, or the session has no transcript. Keep `skip_dep_check=True`, or an instrumentor can silently skip itself. -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `CrewAIAgentNames()` and the exporter to that provider and pass it to `instrument()`. +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `CrewAIAgentNames()` and the exporter to it and pass it to `instrument()`. ## Group a conversation into one session @@ -137,15 +131,15 @@ def handle_message(conversation_id: str, text: str, history: str) -> str: return build_crew(text, history).kickoff().raw ``` -Put the user's message first in the task description, because Maple labels each turn with the first line of the prompt. Name the crew, or its root span gets a new `Crew_` name on every request. `crew_id` and `crew_key` don't work as conversation ids. +Put the user's message first in the task description, because Maple labels each turn with its first line. Give the crew a `name=`. `crew_id` and `crew_key` don't work as conversation ids. For conversational flows, wrap `flow.handle_turn(text, session_id=conversation_id)` in the same `using_session` and set `name = "support_flow"` on the flow class. -Use `kickoff()` or `await crew.kickoff_async()`. The instrumentor doesn't patch `akickoff()`, so with it every model and tool call becomes its own trace. +Use `kickoff()` or `await crew.kickoff_async()`, never `akickoff()`, which isn't traced as one run. ## Streaming crews -`Crew(stream=True)` calls `kickoff()` twice, which gives each message an extra empty turn. Wrap the streamed turn in one span of your own: +`Crew(stream=True)` adds an empty turn to every message. Wrap the streamed turn in one span of your own: ```py from opentelemetry import trace @@ -169,13 +163,13 @@ def stream_message(conversation_id: str, text: str, history: str, send) -> None: ## Flush in short-lived processes -The SDK flushes on a normal interpreter exit, which covers servers and `crewai run`. In serverless handlers and notebooks, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. +Servers and `crewai run` need nothing. In serverless handlers and notebooks, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. ## Check that it works Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session labeled **CrewAI**, with one turn per `kickoff()` starting at `support.kickoff`, `ChatCompletion` model calls with tokens, tool calls like `get_weather.run`, and one lane per agent role. -Cost shows as **unpriced** unless your models go through LiteLLM. That's expected. +Cost shows as **unpriced** unless your models go through LiteLLM, which is expected. ## Troubleshooting @@ -188,7 +182,4 @@ Cost shows as **unpriced** unless your models go through LiteLLM. That's expecte ## Related - [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. -- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. -- [CrewAI telemetry](https://docs.crewai.com/en/telemetry): what CrewAI's own analytics collect and how to turn them off. -- [openinference-instrumentation-crewai](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-crewai): the instrumentor's source. - [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/dspy.md b/apps/landing/src/content/docs/agent-tracing/dspy.md index 6258869b74..d027186a55 100644 --- a/apps/landing/src/content/docs/agent-tracing/dspy.md +++ b/apps/landing/src/content/docs/agent-tracing/dspy.md @@ -7,7 +7,7 @@ navLabel: "DSPy" icon: "python" --- -DSPy has no tracing of its own. OpenInference's `openinference-instrumentation-dspy` records a span for every module, predictor, model call and tool call, but no token counts, no tool names and no conversation id. You add the first two with a small DSPy callback, and the conversation id by wrapping each call in `using_session`. +This guide traces DSPy with OpenInference's DSPy instrumentor, adds tokens and tool names with a small DSPy callback, and groups each conversation into one session with `using_session`. Tested with DSPy 3.4 and `openinference-instrumentation-dspy` 0.1.45 on Python 3.10 or later. @@ -65,13 +65,13 @@ DSPyInstrumentor().instrument(tracer_provider=provider, config=TraceConfig(enabl ThreadingInstrumentor().instrument() ``` -`enable_genai_semconv=True` writes the `gen_ai.*` attributes Maple's session page reads. Without it the session has no transcript and no model. `ThreadingInstrumentor` keeps `dspy.Parallel`, `Evaluate` and thread pool workers in the caller's trace and session. +Without `enable_genai_semconv=True` the session page has no transcript and no model. `ThreadingInstrumentor` keeps `dspy.Parallel` and thread pool workers in the caller's session. -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add the OTLP exporter to that one and pass it to `instrument()` instead of creating a second. +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add the OTLP exporter to it and pass it to `instrument()` instead of creating a second one. ## Add the Maple callback -The callback adds tokens and cost to model spans, names and arguments to tool spans, and an agent span for each module class you wrote. Save it as `maple_dspy.py`: +The callback adds tokens, cost, tool names and agent spans. Save it as `maple_dspy.py`: ```py # maple_dspy.py @@ -82,7 +82,6 @@ from dspy.utils.callback import BaseCallback from openinference.instrumentation import TraceConfig from opentelemetry import trace -# Same switches as the instrumentor: OPENINFERENCE_HIDE_INPUTS / OPENINFERENCE_HIDE_OUTPUTS. _config = TraceConfig() @@ -92,9 +91,6 @@ def _message(role, values): class MapleCallback(BaseCallback): - """Adds what Maple reads and the OpenInference DSPy instrumentor leaves out: - an agent span per program, tool names and arguments, tokens and cost.""" - def __init__(self): self._agents = set() self._lms = {} @@ -120,7 +116,7 @@ class MapleCallback(BaseCallback): trace.get_current_span().set_attribute("gen_ai.output.messages", reply) def on_adapter_format_start(self, call_id, instance, inputs): - # The span is "ChatAdapter.__call__"; without an operation, "chat" in the name reads as a model call. + # Keeps "ChatAdapter.__call__" from counting as a model call trace.get_current_span().set_attribute("gen_ai.operation.name", "invoke_workflow") def on_tool_start(self, call_id, instance, inputs): @@ -135,7 +131,6 @@ class MapleCallback(BaseCallback): def on_lm_end(self, call_id, outputs, exception): lm = self._lms.pop(call_id, None) - # The LM's history holds the provider response. Threads share the LM, so match ours by identity. entry = next((e for e in reversed(lm.history[-16:]) if e["outputs"] is outputs), None) if lm else None if entry is None or getattr(entry["response"], "cache_hit", False): return # history is off, or a cache hit that cost nothing @@ -155,7 +150,7 @@ class MapleCallback(BaseCallback): Register it with the rest of your DSPy configuration: ```py -import tracing # first: sets up the provider and patches DSPy +import tracing # must come first import dspy from maple_dspy import MapleCallback @@ -163,11 +158,11 @@ from maple_dspy import MapleCallback dspy.configure(lm=dspy.LM("openai/gpt-4o-mini", temperature=0), callbacks=[MapleCallback()]) ``` -Every `dspy.Module` subclass you write becomes an agent in Maple, named after its class. Built-in modules like `Predict` and `ReAct` stay steps inside it. Give worker modules descriptive class names (`WeatherWorker`) so each gets its own lane. +Every `dspy.Module` subclass you write shows up as an agent named after its class, so give worker modules descriptive names like `WeatherWorker`. ## Group a conversation into one session -Each call to your module starts a new trace. Maple joins them into one session by `session.id`, which the instrumentor sets inside OpenInference's `using_session` block: +Each call to your module is its own trace. Wrap every call in `using_session` with the conversation id to group them into one session: ```py import dspy @@ -198,13 +193,13 @@ def handle_message(conversation_id: str, question: str, turns: list[dict]) -> st Use the id your app stores the chat under. A new UUID per request gives one session per message, and a constant puts every user in one session. -If you stream with `dspy.streamify`, call it after `dspy.configure(callbacks=[MapleCallback()])`, since it copies the callback list when called. Put `using_session` around the loop that reads the stream, because the program only starts on the first chunk. +If you stream with `dspy.streamify`, call it after `dspy.configure(callbacks=[MapleCallback()])`, and put `using_session` around the loop that reads the stream. To keep prompts and outputs out of your traces, set `OPENINFERENCE_HIDE_INPUTS=true` and `OPENINFERENCE_HIDE_OUTPUTS=true` before `tracing.py` runs. The callback follows both. ## Flush before short-lived processes exit -`BatchSpanProcessor` exports every 5 seconds and flushes on a normal exit. A killed process, a serverless handler or a notebook needs an explicit flush: +Serverless handlers, notebooks and processes that get killed need an explicit flush: ```py from tracing import provider @@ -217,9 +212,7 @@ finally: ## Check that it works -Run a conversation of two or three messages through `handle_message` with one conversation id, including one tool call, with `cache=False` on the LM so every call reaches the provider. In **Agent Sessions** you should see one session with framework **DSPy** and one turn per call, each starting at `ChatAssistant.forward`. - -Model calls are `LM.__call__` spans with a model and tokens. The transcript uses DSPy's `[[ ## field ## ]]` prompt format. Tool calls include `finish.__call__`, which is `dspy.ReAct`'s built-in end-of-loop tool. Cost is DSPy's own estimate, and models DSPy has no price for show as **unpriced**. +Set `cache=False` on the LM, then run two or three messages through `handle_message` with one conversation id, including one tool call. In **Agent Sessions** you should see one session with framework **DSPy**, one turn per call, and `LM.__call__` model calls with tokens. `finish.__call__` is `dspy.ReAct`'s built-in end-of-loop tool. ## Troubleshooting @@ -232,9 +225,5 @@ Model calls are `LM.__call__` spans with a model and tokens. The transcript uses ## Related -- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. -- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. -- [DSPy observability](https://dspy.ai/tutorials/observability/): DSPy's own tracing page, built on MLflow. -- [openinference-instrumentation-dspy](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-dspy): the instrumentor's source. -- [DSPy's `BaseCallback`](https://github.com/stanfordnlp/dspy/blob/main/dspy/utils/callback.py): the hook API `MapleCallback` builds on. +- [Agent Sessions overview](/docs/agent-sessions/overview) - [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/google-adk.md b/apps/landing/src/content/docs/agent-tracing/google-adk.md index 7ae397a500..394fe7e220 100644 --- a/apps/landing/src/content/docs/agent-tracing/google-adk.md +++ b/apps/landing/src/content/docs/agent-tracing/google-adk.md @@ -7,11 +7,9 @@ navLabel: "Google ADK" icon: "googleadk" --- -Google's Agent Development Kit (ADK) emits OpenTelemetry spans for every run, agent, model call and tool call, and stamps the ADK session id on them, so a multi-turn chat groups into one Maple session. You need no instrumentation package. +Google's Agent Development Kit (ADK) emits OpenTelemetry spans on its own, so you need no instrumentation package. You set two environment variables so the transcript is in the format Maple reads, and register a tracer provider when you run a plain `Runner`. -Two things need setup. By default ADK writes prompts and replies into its own attributes, which Maple doesn't read, so you switch it to the GenAI format with two environment variables. And with a plain `Runner`, nothing exports until you register a tracer provider. - -Tested with ADK for Python 2.10. ADK for Go and Kotlin emit the same span names, but their setup isn't covered here. +This guide covers ADK for Python. ## Quick setup with a coding agent @@ -33,7 +31,7 @@ Your ingest key is under **Settings → Ingestion**. EU organizations should say pip install "google-adk>=2.10" litellm opentelemetry-exporter-otlp-proto-http ``` -`litellm` is only needed for non-Gemini models through ADK's `LiteLlm` wrapper. Don't pin a newer OpenTelemetry version: ADK 2.10 caps `opentelemetry-sdk` at 1.42.1. +`litellm` is only needed for non-Gemini models. Don't pin a newer OpenTelemetry version: ADK 2.10 caps `opentelemetry-sdk` at 1.42.1. ## Configure the export and the transcript format @@ -48,13 +46,13 @@ OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=SPAN_ONLY ADK_CAPTURE_MESSAGE_CONTENT_IN_SPANS=false ``` -The `%20` is an encoded space; the Python SDK decodes it. EU organizations use `https://ingest.eu.maple.dev`. Use `SPAN_ONLY` exactly: `true` sends content to log records, which Maple doesn't read for the transcript. +EU organizations use `https://ingest.eu.maple.dev`. The `%20` is an encoded space and must stay. Use `SPAN_ONLY` exactly, because `true` leaves the transcript empty. These settings store every prompt and tool result in Maple. To keep structure and tokens without content, leave `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` unset and delete the two `gen_ai.tool.call.*` lines from the plugin below. ## Register a tracer provider -`adk web` and `adk api_server` build a tracer provider from the environment. A `Runner` in your own app, worker or script doesn't, and its spans go nowhere without a warning. Add a `telemetry.py`: +A `Runner` in your own app, worker or script exports nothing until you register a tracer provider. Add a `telemetry.py`: ```py # telemetry.py @@ -101,8 +99,6 @@ provider.add_span_processor(SkipDuplicateToolSpans(OTLPSpanExporter())) trace.set_tracer_provider(provider) ``` -ADK doesn't record tool arguments or results, so the `ToolCallAttributes` plugin adds them. `SkipDuplicateToolSpans` drops two tool spans that would otherwise count one call twice. - Import `telemetry` as the first line of your entry point and register the plugin on the runner: ```py @@ -137,13 +133,13 @@ runner = Runner( ) ``` -If your app already sets up OpenTelemetry (Logfire, Sentry, a platform agent), add the `SkipDuplicateToolSpans(OTLPSpanExporter())` processor to that provider instead of creating a second one. +If your app already sets up OpenTelemetry (Logfire, Sentry, a platform agent), add the `SkipDuplicateToolSpans(OTLPSpanExporter())` processor to that provider instead. -Under `adk web` or `adk api_server`, skip the provider and register the plugin on your `App`: `App(name="support", root_agent=agent, plugins=[ToolCallAttributes()])`. +`adk web` and `adk api_server` create the provider themselves. There, skip it and register the plugin on your `App`: `App(name="support", root_agent=agent, plugins=[ToolCallAttributes()])`. ## Use one ADK session per conversation -Each `runner.run_async()` call is one turn. Pass your own conversation id as `session_id` on every turn, and the whole conversation is one Maple session: +Each `runner.run_async()` call is one turn. Pass your conversation id as `session_id` on every turn: ```py async def chat(conversation_id: str, user_id: str, text: str) -> str: @@ -158,13 +154,13 @@ async def chat(conversation_id: str, user_id: str, text: str) -> str: return reply ``` -With `auto_create_session=True`, the runner creates the session under that id on the first turn. Don't call `create_session()` without a `session_id` on each request; ADK then mints a new id per turn and every message becomes its own session. +With `auto_create_session=True`, the runner creates the session on the first turn. Don't call `create_session()` without a `session_id`, or every message becomes its own session. -To call an agent like a tool, add it to `sub_agents` with `mode="single_turn"`. `AgentTool` runs the sub-agent under a second session id, which can move the turn into a separate session. +To call an agent like a tool, add it to `sub_agents` with `mode="single_turn"` instead of using `AgentTool`, which can move the turn into a separate session. ## Flush before a short-lived process exits -`BatchSpanProcessor` sends spans every 5 seconds, so a script, notebook cell or job that exits sooner loses the last turn. Flush when the work ends: +A script, notebook cell or job that exits within 5 seconds of its last turn loses that turn. Flush when the work ends: ```py # script.py @@ -190,9 +186,7 @@ In a server, call `provider.shutdown()` from your shutdown hook. On Cloud Run or ## Check that it works -Run a conversation of two or three turns, one of them calling a tool, then open **Agent Sessions**. You should see one session with the framework **Google ADK**, one turn per `run_async()`, a transcript with the tool calls, their arguments and results, and tokens on each model call. - -Cost shows as unpriced, because ADK doesn't record it. +Run a conversation of two or three turns, one calling a tool, then open **Agent Sessions**. You should see one session with the framework **Google ADK**, one turn per `run_async()`, tool calls with their arguments and results, and tokens on each model call. Cost shows as unpriced. ## Troubleshooting @@ -205,7 +199,4 @@ Cost shows as unpriced, because ADK doesn't record it. ## Related - [Agent Sessions overview](/docs/agent-sessions/overview): how Maple builds sessions, turns and checks. -- [Agent tracing guides](/docs/agent-tracing): every framework. - [ADK agent activity traces](https://google.github.io/adk-docs/observability/traces/): ADK's span reference and export setup. -- [LiteLLM](/docs/agent-tracing/litellm): tracing LiteLLM on its own, outside ADK. -- [OpenTelemetry for any agent](/docs/agent-tracing/opentelemetry): the GenAI attributes Maple reads. diff --git a/apps/landing/src/content/docs/agent-tracing/haystack.md b/apps/landing/src/content/docs/agent-tracing/haystack.md index 17ec752d1a..cefe1d9ab4 100644 --- a/apps/landing/src/content/docs/agent-tracing/haystack.md +++ b/apps/landing/src/content/docs/agent-tracing/haystack.md @@ -7,7 +7,7 @@ navLabel: "Haystack" icon: "haystack" --- -Haystack 3 traces its pipelines, components, Agent steps and tool calls through `opentelemetry-haystack`, but it keeps the model, tokens and messages inside `haystack.*` JSON blobs Maple doesn't read, leaves failed tool calls unmarked and has no conversation id. This guide adds `maple_haystack.py`, a subclass of Haystack's tracer that writes the GenAI attributes Maple reads, and a `conversation()` block that groups runs into one session. +This guide adds `maple_haystack.py`, a Haystack tracer that records model, tokens, messages and failed tool calls for Maple, and a `conversation()` block that groups runs into one session. Tested with `haystack-ai` 3.2 and `opentelemetry-haystack` 1.0 on Python 3.10 or later. @@ -32,11 +32,9 @@ pip install "haystack-ai>=3.2" "opentelemetry-haystack>=1.0" \ "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Save this file next to your app as `maple_haystack.py`. It keeps every span and tag Haystack emits and adds `gen_ai.*` attributes to the Agent, model and tool spans: +Save this file next to your app as `maple_haystack.py`: ```py -"""Haystack tracer that adds the OpenTelemetry GenAI attributes Maple reads.""" - import json import logging from collections.abc import Iterator @@ -55,7 +53,6 @@ _conversation_id: ContextVar[str | None] = ContextVar("maple_conversation_id", d @contextmanager def conversation(conversation_id: str) -> Iterator[None]: - """Every Haystack run inside this block joins the same Maple session.""" token = _conversation_id.set(conversation_id) try: yield @@ -110,7 +107,7 @@ class MapleSpan(OpenTelemetrySpan): "gen_ai.usage.cache_read.input_tokens": prompt_details.get("cached_tokens"), "gen_ai.usage.cache_write.input_tokens": prompt_details.get("cache_write_tokens"), "gen_ai.usage.reasoning.output_tokens": (usage.get("completion_tokens_details") or {}).get("reasoning_tokens"), - "gen_ai.usage.cost": usage.get("cost"), # OpenRouter prices every call; other providers leave this out + "gen_ai.usage.cost": usage.get("cost"), } if self._content: attributes["gen_ai.output.messages"] = _messages(replies) @@ -118,12 +115,10 @@ class MapleSpan(OpenTelemetrySpan): def _tool(self, key: str, value: Any) -> None: if key.endswith(".output") and isinstance(value, dict) and "error" in value: - # Haystack records a failed tool as {"error": ...} and leaves the span status unset self._span.set_status(StatusCode.ERROR, str(value["error"])) self._span.set_attribute("error.type", "ToolInvocationError") if self._content: attribute = "gen_ai.tool.call.arguments" if key.endswith(".input") else "gen_ai.tool.call.result" - # Tool results arrive as strings: write them as-is, so a JSON result stays a JSON object self._span.set_attribute(attribute, value if isinstance(value, str) else json.dumps(value, default=str)) @@ -138,7 +133,6 @@ class MapleHaystackTracer(OpenTelemetryTracer): attributes: dict[str, str] = {} operation = None if operation_name == "haystack.agent.run": - # The Agent span has no name: use the pipeline component or the AgentTool that runs it parent = getattr(trace.get_current_span(), "attributes", None) or {} attributes["gen_ai.operation.name"] = "invoke_agent" attributes["gen_ai.agent.name"] = parent.get("haystack.component.name") or parent.get("gen_ai.tool.name") or "agent" @@ -150,7 +144,7 @@ class MapleHaystackTracer(OpenTelemetryTracer): if conversation_id := _conversation_id.get(): attributes["gen_ai.conversation.id"] = conversation_id if not self._content: - tags.pop("haystack.pipeline.input_data", None) # a plain tag, not gated by Haystack's content switch + tags.pop("haystack.pipeline.input_data", None) with self._tracer.start_as_current_span(operation_name, attributes=attributes) as raw_span: span = MapleSpan(raw_span, operation, self._content) @@ -188,13 +182,13 @@ export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" ``` -EU organizations use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. +EU organizations use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. -Import `telemetry` before the first `pipeline.run()` or `agent.run()`. Keep the tracer name `"haystack"`, because Maple uses it to label the sessions as Haystack. If the app already has a `TracerProvider`, skip the provider lines and pass `trace.get_tracer("haystack")` from the existing one. +Import `telemetry` before the first `pipeline.run()` or `agent.run()`. Keep the tracer name `"haystack"`, since Maple labels the sessions by it. If the app already has a `TracerProvider`, skip the provider lines and pass `trace.get_tracer("haystack")` from the existing one. ## Group turns into one session -Haystack's `Agent` is stateless: your app passes the message history into every run, and each run is its own trace. Wrap every run in `conversation()` with your chat's id: +Each run is its own trace. Wrap every run in `conversation()` with your chat's id: ```py from haystack import Pipeline @@ -213,15 +207,15 @@ pipeline.add_component("assistant", agent) def handle_message(chat_id: str, history: list[ChatMessage], text: str) -> list[ChatMessage]: with conversation(chat_id): result = pipeline.run({"assistant": {"messages": [*history, ChatMessage.from_user(text)]}}) - # The Agent adds its system prompt on every run, so keep it out of the stored history + # The Agent re-adds its system prompt on every run, so don't store it return [m for m in result["assistant"]["messages"] if not m.is_from("system")] ``` -Use the id your app already stores for the chat, such as a database row id or a frontend thread id. A new UUID per request gives one session per message, and one id for the whole process merges every user into one session. `conversation()` uses a `ContextVar`, so it stays scoped to the request in async and threaded servers and follows the Agent into its tool threads. +Use the id your app already stores for the chat. A new UUID per request gives one session per message, and one id for the whole process merges every user into one session. ## Agent names, content and tokens -Maple names each agent, and gives it a lane, from the pipeline component that runs it (`assistant` above) or from the `name=` of the `AgentTool` that wraps it. An Agent run directly with `agent.run()` is called `agent`. +Each agent is named after its pipeline component (`assistant` above) or the `name=` of the `AgentTool` that wraps it. An Agent run directly with `agent.run()` is called `agent`. To keep prompts and responses out of Maple, pass `content=False`: @@ -229,7 +223,7 @@ To keep prompts and responses out of Maple, pass `content=False`: tracing.enable_tracing(MapleHaystackTracer(trace.get_tracer("haystack"), content=False)) ``` -Model, tokens, cost, tool names and failures are still recorded. `HAYSTACK_CONTENT_TRACING_ENABLED` has no effect with this tracer. +`HAYSTACK_CONTENT_TRACING_ENABLED` has no effect with this tracer. A streamed `OpenAIChatGenerator` reply carries no token counts unless you ask for them (OpenRouter always sends usage): @@ -237,11 +231,11 @@ A streamed `OpenAIChatGenerator` reply carries no token counts unless you ask fo OpenAIChatGenerator(model="gpt-4o-mini", generation_kwargs={"stream_options": {"include_usage": True}}) ``` -Cost only appears with `OpenRouterChatGenerator`, the one Haystack generator that reports a price. Other providers show tokens and read as **unpriced**. +Cost only appears with `OpenRouterChatGenerator`. Other providers show tokens and read as **unpriced**. ## Flush before short-lived processes exit -`BatchSpanProcessor` exports every 5 seconds. A script, CLI, notebook, cron job or serverless handler that exits sooner loses the last batch: +Scripts, notebooks, cron jobs and serverless handlers need an explicit flush before they exit: ```py try: @@ -255,9 +249,7 @@ Long-running servers only need `provider.shutdown()` in their shutdown hook. ## Check that it works -Run a conversation of at least two turns, one with a tool call, and open **Agent Sessions** filtered by your service name. You should see one session per conversation id, labelled Haystack, with one turn per `pipeline.run()`. Each `haystack.agent.step.llm` span shows up as a model call with model and tokens, and each `haystack.agent.step.tool` span as a tool call, with failed tools marked. - -A tool that returns plain text shows its result in the transcript but not on its tool call row. Return a dict to see the result there too. +Run a conversation of at least two turns, one with a tool call, and open **Agent Sessions** filtered by your service name. You should see one session per conversation id, labelled Haystack, with one turn per `pipeline.run()`, model calls with tokens, and tool calls with failed ones marked. A tool's result shows on its tool call row only if the tool returns a dict. ## Troubleshooting @@ -271,7 +263,3 @@ A tool that returns plain text shows its result in the transcript but not on its ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [Agent tracing guides](/docs/agent-tracing) -- [Trace agents with plain OpenTelemetry](/docs/agent-tracing/opentelemetry) -- [Haystack tracing docs](https://docs.haystack.deepset.ai/docs/tracing) -- [`opentelemetry-haystack` on PyPI](https://pypi.org/project/opentelemetry-haystack/) diff --git a/apps/landing/src/content/docs/agent-tracing/langchain.md b/apps/landing/src/content/docs/agent-tracing/langchain.md index 539d3069cc..18a0bc7e5f 100644 --- a/apps/landing/src/content/docs/agent-tracing/langchain.md +++ b/apps/landing/src/content/docs/agent-tracing/langchain.md @@ -7,13 +7,13 @@ navLabel: "LangChain & LangGraph" icon: "langchain" --- -LangChain and LangGraph report every run through callbacks, and OpenInference's `openinference-instrumentation-langchain` turns those runs into OpenTelemetry spans. Every `invoke()` starts a new trace, so you have to pass the conversation's `thread_id` for Maple to group a chat into one session. +OpenInference's `openinference-instrumentation-langchain` sends LangChain and LangGraph runs to Maple. Pass the conversation's `thread_id` on every call so a chat becomes one session. -Tested with LangChain 1.4 (`create_agent`), LangGraph 1.2 and `openinference-instrumentation-langchain` 0.1.76 on Python 3.10 or later. LangChain.js isn't covered yet. +This guide covers Python 3.10 or later. LangChain.js isn't covered yet. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-langchain](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-langchain) skill, which contains every step of this guide. +Paste this prompt into Claude Code, Codex, Cursor or another coding agent. It installs the [maple-agent-tracing-langchain](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-langchain) skill and follows it. ```text Set up Maple agent tracing for LangChain & LangGraph in this project. @@ -33,7 +33,7 @@ pip install "langchain>=1.4" "langgraph>=1.2" "langchain-openai>=1.6" \ "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Pin `openinference-instrumentation` explicitly. Older versions lack the GenAI output this setup depends on. +Keep the explicit `openinference-instrumentation` pin. Older versions don't work with this setup. ## Point the exporter at Maple @@ -61,13 +61,11 @@ from opentelemetry.sdk.trace.export import BatchSpanProcessor # The name= you gave create_agent(), and graph nodes that act as agents AGENT_NAMES = {"assistant"} -# LangGraph's tool node and prompt templates: steps, not tool or model calls +# Your tool node and prompt templates STEP_NAMES = {"tools", "ChatPromptTemplate"} class AgentSpans(SpanProcessor): - """Names your agents' spans for Maple (one lane per agent) and keeps graph steps out of the tool and model counts.""" - def on_start(self, span, parent_context=None): if span.instrumentation_scope.name != "openinference.instrumentation.langchain": return @@ -78,7 +76,7 @@ class AgentSpans(SpanProcessor): span.set_attribute("gen_ai.operation.name", "invoke_workflow") -provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider = TracerProvider() provider.add_span_processor(AgentSpans()) provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) @@ -89,15 +87,15 @@ LangChainInstrumentor().instrument( ) ``` -`enable_genai_semconv=True` is required. Without it, Maple ignores the `thread_id` and shows the transcript as raw JSON. +Keep `enable_genai_semconv=True`. Without it, sessions don't group and the transcript shows as raw JSON. -`AgentSpans` names your agents so Maple gives each one its own lane. Put every agent's `name=` in `AGENT_NAMES`, sub-agents included. `STEP_NAMES` keeps graph steps out of the tool and model counts; if your tool node has another name containing "tool", like `run_tools`, add it there. +Put every agent's `name=` in `AGENT_NAMES`, sub-agents included, so each gets its own lane. If your tool node isn't called `tools`, add its name to `STEP_NAMES`. -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `AgentSpans()` and the exporter to that provider and pass it to `instrument()`. +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `AgentSpans()` and the exporter to it and pass it to `instrument()`. ## Group a conversation with thread_id -Maple groups turns into a session by `gen_ai.conversation.id`, which the instrumentor fills from the run's `thread_id`. Pass your app's conversation id on every `invoke()`, `stream()` and `Command(resume=...)`: +Pass your app's conversation id as `thread_id` on every `invoke()`, `stream()` and `Command(resume=...)`: ```py import tracing # first, before the first invoke() @@ -122,15 +120,13 @@ def handle_message(conversation_id: str, text: str) -> str: return result["messages"][-1].content ``` -This works with or without a checkpointer. A plain chain (`prompt | model`, no graph) doesn't read `configurable`, so pass `{"metadata": {"thread_id": conversation_id}}` instead. Agents called from inside a tool inherit the caller's id. - -Use the id your app stores the chat under. A new UUID per request gives you one session per message. +Use the id your app stores the chat under. A new UUID per request gives you one session per message. A plain chain (`prompt | model`, no graph) ignores `configurable`, so pass `{"metadata": {"thread_id": conversation_id}}` instead. -`stream_usage=True` makes `ChatOpenAI` report tokens on streamed replies when it talks to a server other than api.openai.com, such as a custom `base_url`, vLLM or a gateway. +Keep `stream_usage=True` if `ChatOpenAI` uses a custom `base_url`, vLLM or a gateway. Without it, streamed replies have no tokens. ## Flush in short-lived processes -The SDK flushes on a normal interpreter exit, and long-running servers need nothing. In serverless handlers, notebooks and task workers, flush after each run: +Long-running servers need nothing. In serverless handlers, notebooks and task workers, flush after each run: ```py from tracing import provider @@ -138,16 +134,16 @@ from tracing import provider try: handle_message("conv-42", "What's the weather in Berlin?") finally: - provider.force_flush() # serverless: before returning; notebooks: after each run + provider.force_flush() ``` -On LangGraph Server (`langgraph dev` or self-hosted), import `tracing` at the top of the graph module that `langgraph.json` points to, and set the `OTEL_*` variables in the server's environment. Each server thread already carries its `thread_id`. +On LangGraph Server (`langgraph dev` or self-hosted), import `tracing` at the top of the graph module that `langgraph.json` points to, and set the `OTEL_*` variables in the server's environment. Server threads group into sessions without extra code. ## Check that it works Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session with one turn per `invoke()`, a readable transcript, `ChatOpenAI` model calls with tokens, and tool calls named after your tools. -The framework shows as **Unidentified** and cost as **unpriced**. Both are expected. With a checkpointer, every turn's label repeats the conversation's first message, but the transcript inside each turn is correct. +The framework shows as **Unidentified** and cost as **unpriced**, which is expected. With a checkpointer, every turn is labeled with the conversation's first message. The transcript inside each turn is still correct. ## Troubleshooting @@ -160,7 +156,4 @@ The framework shows as **Unidentified** and cost as **unpriced**. Both are expec ## Related - [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. -- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. -- [openinference-instrumentation-langchain](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-langchain): the instrumentor's source. -- [Trace with OpenTelemetry](https://docs.langchain.com/langsmith/trace-with-opentelemetry): LangSmith's OpenTelemetry export. - [OpenRouter](/docs/agent-tracing/openrouter): if your models go through OpenRouter. diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md index 3a13a49baa..716a61f814 100644 --- a/apps/landing/src/content/docs/agent-tracing/litellm.md +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -7,9 +7,7 @@ navLabel: "LiteLLM" icon: "litellm" --- -LiteLLM's OpenTelemetry logger writes one `chat ` span per model call, with the model, tokens, prompt and reply. It has no agent loop or notion of a conversation, so your code adds the agent and tool spans, and you pass a session id on every call so Maple can group the turns. - -Tested with `litellm` 1.103.0 and the OpenTelemetry Python SDK 1.43.0 on Python 3.12. +LiteLLM traces each model call. Your code adds the agent and tool spans and passes a session id on every call so Maple groups the turns into one session. ## Quick setup with a coding agent @@ -27,7 +25,7 @@ Your ingest key is in **Settings → Ingestion**. EU organizations should say EU ## Trace in your app or at the proxy -If your code calls `litellm.acompletion()`, trace in your app with the next sections. If your app sends OpenAI-compatible requests to a LiteLLM Proxy you run, see [Trace at the LiteLLM Proxy](#trace-at-the-litellm-proxy). Either way your app emits the agent and tool spans. Trace model calls in one place only, or every call shows up twice. +If your code calls `litellm.acompletion()`, follow the next sections. If your app calls a LiteLLM Proxy you run, see [Trace at the LiteLLM Proxy](#trace-at-the-litellm-proxy). Trace model calls in one place only, or every call shows up twice. ## Install LiteLLM and the exporter @@ -35,7 +33,7 @@ If your code calls `litellm.acompletion()`, trace in your app with the next sect pip install "litellm==1.103.0" "opentelemetry-sdk==1.43.0" "opentelemetry-exporter-otlp-proto-http==1.43.0" ``` -Keep OpenTelemetry below 1.44 on LiteLLM 1.103. Newer versions break LiteLLM's v2 logger ([BerriAI/litellm#41990](https://github.com/BerriAI/litellm/issues/41990)). The fix ships in LiteLLM 1.104, after which you can drop the pin. +Keep OpenTelemetry at 1.43. Version 1.44 and later break LiteLLM 1.103's logger. Point the exporter at Maple: @@ -49,7 +47,7 @@ For an EU organization, use `https://ingest.eu.maple.dev`. ## Register LiteLLM's v2 logger -Use the v2 logger (`OpenTelemetryV2`). The default v1 logger never records a conversation id, so every call becomes its own session. Hand v2 your own `TracerProvider` so LiteLLM's spans and yours share one exporter: +Use the v2 logger (`OpenTelemetryV2`). The default v1 logger makes every call its own session. Pass it your `TracerProvider`: ```py # tracing.py @@ -68,7 +66,6 @@ provider = TracerProvider( {"service.name": "support-agent", "deployment.environment.name": "production"} ) ) -# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) @@ -82,13 +79,13 @@ litellm.callbacks = [ tracer = trace.get_tracer("support-agent") ``` -Import `tracing` at the top of your entry point, before the first model call. If your app already has a `TracerProvider`, pass that one as `tracer_provider=`. Don't also add `"otel"` to `litellm.callbacks`, which registers a second logger. +Import `tracing` at the top of your entry point. If your app already has a `TracerProvider`, pass that one. Don't also add `"otel"` to `litellm.callbacks`, which registers a second logger. -`capture_message_content="span_only"` records prompts and replies on the span, which Maple needs for the transcript. Use `"no_content"` to keep them out of Maple. +`"span_only"` records prompts and replies for the transcript. Use `"no_content"` to keep them out of Maple. ## Wrap the agent loop in agent and tool spans -Wrap each agent run in an `invoke_agent` span and each tool call in an `execute_tool` span. LiteLLM's `chat` spans nest under the current span, so a run is one trace. The v2 logger only traces the async API, so call `acompletion()`, not `completion()`: +Wrap each agent run in an `invoke_agent` span and each tool call in an `execute_tool` span. The v2 logger only traces `acompletion()`, not `completion()`: ```py # agent.py @@ -133,7 +130,6 @@ async def run_tool(agent: Agent, call) -> str: if inspect.isawaitable(result): result = await result except Exception as exc: - # The model gets the error as the tool result; the span is marked failed. span.set_status(StatusCode.ERROR, str(exc)) span.set_attribute("error.type", type(exc).__name__) result = {"error": str(exc)} @@ -160,11 +156,11 @@ async def run_agent(agent: Agent, conversation_id: str, messages: list) -> str: messages.append({"role": "tool", "tool_call_id": call.id, "content": output}) ``` -For sub-agents, give each agent its own `name` and call `run_agent` for the worker from inside a tool function, with the same conversation id. Maple draws one lane per agent name. +For sub-agents, call `run_agent` for the worker from inside a tool function, with its own `name` and the same conversation id. ## Group every turn into one session -The v2 logger turns `litellm_session_id=` into `gen_ai.conversation.id`, which Maple uses as the session key. Pass the chat or thread id your app already has, stable for the whole conversation: +`litellm_session_id=` sets the session. Pass the chat or thread id your app already has, stable for the whole conversation: ```py from agent import Agent, run_agent @@ -179,11 +175,11 @@ async def handle_message(chat_id: str, text: str) -> str: return await run_agent(assistant, chat_id, messages) ``` -If you already pass `metadata`, `metadata={"session_id": ...}` works too. Don't set `gen_ai.conversation.id` on your own `invoke_agent` span, or the session is labeled **Unidentified** instead of **LiteLLM**. +Don't set `gen_ai.conversation.id` on your own `invoke_agent` span, or the session is labeled **Unidentified** instead of **LiteLLM**. For streaming, pass `stream_options={"include_usage": True}` and consume the stream inside the agent span, or the streamed call has no token counts. -Cost shows as unpriced. LiteLLM writes its price to an attribute Maple doesn't read. The [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-litellm) has a recipe that reports each turn's total. +Cost shows as unpriced. The [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-litellm) has a recipe that reports each turn's total. ## Trace at the LiteLLM Proxy @@ -212,7 +208,7 @@ OTEL_ENVIRONMENT_NAME=production OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT=span_only ``` -The `ghcr.io/berriai/litellm` image works as is. A pip-installed proxy needs the OpenTelemetry pin plus the FastAPI instrumentation, which continues your app's trace: +The `ghcr.io/berriai/litellm` image works as is. A pip-installed proxy needs these packages: ```bash pip install "litellm[proxy]==1.103.0" "opentelemetry-sdk==1.43.0" \ @@ -233,17 +229,17 @@ client = AsyncOpenAI(base_url="http://localhost:4000", api_key=os.environ["LITEL async def call_model(conversation_id: str, messages: list, tools: list | None): headers = {"x-litellm-session-id": conversation_id} - propagate.inject(headers) # adds traceparent for the current span + propagate.inject(headers) return await client.chat.completions.create( model="gpt-4o-mini", messages=messages, tools=tools, extra_headers=headers ) ``` -Don't also instrument the OpenAI client in the app, or calls and tokens double. On this path Maple currently shows the LLM call count at 2x the real number in the sessions list and 3x on the session page, because it counts the proxy's auth and server spans. Tokens, cost and the transcript are correct. +Don't also instrument the OpenAI client in the app, or calls and tokens double. On this path the LLM call count shows 2x (sessions list) or 3x (session page) the real number. Tokens, cost and the transcript are correct. ## Flush before a short-lived process exits -LiteLLM creates its span after the call returns, from a background queue. A script, Lambda or notebook cell that ends right after its last call loses that span unless you drain the queue and flush: +A script, Lambda or notebook cell that ends right after its last call loses that call's span. Drain LiteLLM's queue and flush: ```py import asyncio @@ -274,7 +270,7 @@ A long-running server needs this only in its shutdown hook. ## Check that it works -Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session with the id you passed and framework **LiteLLM**, one turn per message, and a transcript with the prompts, replies and tool calls. Each turn holds your `invoke_agent` span with `chat ` and `execute_tool ` spans inside it. +Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session with the id you passed and framework **LiteLLM**, one turn per message, and a transcript with the prompts, replies and tool calls. ## Troubleshooting @@ -287,8 +283,4 @@ Run a conversation with two messages and a tool call, then open **Agent Sessions ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [All agent tracing guides](/docs/agent-tracing) -- [Trace OpenRouter calls with Broadcast](/docs/agent-tracing/openrouter) - [LiteLLM: OpenTelemetry v2](https://docs.litellm.ai/docs/observability/opentelemetry_v2) -- [LiteLLM: OpenTelemetry (v1)](https://docs.litellm.ai/docs/observability/opentelemetry_integration) -- [Instrument a Python application](/docs/guides/instrumentation-python) diff --git a/apps/landing/src/content/docs/agent-tracing/llamaindex.md b/apps/landing/src/content/docs/agent-tracing/llamaindex.md index 91f341ed6f..ba9cd6c95b 100644 --- a/apps/landing/src/content/docs/agent-tracing/llamaindex.md +++ b/apps/landing/src/content/docs/agent-tracing/llamaindex.md @@ -7,13 +7,11 @@ navLabel: "LlamaIndex" icon: "llamaindex" --- -OpenInference's `openinference-instrumentation-llama-index` turns LlamaIndex agent runs, workflow steps, model calls and tool calls into OpenTelemetry spans. It needs a small span processor to avoid counting each model call two or three times, and you have to wrap every `agent.run()` in a conversation id, or each message becomes its own session. - -Tested with llama-index-core 0.14.25 and `openinference-instrumentation-llama-index` 4.5.2 on Python 3.10 or later, using `FunctionAgent`, `AgentWorkflow` and custom `Workflow` classes. +OpenInference's `openinference-instrumentation-llama-index` sends LlamaIndex agents and workflows to Maple. You add a small span processor and wrap every `agent.run()` in a conversation id so a chat becomes one session. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-llamaindex](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-llamaindex) skill, which contains every step of this guide. +Paste this prompt into Claude Code, Codex, Cursor or another coding agent. It installs the [maple-agent-tracing-llamaindex](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-llamaindex) skill and follows it. ```text Set up Maple agent tracing for LlamaIndex in this project. @@ -32,9 +30,9 @@ pip install "llama-index-core>=0.14.25" "openinference-instrumentation-llama-ind "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Add your model package (`llama-index-llms-openai`, `llama-index-llms-openrouter`, ...) as usual. On llama-index-core older than 0.14.19 the instrumentor logs a dependency conflict and instruments nothing. +Add your model package (`llama-index-llms-openai`, `llama-index-llms-openrouter`, ...) as usual. -Use this instead of LlamaIndex's own `llama-index-observability-otel`, which puts the prompt and reply in span events that Maple doesn't read. Don't run both. +If the app uses LlamaIndex's own `llama-index-observability-otel`, remove it. Maple can't read its transcripts, and running both doubles every span. ## Point the exporter at Maple @@ -66,21 +64,16 @@ LLM_METHODS = (".chat", ".achat", ".stream_chat", ".astream_chat", class LlamaIndexForMaple(SpanProcessor): - """Sits in front of the exporter: one span per model and tool call, agent names, no false HITL failures.""" - def __init__(self, exporter_processor: SpanProcessor): self._next = exporter_processor self._open_llm_spans = {} def on_start(self, span, parent_context=None): - # Maple labels a span "LlamaIndex" by a llamaindex.* key; OpenInference writes none span.set_attribute("llamaindex.instrumentor", "openinference") - # instrument_tags({"gen_ai.agent.name": ...}) becomes an attribute, so sub-agents get lanes agent_name = active_instrument_tags.get().get("gen_ai.agent.name") if agent_name: span.set_attribute("gen_ai.agent.name", agent_name) if span.name.endswith((".call_tool", ".aggregate_tool_results")): - # agent workflow steps around the tool span; by name alone Maple would count them as tool calls span.set_attribute("gen_ai.operation.name", "invoke_workflow") if span.name.endswith(LLM_METHODS): self._open_llm_spans[span.context.span_id] = span @@ -89,12 +82,12 @@ class LlamaIndexForMaple(SpanProcessor): def on_end(self, span): self._open_llm_spans.pop(span.context.span_id, None) if span.name.endswith("._prepare_chat_with_tools"): - return # builds the request; never calls the model + return if (span.status.description or "").startswith("WaitingForEvent"): - return # ctx.wait_for_event() suspends the tool and replays it later; not a failure + return outer = self._open_llm_spans.get(span.parent.span_id) if span.parent else None if outer is not None and outer.name == span.name: - outer.set_attributes(span.attributes) # the inner twin holds the messages and usage + outer.set_attributes(span.attributes) return self._next.on_end(span) @@ -105,7 +98,7 @@ class LlamaIndexForMaple(SpanProcessor): return self._next.force_flush(timeout_millis) -provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider = TracerProvider() provider.add_span_processor(LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))) trace.set_tracer_provider(provider) @@ -115,11 +108,9 @@ LlamaIndexInstrumentor().instrument( ) ``` -`enable_genai_semconv=True` is required. Without it the session page has no transcript. - -`LlamaIndexForMaple` wraps the exporter, so always add the exporter through it. It merges the nested spans LlamaIndex opens around each model call into one, keeps workflow steps out of the tool count, copies agent names from `instrument_tags`, and labels the framework as LlamaIndex. +Keep `enable_genai_semconv=True`, or the session has no transcript. Always add the exporter through `LlamaIndexForMaple`, never directly. -If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))` to that provider and pass it to `instrument()`. +If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or Sentry), add `LlamaIndexForMaple(BatchSpanProcessor(OTLPSpanExporter()))` to it and pass it to `instrument()`. ## Group a conversation into one session @@ -148,15 +139,15 @@ async def handle_message(conversation_id: str, text: str): await handler ``` -Only the `agent.run()` call has to be inside the `with`. Every step, tool and model call of the run inherits the id, so you can consume the stream outside it. +Only the `agent.run()` call needs to be inside the `with`. Consume the stream outside it. -Keep one `Context` per conversation. The `Context` object itself isn't a session id, and a new UUID per request gives you one session per message. +Keep one `Context` per conversation. A new UUID per request gives you one session per message. -For multi-agent workflows, run the whole workflow inside `using_session(conversation_id)` and wrap each sub-agent's `run()` in its own `instrument_tags({"gen_ai.agent.name": agent.name})`. Maple then shows each agent in its own lane. `AgentWorkflow` handoffs happen inside one span, so they show as a single agent. +For multi-agent workflows, run the whole workflow inside `using_session(conversation_id)` and wrap each sub-agent's `run()` in its own `instrument_tags({"gen_ai.agent.name": agent.name})` to give each agent its own lane. `AgentWorkflow` handoffs show as a single agent. ## Get tokens on streamed calls -`FunctionAgent` streams every model call. OpenAI's API only sends usage on a stream when asked, so pass `stream_options` on OpenAI and OpenAI-compatible models: +`FunctionAgent` streams its model calls, and OpenAI only reports tokens on a stream when asked. Pass `stream_options` on OpenAI and OpenAI-compatible models: ```py from llama_index.llms.openai import OpenAI @@ -164,21 +155,21 @@ from llama_index.llms.openai import OpenAI llm = OpenAI(model="gpt-4o-mini", additional_kwargs={"stream_options": {"include_usage": True}}) ``` -OpenRouter sends usage without it. With `OpenAILike` or `OpenRouter`, also pass `is_function_calling_model=True`, or the agent never calls tools. +With `OpenAILike` or `OpenRouter`, also pass `is_function_calling_model=True`, or the agent never calls tools. ## Flush in short-lived processes -The SDK flushes on a normal interpreter exit, and long-running servers need nothing. In serverless handlers, notebooks and task workers, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. +Long-running servers need nothing. In serverless handlers, notebooks and task workers, import `provider` from `tracing` and call `provider.force_flush()` in a `finally` after each run. ## Check that it works Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. Within a minute you should see one session labeled **LlamaIndex**, with one turn per `agent.run()`, a transcript, one model call per request with tokens, and `FunctionTool.acall` tool calls. -Cost shows as **unpriced**, which is expected. Streamed model calls show about 1 ms of duration because the span ends when LlamaIndex hands back the stream. +Cost shows as **unpriced** and streamed model calls last about 1 ms. Both are expected. ## Troubleshooting -- **No spans at all.** Import `tracing` before the first `agent.run()`, check for `DependencyConflict` in the logs (llama-index-core older than 0.14.19), and look for exporter errors. +- **No spans at all.** Import `tracing` before the first `agent.run()`, check the logs for `DependencyConflict` (upgrade llama-index-core) and exporter errors. - **One session per message.** `agent.run()` isn't inside `using_session(...)`, or the id changes per request. - **Each model or tool call counted two or three times.** The exporter was added directly. Add it through `LlamaIndexForMaple`. - **No tokens on streamed calls.** Add `stream_options={"include_usage": True}` through `additional_kwargs`. @@ -187,7 +178,4 @@ Cost shows as **unpriced**, which is expected. Streamed model calls show about 1 ## Related - [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. -- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. -- [LlamaIndex observability](https://developers.llamaindex.ai/python/framework/module_guides/observability/): LlamaIndex's own page on tracing. -- [openinference-instrumentation-llama-index](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-llama-index): the instrumentor's source and `TraceConfig` options. - [OpenRouter](/docs/agent-tracing/openrouter): cost per call if your models go through OpenRouter. diff --git a/apps/landing/src/content/docs/agent-tracing/mastra.md b/apps/landing/src/content/docs/agent-tracing/mastra.md index 776a87eed0..50c2d9a7a8 100644 --- a/apps/landing/src/content/docs/agent-tracing/mastra.md +++ b/apps/landing/src/content/docs/agent-tracing/mastra.md @@ -7,15 +7,15 @@ navLabel: "Mastra" icon: "mastra" --- -Mastra traces every agent run, model call and tool call itself, and `@mastra/otel-exporter` sends those spans to Maple as OpenTelemetry GenAI spans. You don't need an OpenTelemetry SDK or an instrumentation package. +Mastra traces agent runs, model calls and tool calls itself, and `@mastra/otel-exporter` sends those spans to Maple. You don't need an OpenTelemetry SDK or an instrumentation package. -The session id is Mastra's memory thread id, so every call of a conversation must pass the same `memory: { thread }`. You also add a short span processor that fills in the prompt Mastra leaves off each model call. +The session id is Mastra's memory thread id, so every call of a conversation must pass the same `memory: { thread }`. -Tested with `@mastra/core` 1.71, `@mastra/observability` 1.18 and `@mastra/otel-exporter` 1.4 on Node.js 22.13 or newer. +You need `@mastra/core` 1.x and Node.js 22.13 or newer. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-mastra](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-mastra) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-mastra](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-mastra) skill and follows it. ```text Set up Maple agent tracing for Mastra in this project. @@ -33,11 +33,11 @@ Your ingest key is in **Settings → Ingestion**. npm install @mastra/observability@latest @mastra/otel-exporter@latest ``` -Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week. With mismatched releases, the exporter can pick the wrong span as the model call. +Keep `@mastra/core`, `@mastra/observability` and `@mastra/otel-exporter` on releases from the same week, or the exporter can pick the wrong span as the model call. ## Add the Maple span processor -This processor fixes three gaps in what Mastra 1.71 exports. It copies each model call's prompt onto its `chat` span, gives sub-agents the conversation's thread id instead of their own, and drops the raw provider response (headers, cookies, full body) from step spans. +Without this processor the transcript has no user messages, sub-agents land in separate sessions, and raw provider responses (including cookies) are exported. ```ts // src/mastra/maple-span-processor.ts @@ -109,13 +109,13 @@ export const mastra = new Mastra({ For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. -The endpoint, protocol and key have to be in code. This exporter ignores the `OTEL_EXPORTER_OTLP_*` variables and defaults to `http/json`. `observability` must be an `Observability` instance, since a plain object silently installs a no-op. +Set the endpoint, protocol and key in code, since this exporter ignores the `OTEL_EXPORTER_OTLP_*` variables. `observability` must be an `Observability` instance; a plain object silently traces nothing. Only agents and workflows registered on this `Mastra` instance are traced. Get them with `mastra.getAgent()` or `mastra.getWorkflow()`. ## Pass the thread id on every call -Maple groups traces into sessions by `gen_ai.conversation.id`, which Mastra sets from the memory thread. Pass the same thread on every call of a conversation: +Pass the same thread on every call of a conversation: ```ts const agent = mastra.getAgent("supportAgent") @@ -128,7 +128,7 @@ export async function handleMessage(chatId: string, userId: string, text: string } ``` -Use the chat id your app already has. It must stay the same for the whole conversation and differ between conversations. An agent that already uses `Memory` with this option needs nothing more. `agent.stream()` takes the same option; read the stream to the end, since the spans are exported when it finishes. +Use the chat id your app already has. It must stay the same for the whole conversation and differ between conversations. `agent.stream()` takes the same option; read the stream to the end, since the spans are exported when it finishes. Workflow runs, and agents called without `memory`, have no thread. Put the id in the root span's metadata instead: @@ -140,30 +140,27 @@ const result = await run.start({ }) ``` -Give every `Agent` a distinct `name`, since Maple draws one lane per agent name. When a workflow step calls an agent, pass it the step's `tracingContext` (`agent.generate(prompt, { tracingContext })`) so the agent joins the workflow's trace. +Give every `Agent` a distinct `name`, or sub-agents share one lane. When a workflow step calls an agent, pass it the step's `tracingContext` (`agent.generate(prompt, { tracingContext })`) so the agent joins the workflow's trace. ## Flush in scripts and serverless functions -The exporter sends spans every 5 seconds. In a script, call `await mastra.shutdown()` in a `finally` block before exiting. In a serverless handler, call `await mastra.observability.flush()` at the end of each request, after any streamed response has finished. +A short-lived process can exit before its spans are sent. In a script, call `await mastra.shutdown()` in a `finally` block before exiting. In a serverless handler, call `await mastra.observability.flush()` at the end of each request, after any streamed response has finished. ## Check that it works Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session named after your thread id with framework **Mastra**, one turn per `generate()` or `stream()` call, and a transcript with the prompts, replies and tool calls. -Cost shows as unpriced, because Mastra doesn't report it. If nothing arrives, set `logLevel: "debug"` on `OtelExporter` to log each export as `Export completed` or `Export FAILED` with the reason. +Cost shows as unpriced because Mastra doesn't report it. If nothing arrives, set `logLevel: "debug"` on `OtelExporter` to log each export as `Export completed` or `Export FAILED` with the reason. ## Troubleshooting - **Nothing arrives and there is no error.** `observability` is a plain object instead of `new Observability(...)`, or the agent isn't registered on the `Mastra` instance. - **`http/protobuf exporter is not installed` at startup.** The install skipped optional dependencies. Install `@opentelemetry/exporter-trace-otlp-proto`. - **Every message is its own session.** The call has no `memory: { thread, resource }`, or the thread id changes per request. For workflows, use `tracingOptions.metadata.threadId`. -- **The transcript has replies but no user messages, or a supervisor run lands in a session named `-`.** `mapleSpanProcessor` is missing from `spanOutputProcessors`. +- **The transcript has no user messages, or sub-agents land in a session named `-`.** Add `mapleSpanProcessor` to `spanOutputProcessors`. - **A failed tool shows as successful.** The tool returned an error value. Throw an `Error` instead. ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [All agent tracing guides](/docs/agent-tracing) - [Mastra: OpenTelemetry exporter](https://mastra.ai/docs/observability/tracing/exporters/otel) -- [Mastra: tracing overview](https://mastra.ai/docs/observability/tracing/overview) -- [Trace Vercel AI SDK agents](/docs/agent-tracing/vercel-ai-sdk) diff --git a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md index e0b9ade0df..22746f9272 100644 --- a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md +++ b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md @@ -7,9 +7,7 @@ navLabel: "Microsoft Agent Framework" icon: "dotnet" --- -Microsoft Agent Framework (MAF) emits OpenTelemetry spans for every agent run, model call and tool call in Python and .NET. It does not set a conversation id when your app keeps chat history itself, so you add one with a short span processor, or every turn shows up as its own session. - -Tested with `agent-framework` 1.19 (Python), `Microsoft.Agents.AI` 1.22 (.NET) and Semantic Kernel 1.44 (Python). +Microsoft Agent Framework (MAF) emits OpenTelemetry spans in Python and .NET. You point them at Maple and add a short span processor that sets the conversation id, or every turn shows up as its own session. ## Quick setup with a coding agent @@ -27,13 +25,13 @@ Your ingest key is in **Settings → Ingestion**. EU organizations should say EU ## Export traces from Python -MAF installs no exporter, so add the OTLP/HTTP one: +Install MAF with the OTLP/HTTP exporter: ```bash pip install "agent-framework-core>=1.19.0" "agent-framework-openai>=1.14.4" opentelemetry-exporter-otlp-proto-http ``` -Call `configure_otel_providers()` once at startup, before you create agents. `enable_sensitive_data=True` records prompts, replies and tool arguments and results; without it the transcript is empty. +Call `configure_otel_providers()` once at startup, before you create agents. Without `enable_sensitive_data=True` the transcript is empty. ```py # telemetry.py @@ -58,7 +56,7 @@ trace.get_tracer_provider().add_span_processor(ConversationIdProcessor()) Keep `otlp_protocol="http/protobuf"`. MAF defaults to gRPC, which Maple doesn't accept. -If your app already has a `TracerProvider` (Azure Monitor, Logfire, your own), don't call `configure_otel_providers()`. Add Maple's exporter and `ConversationIdProcessor` to your provider, then call `enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)` from `agent_framework.observability`. +If your app already has a `TracerProvider`, don't call `configure_otel_providers()`. Add Maple's exporter and `ConversationIdProcessor` to your provider, then call `enable_instrumentation(enable_sensitive_data=True, enable_message_events=False)` from `agent_framework.observability`. ## Export traces from .NET @@ -68,8 +66,6 @@ dotnet add package Microsoft.Agents.AI.OpenAI --version 1.22.0 dotnet add package OpenTelemetry.Exporter.OpenTelemetryProtocol --version 1.19.1 ``` -`UseOpenTelemetry()` on the agent also instruments its chat client: - ```csharp using System.ClientModel; using Microsoft.Agents.AI; @@ -106,11 +102,11 @@ AIAgent agent = openAi.GetChatClient("gpt-4o-mini").AsIChatClient() .Build(); ``` -Keep the leading `*` in `AddSource`: the default source names start with `Experimental.`, so `AddSource("Microsoft.Agents.AI.*")` matches nothing. Keep `/v1/traces` in the endpoint and the `HttpProtobuf` line. Pass each tool a `name`, or a local function in `Program.cs` shows up as something like `_Main_g_GetWeather_0_3`. +Keep the leading `*` in `AddSource` (the source names start with `Experimental.`), `/v1/traces` in the endpoint, and the `HttpProtobuf` line. Pass each tool a `name`, or local functions show up under compiler-generated names. ## Group each conversation into one session -Maple groups turns by `gen_ai.conversation.id`. MAF sets it only when the provider stores the conversation (Responses API with `store=True`, Foundry agents). This processor stamps your id on every span started inside a `conversation()` block: +Maple groups turns into a session by `gen_ai.conversation.id`. This processor sets it on every span started inside a `conversation()` block: ```py # maple_tracing.py @@ -139,21 +135,21 @@ def conversation(conversation_id: str): _conversation_id.reset(token) ``` -Wrap each request in it. `session.session_id` works as the id, or create the session with your own chat id: `agent.create_session(session_id=chat_id)`. +Wrap each request in it. Use `session.session_id` as the id, or create the session with your own chat id: `agent.create_session(session_id=chat_id)`. ```py -from agent_framework import AgentSession +from agent_framework import Agent, AgentSession from maple_tracing import conversation -async def handle_message(session: AgentSession, text: str) -> str: +async def handle_message(agent: Agent, session: AgentSession, text: str) -> str: with conversation(session.session_id): response = await agent.run(text, session=session) return response.text ``` -When streaming, keep the whole `async for` loop inside the block. Build workflows inside it too, or `WorkflowBuilder.build()` shows up as a separate one-span session. Don't use `gen_ai.agent.id` or the `conversation_id` chat option as the id; the first is shared by every user and the second turns off MAF's in-memory history. +When streaming, keep the whole `async for` loop inside the block. Don't pass the `conversation_id` chat option instead; it turns off MAF's in-memory history. In .NET, use an `Activity` processor with an `AsyncLocal` and set it before `RunAsync`: @@ -180,7 +176,7 @@ var response = await agent.RunAsync(userMessage, session); ## Flush before a script exits -`configure_otel_providers()` batches spans. Scripts, CLIs and notebooks that exit without flushing lose their last turns, so shut the providers down in a `finally`: +Scripts, CLIs and notebooks that exit without flushing lose their last turns. Shut the providers down in a `finally`: ```py from opentelemetry import _logs, metrics, trace @@ -188,7 +184,7 @@ from opentelemetry import _logs, metrics, trace def shutdown_telemetry() -> None: for provider in (trace.get_tracer_provider(), metrics.get_meter_provider(), _logs.get_logger_provider()): - provider.shutdown() # exports whatever is still buffered + provider.shutdown() try: @@ -201,7 +197,7 @@ In a serverless handler, call `trace.get_tracer_provider().force_flush()` before ## Semantic Kernel -Semantic Kernel (SK) reads its telemetry switches at import time, so set them before the first `import semantic_kernel`. SK brings no exporter, so configure your own provider with the same `ConversationIdProcessor`: +Semantic Kernel (SK) reads its telemetry switches at import time, so set them before the first `import semantic_kernel`. Configure your own provider with the same `ConversationIdProcessor`: ```bash pip install "semantic-kernel>=1.44.1" opentelemetry-sdk opentelemetry-exporter-otlp-proto-http @@ -235,11 +231,11 @@ provider.add_span_processor( trace.set_tracer_provider(provider) ``` -Wrap each turn in `with conversation(thread_id):`. The transcript comes from `ChatCompletionAgent`, so pass messages positionally (`await agent.get_response(text, thread=thread)`); the `messages=` keyword records an empty input. Code that calls the kernel without an agent has no transcript in Maple. +Wrap each turn in `with conversation(thread.id):`. Pass messages positionally, as in `await agent.get_response(text, thread=thread)`; the `messages=` keyword records an empty input. Only `ChatCompletionAgent` calls produce a transcript; calling the kernel directly doesn't. ## Check that it works -Run a conversation of two or three turns where one turn calls a tool. Within about a minute, **Agent Sessions** shows one session for it, labeled **Microsoft Agent Framework** or **Semantic Kernel** (.NET shows **Unidentified**), with one turn per `agent.run()`, the transcript, and tool calls with their arguments and results. Cost shows as unpriced because MAF doesn't emit one. +Run a conversation of two or three turns where one turn calls a tool. Within about a minute, **Agent Sessions** shows one session for it, labeled **Microsoft Agent Framework** or **Semantic Kernel** (.NET shows **Unidentified**), with one turn per `agent.run()`, the transcript, and tool calls with their arguments and results. Cost shows as unpriced; MAF doesn't emit cost. ## Troubleshooting @@ -251,8 +247,7 @@ Run a conversation of two or three turns where one turn calls a tool. Within abo ## Related -- [Agent Sessions](/docs/agent-sessions/overview): what a session, turn, model call and tool call are in Maple. -- [Trace your AI agent](/docs/agent-tracing): guides for other frameworks. +- [Agent Sessions](/docs/agent-sessions/overview): reading a session in Maple. - [Python instrumentation](/docs/guides/instrumentation-python) and [.NET instrumentation](/docs/guides/instrumentation-csharp): tracing the rest of the service. - [Agent Framework observability](https://learn.microsoft.com/en-us/agent-framework/agents/observability): Microsoft's reference for the settings above. - [Semantic Kernel telemetry](https://learn.microsoft.com/en-us/semantic-kernel/concepts/enterprise-readiness/observability/): the SK diagnostics switches. diff --git a/apps/landing/src/content/docs/agent-tracing/openai-agents.md b/apps/landing/src/content/docs/agent-tracing/openai-agents.md index 45585c0c0e..b897ddcf6b 100644 --- a/apps/landing/src/content/docs/agent-tracing/openai-agents.md +++ b/apps/landing/src/content/docs/agent-tracing/openai-agents.md @@ -7,11 +7,7 @@ navLabel: "OpenAI Agents SDK" icon: "openai" --- -The OpenAI Agents SDK traces every run, but it uploads those traces to the OpenAI dashboard instead of exporting OpenTelemetry. OpenInference's `openinference-instrumentation-openai-agents` bridges them to OpenTelemetry so Maple can read them. - -The part to get right is the session id. The SDK's `group_id` never reaches OpenTelemetry, so you wrap each run in OpenInference's `using_session`, or every message shows up as its own session. - -Tested with `openai-agents` 0.22 and `openinference-instrumentation-openai-agents` 2.5 on Python 3.10 to 3.14. +OpenInference's `openinference-instrumentation-openai-agents` exports OpenAI Agents SDK runs to Maple. Wrap each run in its `using_session` helper, or every message shows up as its own session. ## Quick setup with a coding agent @@ -41,7 +37,7 @@ export OTEL_EXPORTER_OTLP_ENDPOINT=https://ingest.maple.dev export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" ``` -EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code instead, give the full URL ending in `/v1/traces`. +EU organizations use `https://ingest.eu.maple.dev`. If you pass `endpoint=` to `OTLPSpanExporter` in code, give the full URL ending in `/v1/traces`. ## Register the bridge @@ -109,19 +105,19 @@ OpenAIAgentsInstrumentor().instrument( ) ``` -`enable_genai_semconv=True` makes the bridge write the OpenTelemetry GenAI attributes that Maple builds the transcript from. Without it, the session page is empty. +Without `enable_genai_semconv=True`, the session page is empty. -`MapleSpanFixes` corrects three things the bridge records wrong: tool arguments (it would copy the tool's JSON schema), handoff tool names, and the reply of streamed Chat Completions calls. It has to run before the OpenInference processor, which is why `set_trace_processors` comes first and `instrument()` gets `exclusive_processor=False`. +`MapleSpanFixes` fixes tool arguments, handoff tool names and streamed replies. It has to run before the OpenInference processor, so keep the order shown. -`set_trace_processors` also stops the upload to OpenAI, so tracing needs no OpenAI key. Don't use `set_tracing_disabled(True)` or `OPENAI_AGENTS_DISABLE_TRACING=1` for that. They turn off the pipeline the bridge reads from, and you get no spans. +`set_trace_processors` already stops the upload to OpenAI. Don't use `set_tracing_disabled(True)` or `OPENAI_AGENTS_DISABLE_TRACING=1` for that, or you get no spans. If the app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add the OTLP exporter to it and pass it to `instrument()` instead of creating a second one. -If an agent uses `OpenAIChatCompletionsModel` with a base URL other than OpenAI's (OpenRouter, LiteLLM, vLLM, Ollama), give it `model_settings=ModelSettings(include_usage=True)`. Otherwise streamed turns report zero tokens. +If an agent uses `OpenAIChatCompletionsModel` with a non-OpenAI base URL (OpenRouter, LiteLLM, vLLM, Ollama), give it `model_settings=ModelSettings(include_usage=True)`. Otherwise streamed turns report zero tokens. ## Group each conversation into one session -Maple groups this framework's traces by `session.id`, and the bridge only sets it inside `using_session`. Wrap every run: +Wrap every run in `using_session` with the conversation's id: ```py from agents import RunConfig, Runner, SQLiteSession @@ -139,15 +135,15 @@ async def handle_message(conversation_id: str, text: str) -> str: return result.final_output ``` -Use the id your app already stores the chat under, and give the SDK session the same one. A new UUID per request gives you one session per message, and a constant puts every user in one session. +Use the id your app already stores the chat under, and give the SDK session the same one. A new UUID per request gives you one session per message. -For streaming, call `Runner.run_streamed` inside the `with` block. Iterating `stream_events()` after the block is fine. +For streaming, call `Runner.run_streamed` inside the `with` block. -`workflow_name` names each trace's root span. Keep `chat`, `completion` and `tool` out of it, or Maple counts a phantom model or tool call per run. A name ending in `workflow` or `agent` is safe. +Keep `chat`, `completion` and `tool` out of `workflow_name`, or Maple counts a phantom model or tool call per run. ## Flush in short-lived processes -The provider flushes on a normal interpreter exit. A killed process, a serverless invocation or a notebook needs an explicit flush: +In serverless functions and notebooks, flush explicitly: ```py from tracing import provider @@ -158,17 +154,15 @@ finally: provider.force_flush() # serverless: before returning; notebooks: after each run ``` -The SDK's `flush_traces()` doesn't help here, because the spans wait in the OpenTelemetry batch processor. +The SDK's `flush_traces()` doesn't flush these spans. ## Check that it works -Send two or three messages with the same conversation id, one of them using a tool, then open **Agent Sessions**. You should see one session with the framework **OpenAI Agents SDK**, one turn per `Runner.run`, a transcript with the tool calls and their arguments, and input and output tokens on every model call. - -Cost shows as unpriced, because neither the SDK nor the bridge records it. +Send two or three messages with the same conversation id, one using a tool, then open **Agent Sessions**. You should see one session with the framework **OpenAI Agents SDK**, one turn per `Runner.run`, tool calls with their arguments, and tokens on every model call. Cost shows as unpriced. ## TypeScript -The TypeScript bridge, `@arizeai/openinference-instrumentation-openai-agents`, works with less detail: the framework shows as **Unidentified**, model replies are missing from the transcript, there are no agent lanes, and the session id has to be set as `gen_ai.conversation.id`. The [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) has the setup. For the full session page in TypeScript, emit the GenAI attributes yourself as in the [provider SDKs guide](/docs/agent-tracing/provider-sdks). +The TypeScript bridge, `@arizeai/openinference-instrumentation-openai-agents`, gives less detail: the framework shows as **Unidentified**, model replies are missing from the transcript, and there are no agent lanes. It takes the session id as `gen_ai.conversation.id`. The [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) has the setup. For the full session page, emit the GenAI attributes yourself as in the [provider SDKs guide](/docs/agent-tracing/provider-sdks). ## Troubleshooting @@ -181,7 +175,5 @@ The TypeScript bridge, `@arizeai/openinference-instrumentation-openai-agents`, w ## Related - [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. -- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. -- [Tracing in the OpenAI Agents SDK](https://openai.github.io/openai-agents-python/tracing/): the SDK's span types, `trace()`, and the sensitive-data switch. -- [openinference-instrumentation-openai-agents](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-openai-agents): the bridge's source. +- [Tracing in the OpenAI Agents SDK](https://openai.github.io/openai-agents-python/tracing/): the switch that turns off prompt and reply capture. - [OpenRouter](/docs/agent-tracing/openrouter) and [LiteLLM](/docs/agent-tracing/litellm): if your models go through either gateway. diff --git a/apps/landing/src/content/docs/agent-tracing/openrouter.md b/apps/landing/src/content/docs/agent-tracing/openrouter.md index 8e67a2bbd5..1887e384d5 100644 --- a/apps/landing/src/content/docs/agent-tracing/openrouter.md +++ b/apps/landing/src/content/docs/agent-tracing/openrouter.md @@ -7,13 +7,11 @@ navLabel: "OpenRouter" icon: "openrouter" --- -OpenRouter Broadcast exports a trace for every request made with your OpenRouter account, with the model, tokens, the cost OpenRouter charged, and the prompt and completion. You set it up once in the OpenRouter dashboard. The one thing to change in your code is the `session_id` field: without it, every model call is its own session. - -Code samples use the `openai` SDK (npm 7.23, PyPI 3.20), `@openrouter/ai-sdk-provider` 3.1 and `@openrouter/sdk` 1.3. +OpenRouter Broadcast sends a trace of every request on your OpenRouter account to Maple, with tokens, cost, prompt and completion. You set it up once in the OpenRouter dashboard, then add a `session_id` to your requests. Without it, every model call is its own session. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-openrouter](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openrouter) skill, which contains every step of this guide. +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-openrouter](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openrouter) skill and follows it. ```text Set up Maple agent tracing for OpenRouter in this project. @@ -23,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. The agent can't change your OpenRouter dashboard, so it ends by printing the values to enter in the next section. +Your ingest key is in **Settings → Ingestion**. The agent prints the values for the OpenRouter dashboard, which you enter yourself as described in the next section. ## Point Broadcast at Maple @@ -42,11 +40,11 @@ Your ingest key is in **Settings → Ingestion**. The agent can't change your Op 4. Click **Test Connection**. OpenRouter only saves the destination if the test passes. -Leave the sampling rate at 1.0. Sampling is per session, so a lower rate drops whole conversations. Leave the API key filter empty to export every key, and if your app calls `eu.openrouter.ai`, add the Europe data region. +Leave the sampling rate at 1.0, since a lower rate drops whole conversations. Leave the API key filter empty. If your app calls `eu.openrouter.ai`, add the Europe data region. ## Send a session id with every request -Maple groups calls into a session by the `session_id` field of your request body (or the `x-session-id` header). Send the same id on every request of a conversation, such as your chat thread id, and a new one for each conversation. +Send a `session_id` field in the request body (or an `x-session-id` header). Use the same id on every request of a conversation, such as your chat thread id, and a new one for each conversation. With the `openai` SDK in TypeScript, the field isn't in the types, so it needs a `@ts-expect-error`: @@ -101,14 +99,13 @@ OpenRouter's own SDKs have a typed field: `sessionId` in `@openrouter/sdk` and ` ## Nest Broadcast under your own traces -Skip this if OpenRouter is your only source of traces. If your app already sends its own traces to Maple, every model call is otherwise recorded twice. Pass the active span's ids in the `trace` field, and OpenRouter places its spans inside your trace. +Skip this if OpenRouter is your only source of traces. If your app also sends its own traces to Maple, every model call is recorded twice. Pass the active span's ids in the `trace` field so OpenRouter places its spans inside your trace. In TypeScript, wrap `fetch` and pass it to the client (`new OpenAI({ baseURL, apiKey, fetch: openRouterFetch })` or `createOpenRouter({ apiKey, fetch: openRouterFetch })`): ```ts import { trace } from "@opentelemetry/api" -// Nests each OpenRouter Broadcast trace under the span that made the request. export const openRouterFetch: typeof fetch = (input, init) => { const span = trace.getActiveSpan()?.spanContext() if (span && typeof init?.body === "string") { @@ -142,13 +139,13 @@ client.chat.completions.create(model=model, messages=messages, extra_body=openro Use the same value for `session_id` as your framework's conversation id. -Broadcast only sees requests to OpenRouter, so it has no tool calls or agent names. For those, instrument your app with its [framework guide](/docs/agent-tracing) and nest Broadcast under it as shown here. +Broadcast has no tool calls or agent names. For those, trace your app with its [framework guide](/docs/agent-tracing) and nest Broadcast under it as shown here. ## Check that it works -Broadcast sends traces from OpenRouter's servers, so your app needs no flush, but expect about a minute of delay. Run a conversation of three turns with the same `session_id`, then open **Agent Sessions** and filter by service `openrouter`. +Run a conversation of three turns with the same `session_id`. After about a minute, open **Agent Sessions** and filter by service `openrouter`. -You should see one session named after your `session_id`, with vendor **OpenRouter**, an `LLM Generation` span per model call, and tokens and cost in USD. The transcript shows each call's prompt and completion as a raw JSON block. To keep content out of Maple, turn on **Privacy Mode** on the destination. Tokens and cost still arrive. +You should see one session named after your `session_id`, with vendor **OpenRouter**, an `LLM Generation` span per model call, and tokens and cost in USD. The transcript shows each call's prompt and completion as a raw JSON block. To keep content out of Maple, turn on **Privacy Mode** on the destination. ## Troubleshooting @@ -161,6 +158,5 @@ You should see one session named after your `session_id`, with vendor **OpenRout ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [Agent tracing guides](/docs/agent-tracing) -- [OpenRouter Broadcast](https://openrouter.ai/docs/guides/features/broadcast) and its [OpenTelemetry Collector destination](https://openrouter.ai/docs/guides/features/broadcast/otel-collector) +- [OpenRouter Broadcast](https://openrouter.ai/docs/guides/features/broadcast) - [Vercel AI SDK](/docs/agent-tracing/vercel-ai-sdk), if you call OpenRouter through `@openrouter/ai-sdk-provider` diff --git a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md index 1960db1524..a8331db33e 100644 --- a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md +++ b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md @@ -7,11 +7,9 @@ navLabel: "Any language (OTel GenAI)" icon: "opentelemetry" --- -Use this guide when no other guide covers your agent, for example a hand-written agent loop. You write the agent spans yourself with any OpenTelemetry SDK. Maple needs three kinds: an `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. The details that break most setups are the conversation id and sending messages as JSON strings. +Use this guide when no other guide covers your agent, for example a hand-written agent loop. You write the agent spans yourself with any OpenTelemetry SDK. Maple needs three kinds: an `invoke_agent` span per user turn, a `chat` span per model call and an `execute_tool` span per tool call. -Tested with the OpenTelemetry JS SDK 2.11 on Node.js 26 and the Python SDK 1.45 on Python 3.14, calling OpenRouter, against the [GenAI conventions](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. - -If you use a framework, check the [framework guides](/docs/agent-tracing) first. Most of them emit these spans for you. +Tested with the OpenTelemetry JS SDK 2.11 on Node.js 26 and the Python SDK 1.45 on Python 3.14, against the [GenAI conventions](https://github.com/open-telemetry/semantic-conventions-genai) as of September 2026. ## Quick setup with a coding agent @@ -38,21 +36,21 @@ invoke_agent support gen_ai.conversation.id = chat_42 └── chat openai/gpt-4o-mini model call: final answer ``` -Maple classifies a span by `gen_ai.operation.name`, not by its name. Agent Sessions ignores a span without that attribute, even if it carries a model or token counts. +Set `gen_ai.operation.name` on every span. Agent Sessions ignores spans without it. | Span (kind) | Attribute | Value | | --- | --- | --- | | `invoke_agent` (`INTERNAL`) | `gen_ai.operation.name` | `invoke_agent` | -| | `gen_ai.agent.name` | `support`. Each sub-agent gets a lane named after it. | +| | `gen_ai.agent.name` | `support` | | | `gen_ai.conversation.id` | your chat or thread id, see [sessions](#group-turns-into-one-session) | | | `gen_ai.input.messages`, `gen_ai.output.messages` | optional, JSON string | -| `chat` (`CLIENT`) | `gen_ai.operation.name` | `chat` (`generate_content` and `text_completion` also work) | +| `chat` (`CLIENT`) | `gen_ai.operation.name` | `chat` | | | `gen_ai.provider.name` | the API you called: `openai`, `anthropic`, `gcp.gemini`, `openrouter`... | | | `gen_ai.request.model`, `gen_ai.response.model` | model ids | | | `gen_ai.response.id` | the provider's response id | | | `gen_ai.usage.input_tokens`, `gen_ai.usage.output_tokens` | int | | | `gen_ai.usage.cache_read.input_tokens`, `gen_ai.usage.cache_write.input_tokens`, `gen_ai.usage.reasoning.output_tokens` | int, optional | -| | `gen_ai.usage.cost` | double in USD, optional (not part of the spec) | +| | `gen_ai.usage.cost` | double in USD, optional | | | `gen_ai.system_instructions`, `gen_ai.input.messages`, `gen_ai.output.messages` | JSON string | | | `gen_ai.response.finish_reasons` | string array, e.g. `["stop"]` | | | `gen_ai.response.time_to_first_chunk` | double, in **seconds**, streamed calls | @@ -62,9 +60,9 @@ Maple classifies a span by `gen_ai.operation.name`, not by its name. Agent Sessi | | `gen_ai.tool.call.arguments` | JSON string of an object | | | `gen_ai.tool.call.result` | JSON string of an object or array (a bare string is dropped) | -Mark a failed span of any kind with status `ERROR` and an `error.type` attribute. When a tool fails, you can still return the error to the model as its result; the span status is what Maple counts. +Mark a failed span with status `ERROR` and an `error.type` attribute, even when you return a tool's error to the model as its result. -Put token usage on `chat` spans only, copied from the provider's response as is. Maple reads the input count by `gen_ai.provider.name`: for `anthropic`, send Anthropic's raw `input_tokens`, which exclude cache, or cached tokens are counted twice. On OpenAI's streaming API, set `stream_options: { include_usage: true }`, or streamed calls report no tokens. Maple never prices tokens, so a session without `gen_ai.usage.cost` shows as unpriced. OpenRouter returns the cost in `usage.cost`. +Put token usage on `chat` spans only, copied from the provider's response as is. For `anthropic`, send Anthropic's raw `input_tokens`, which exclude cache, or cached tokens are counted twice. On OpenAI's streaming API, set `stream_options: { include_usage: true }`, or streamed calls report no tokens. Maple doesn't price tokens, so a session without `gen_ai.usage.cost` shows as unpriced. OpenRouter returns the cost in `usage.cost`. ### The message format @@ -83,9 +81,9 @@ Put token usage on `chat` spans only, copied from the provider's response as is. Output messages add a `finish_reason` to each message. `gen_ai.system_instructions` is an array of parts without a role: `[{"type":"text","content":"You are a concise assistant."}]`. -Always set these as a string holding JSON. Maple drops plain text, and it doesn't read messages from span events, logs or indexed keys like `gen_ai.prompt.0.content`. Leave `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` and `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` unset: a long history cut mid-string no longer parses and Maple drops the whole attribute. +Set these as JSON strings. Maple drops plain text and doesn't read span events, logs or indexed keys like `gen_ai.prompt.0.content`. Leave `OTEL_ATTRIBUTE_VALUE_LENGTH_LIMIT` and `OTEL_SPAN_ATTRIBUTE_VALUE_LENGTH_LIMIT` unset, because a truncated message array no longer parses. -To keep prompts and results out of Maple, skip the five content attributes. Models, tokens, tool names and errors still show up, with an empty transcript. +To keep prompts and results out of Maple, skip the five content attributes. Everything else still shows up, with an empty transcript. ## Export spans to Maple @@ -148,18 +146,17 @@ provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) ``` -If your app already has a `TracerProvider` (from Sentry, Datadog, `opentelemetry-instrument` or `NodeSDK`), add the `BatchSpanProcessor` to it instead of creating a second one. +If your app already has a `TracerProvider` (Sentry, Datadog, `opentelemetry-instrument`, `NodeSDK`), add the `BatchSpanProcessor` to it instead of creating a second one. -Name the tracer after your app, like `support-agent`. A tracer named after a framework or gateway, such as `openrouter` or `langsmith`, makes Maple read that framework's session key and ignore `gen_ai.conversation.id`. +Name the tracer after your app, like `support-agent`. A tracer named after a framework or gateway, such as `openrouter` or `langsmith`, makes Maple ignore `gen_ai.conversation.id`. ## Instrument the agent loop -Complete, tested loops that stream, call tools and record every attribute above are in the skill: [TypeScript](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/typescript.md), [Python](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/python.md) and [Go](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/go.md) (not compiled; run `go vet`). Keep your own loop and wrap its existing calls. +Complete loops that stream, call tools and record every attribute above are in the skill: [TypeScript](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/typescript.md), [Python](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/python.md) and [Go](https://github.com/MapleTechLabs/maple/blob/main/skills/maple-agent-tracing-opentelemetry/references/go.md) (untested; run `go vet`). Wrap your own loop's existing calls the same way. This is the `execute_tool` span from the TypeScript version. The `chat` span follows the same pattern around each model call: ```ts -// One tool call = one `execute_tool` span. A failure is marked on the span and returned to the model. async function runTool(agent: Agent, call: ToolCall) { const name = call.function.name return tracer.startActiveSpan( @@ -196,7 +193,7 @@ Start the `chat` and `execute_tool` spans inside the `invoke_agent` span's callb ## Group turns into one session -Set `gen_ai.conversation.id` on the `invoke_agent` span of every turn, using the id your app already has for the conversation. Every span in that trace joins the session. A chat backend passes it once per user message: +Set `gen_ai.conversation.id` on the `invoke_agent` span of every turn, using the id your app already has for the conversation. A chat backend passes it once per user message: ```ts // One history per conversation. Store it in your database in a real backend. @@ -210,12 +207,11 @@ export async function handleMessage(chatId: string, text: string, onText?: (delt } ``` -Without the id, each trace becomes its own one-turn session named `trace:`. A UUID generated per request, or the trace id, has the same effect. Don't give sub-agents their own id: they inherit the session from the trace, and with two ids in one trace Maple keeps only one. +Without the id, or with one generated per request, each trace becomes its own one-turn session named `trace:`. Don't give sub-agents their own id; they inherit the session from the trace. If a framework writes its session id under a key Maple doesn't read, wrap each turn in your own span that carries `maple_ai.session.id`, and run the framework inside it: ```ts -// chatId comes from your request; frameworkAgent is the framework's agent await tracer.startActiveSpan( "invoke_agent support", { @@ -235,7 +231,7 @@ await tracer.startActiveSpan( ) ``` -Put `maple_ai.session.id` only on your own wrapper span. On a framework's spans it replaces the framework's attribute decoding. +Put `maple_ai.session.id` only on your own wrapper span, never on the framework's spans. ## Flush before a short-lived process exits @@ -243,7 +239,7 @@ Put `maple_ai.session.id` only on your own wrapper span. On a framework's spans ## Check that it works -Run one conversation with two messages and a tool call, then open **Agent Sessions** in Maple. After a few seconds you should see one session with your conversation id, one turn per message with its transcript, and `chat` and `execute_tool` spans nested under each turn's `invoke_agent` span. Hand-written spans show the framework as **Unidentified**, which is expected. +Run one conversation with two messages and a tool call, then open **Agent Sessions** in Maple. After a few seconds you should see one session with your conversation id, one turn per message with its transcript, and `chat` and `execute_tool` spans nested under each turn's `invoke_agent` span. The framework shows as **Unidentified**, which is expected. ## Troubleshooting @@ -258,5 +254,4 @@ Run one conversation with two messages and a tool call, then open **Agent Sessio - [Agent Sessions overview](/docs/agent-sessions/overview) - [All agent tracing guides](/docs/agent-tracing) - [Provider SDKs](/docs/agent-tracing/provider-sdks), for auto-instrumented OpenAI, Anthropic and Gemini clients -- [OpenTelemetry GenAI semantic conventions](https://github.com/open-telemetry/semantic-conventions-genai) - [GenAI spans](https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-spans.md) and [GenAI agent spans](https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-agent-spans.md) diff --git a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md index b712506426..bf421346fd 100644 --- a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md +++ b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md @@ -7,9 +7,7 @@ navLabel: "OpenAI, Anthropic & Gemini SDKs" icon: "openai" --- -If your agent is your own loop around `client.chat.completions.create`, `client.messages.create` or `client.models.generate_content`, an instrumentation library can record each model call. It can't see where a turn starts, which conversation it belongs to, or the tools your code runs, so you add those spans yourself and put the conversation id on the turn span. You also have to turn on content capture, which is off by default. - -Tested with Python `openai` 3.20.0 and `anthropic` 1.8.0 (instrumentations 1.2b0) and TypeScript `openai` 7.23.0. The Gemini path follows the instrumentation's documentation and hasn't been run against a live model yet. If you use an agent framework on top of these SDKs, use [that framework's guide](/docs/agent-tracing) instead. +Use this guide when your agent is your own loop around `client.chat.completions.create`, `client.messages.create` or `client.models.generate_content`. An instrumentation library records each model call, and you add the turn and tool spans and put the conversation id on the turn span. If you use an agent framework on top of these SDKs, use [that framework's guide](/docs/agent-tracing) instead. ## Quick setup with a coding agent @@ -27,8 +25,6 @@ Your ingest key is in **Settings → Ingestion**. EU organizations should say EU ## Configure the exporter -Both languages read the standard OpenTelemetry variables: - ```bash export OTEL_EXPORTER_OTLP_ENDPOINT="https://ingest.maple.dev" export OTEL_EXPORTER_OTLP_HEADERS="Authorization=Bearer YOUR_INGEST_KEY" @@ -36,11 +32,11 @@ export OTEL_EXPORTER_OTLP_PROTOCOL="http/protobuf" export OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT="SPAN_ONLY" ``` -For an EU organization, use `https://ingest.eu.maple.dev`. `SPAN_ONLY` records prompts and replies as span attributes, which Maple needs for the transcript. Leave it unset to keep content out of Maple. `EVENT_ONLY` or `true` leaves the transcript empty. +For an EU organization, use `https://ingest.eu.maple.dev`. `SPAN_ONLY` records prompts and replies for the transcript. Leave it unset to keep content out of Maple. `EVENT_ONLY` or `true` leaves the transcript empty. ## Python: install the GenAI instrumentation -Use the official OpenTelemetry GenAI packages. Note the `genai` in the name: `opentelemetry-instrumentation-openai` is a different project (OpenLLMetry), and `opentelemetry-instrumentation-openai-v2` is deprecated. +Install the `genai` packages below, not `opentelemetry-instrumentation-openai` or `opentelemetry-instrumentation-openai-v2`. ```bash pip install "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" \ @@ -59,7 +55,6 @@ from opentelemetry.sdk.trace import TracerProvider from opentelemetry.sdk.trace.export import BatchSpanProcessor provider = TracerProvider(resource=Resource.create({"service.name": "support-agent"})) -# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) @@ -68,11 +63,11 @@ OpenAIInstrumentor().instrument() # from opentelemetry.instrumentation.google_genai import GoogleGenAiSdkInstrumentor ``` -Import `tracing` first in your entry point so the SDK is patched before the first request. If your app already has a `TracerProvider` (Sentry, Logfire, Datadog), add the `BatchSpanProcessor` to it instead of creating a second. +Import `tracing` first in your entry point. If your app already has a `TracerProvider` (Sentry, Logfire, Datadog), add the `BatchSpanProcessor` to it instead. ## Python: wrap each turn and tool call -Maple groups traces into a session by `gen_ai.conversation.id`. Put it on an `invoke_agent` span that wraps the whole turn, so the model and tool calls inside become one trace. Use the id your app already stores for the conversation (thread id, ticket id), not a new UUID per request. +Wrap each turn in an `invoke_agent` span that carries `gen_ai.conversation.id`. Use the id your app already stores for the conversation (thread id, ticket id), not a new UUID per request. ```py # agent_tracing.py @@ -85,7 +80,6 @@ from opentelemetry.trace import Status, StatusCode tracer = trace.get_tracer("support-agent") -# Same switch the instrumentors read, so one env var controls content everywhere. CAPTURE_CONTENT = os.environ.get( "OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT", "" ).upper() in ("SPAN_ONLY", "SPAN_AND_EVENT") @@ -114,7 +108,6 @@ def run_tool(call_id: str, name: str, arguments: str, tool) -> str: try: result = json.dumps(tool(**json.loads(arguments))) except Exception as exc: - # The exception never leaves this block, so mark the span failed by hand. span.record_exception(exc) span.set_status(Status(StatusCode.ERROR, str(exc))) span.set_attribute("error.type", type(exc).__qualname__) @@ -157,7 +150,7 @@ With Anthropic, run each `tool_use` block with `run_tool(block.id, block.name, j ## TypeScript: set up the exporter and helper -There is no OpenTelemetry instrumentation that works with `openai` 7 or writes messages where Maple reads them, so a small helper records the turn, each tool call and each model call. +No OpenTelemetry instrumentation works for `openai` 7 in TypeScript, so this helper records the turn, each tool call and each model call. ```bash npm install @opentelemetry/api @opentelemetry/sdk-node @opentelemetry/exporter-trace-otlp-proto @@ -168,7 +161,6 @@ npm install @opentelemetry/api @opentelemetry/sdk-node @opentelemetry/exporter-t import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto" import { NodeSDK, tracing } from "@opentelemetry/sdk-node" -// The exporter reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS. export const spanProcessor = new tracing.BatchSpanProcessor(new OTLPTraceExporter()) export const sdk = new NodeSDK({ serviceName: "support-agent", spanProcessors: [spanProcessor] }) @@ -258,7 +250,7 @@ export function tracedChat(client: OpenAI, params: ChatParams, onText?: (delta: }) } -// OpenAI chat messages -> the {role, parts} shape Maple renders as a transcript. Text and tool calls only. +// Converts OpenAI messages to Maple's transcript format. Text and tool calls only. function toGenAiMessage(message: OpenAI.Chat.ChatCompletionMessageParam | OpenAI.Chat.ChatCompletionMessage) { if (message.role === "tool") { return { role: "tool", parts: [{ type: "tool_call_response", id: message.tool_call_id, response: message.content }] } @@ -337,11 +329,11 @@ For `@anthropic-ai/sdk` or `@google/genai`, copy `tracedChat` and map that SDK's ## Sub-agents, streaming and cost -A sub-agent is an agent run inside a tool call. Call it through `run_tool` and wrap its loop in `agent_span` with its own name and no conversation id. Maple draws one lane per agent name. +Run a sub-agent inside `run_tool` and wrap its loop in `agent_span` with its own name and no conversation id. When you stream OpenAI Chat Completions in Python, pass `stream_options={"include_usage": True}`, or the call shows 0 tokens. The TypeScript helper sets it for you. -Python sessions show as unpriced, because the instrumentations record no cost. The TypeScript helper records OpenRouter's `usage.cost` when you call through OpenRouter. +Python sessions show as unpriced. The TypeScript helper records cost only when you call through OpenRouter. ## Flush before a short-lived process exits @@ -349,7 +341,7 @@ In Python, a serverless handler or notebook should call `provider.force_flush()` ## Check that it works -Run a conversation of two or three messages with a tool call, flush, and open **Agent Sessions** in Maple. You should see one session with your conversation id, one turn per user message, and a transcript with the replies and tool calls. The framework column says **Unidentified**, which is expected for this setup. +Run a conversation of two or three messages with a tool call, flush, and open **Agent Sessions** in Maple. You should see one session with your conversation id, one turn per user message, and a transcript with the replies and tool calls. The framework is **Unidentified** for this setup. ## Troubleshooting @@ -362,7 +354,5 @@ Run a conversation of two or three messages with a tool call, flush, and open ** ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [Agent tracing guides](/docs/agent-tracing) - [OpenRouter](/docs/agent-tracing/openrouter), if your calls go through OpenRouter - [OpenTelemetry GenAI instrumentations for Python](https://github.com/open-telemetry/opentelemetry-python-genai) -- [OpenTelemetry GenAI semantic conventions](https://opentelemetry.io/docs/specs/semconv/gen-ai/) diff --git a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md index 18ec0f81e6..1531514546 100644 --- a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md @@ -7,13 +7,11 @@ navLabel: "Pydantic AI" icon: "pydantic" --- -Pydantic AI emits OpenTelemetry spans for every run, model call and tool call, with prompts, replies and token counts. You give it a `TracerProvider` that exports to Maple. The one thing to get right is the conversation id: unless you pass one, Pydantic AI generates a new id for every run, and each message becomes its own session. - -Tested with `pydantic-ai-slim` 2.51.0, the OpenTelemetry Python SDK 1.45.0, Logfire 5.1.1 and Python 3.10+. +Pydantic AI already emits OpenTelemetry spans for runs, model calls and tool calls. You export them to Maple and pass a conversation id on every run. Without the id, each message becomes its own session. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-pydantic-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-pydantic-ai) skill, which contains every step of this guide. +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-pydantic-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-pydantic-ai) skill and follows it. ```text Set up Maple agent tracing for Pydantic AI in this project. @@ -57,7 +55,6 @@ provider = TracerProvider( {"service.name": "support-agent", "deployment.environment.name": "production"} ) ) -# Reads OTEL_EXPORTER_OTLP_ENDPOINT and OTEL_EXPORTER_OTLP_HEADERS provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) @@ -70,13 +67,13 @@ Agent.instrument_all( ) ``` -Import `tracing` at the top of your entry point (`main.py`, the FastAPI app module, the worker). It must run before the first `agent.run()`. Agents created earlier are still covered. +Import `tracing` at the top of your entry point (`main.py`, the FastAPI app module, the worker), before the first `agent.run()`. If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Sentry or your own setup), add the `BatchSpanProcessor` to it instead of creating a second one, and call `Agent.instrument_all(InstrumentationSettings(include_content=True, include_binary_content=False))` without `tracer_provider`. ### If you already use Logfire -Logfire brings its own OpenTelemetry SDK and exporter, so skip the `pip install` above (Logfire 5.1 pins `opentelemetry-sdk` below 1.45 and the install won't resolve). Keep the three environment variables. Logfire exports to Maple whenever `OTEL_EXPORTER_OTLP_ENDPOINT` is set. +Skip the `pip install` above, because it conflicts with Logfire's OpenTelemetry pins. Keep the three environment variables. Logfire exports to Maple whenever `OTEL_EXPORTER_OTLP_ENDPOINT` is set. ```py import logfire @@ -89,7 +86,7 @@ Logfire scrubs tool arguments and results by default. If they arrive as `[Scrubb ## Pass the conversation id on every run -Maple groups traces into sessions by `gen_ai.conversation.id`. Pydantic AI takes it from the `conversation_id=` you pass, then from the last message of `message_history`, and otherwise generates a new UUID7. Pass the chat or thread id your app already has on every `run()`, `run_stream()` and `iter()`: +Pass the chat or thread id your app already has as `conversation_id=` on every `run()`, `run_stream()` and `iter()`: ```py from pydantic_ai import Agent @@ -115,7 +112,7 @@ async def stream_reply(chat_id: str, text: str, history: list): ### Sub-agents -When a tool calls another agent's `run()`, that run generates its own id unless you pass the caller's. Pass `conversation_id` and `usage` from the tool's context. Give every agent a `name=`, which Maple uses to draw one lane per agent. +When a tool runs another agent, pass `conversation_id` and `usage` from the tool's context. Give every agent a `name=` so each one gets its own lane. ```py from pydantic_ai import Agent, RunContext @@ -137,7 +134,7 @@ async def research_weather(ctx: RunContext[None], city: str) -> str: ## Flush short-lived processes -The SDK flushes on a normal interpreter exit. A Lambda that freezes after returning, a killed worker or a notebook kernel doesn't exit normally, so flush explicitly: +A Lambda, a killed worker or a notebook kernel doesn't exit normally, so flush explicitly: ```py import asyncio @@ -156,21 +153,19 @@ In a script or CLI, call `provider.shutdown()` at the end. With Logfire, use `lo ## Check that it works -Pydantic AI prints an `observability: off` banner on the first run when no instrumentation is set up. If you still see it, `tracing.py` didn't run before your first `agent.run()`. +If Pydantic AI prints an `observability: off` banner on the first run, `tracing.py` didn't run before your first `agent.run()`. -Run a conversation with two messages and a tool call, then open **Agent Sessions**. You should see one session with your conversation id and framework **Pydantic AI**, one turn per `run()`, a transcript with prompts, replies and tool calls, and token counts on every model call. Cost shows as unpriced, because Maple doesn't read the cost attribute Pydantic AI writes. +Run a conversation with two messages and a tool call, then open **Agent Sessions**. You should see one session with your conversation id and framework **Pydantic AI**, one turn per `run()`, a transcript with prompts, replies and tool calls, and token counts on every model call. Cost shows as unpriced. ## Troubleshooting - **Every message is its own session.** Pass `conversation_id=` on every `run()`, `run_stream()` and `iter()`. - **A multi-agent run is split into several turns or sessions.** Pass `conversation_id=ctx.conversation_id` to every nested `run()`. - **Tool arguments or results read `[Scrubbed due to ...]`.** Logfire's scrubbing matched a word like `session` or `auth`. Pass `scrubbing=logfire.ScrubbingOptions(callback=...)` that keeps `gen_ai.tool.call.arguments` and `gen_ai.tool.call.result`, or `scrubbing=False`. -- **A failed tool shows as successful.** The tool returned an error value. Raise `ToolFailed("...")` (Pydantic AI 2.16+) so the call is marked failed and the model still sees the message. +- **A failed tool shows as successful.** The tool returned an error value. Raise `ToolFailed("...")` so the call is marked failed and the model still sees the message. - **Spans show up twice.** Another instrumentor (Logfire's `instrument_openai()`, OpenInference, OpenLLMetry) also traces the model client. Remove it and keep Pydantic AI's. ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [All agent tracing guides](/docs/agent-tracing) -- [Pydantic AI: debugging and monitoring with OpenTelemetry](https://pydantic.dev/docs/ai/integrations/logfire/) - [Instrument a Python application](/docs/guides/instrumentation-python) diff --git a/apps/landing/src/content/docs/agent-tracing/smolagents.md b/apps/landing/src/content/docs/agent-tracing/smolagents.md index c7d239a63c..17d15344f1 100644 --- a/apps/landing/src/content/docs/agent-tracing/smolagents.md +++ b/apps/landing/src/content/docs/agent-tracing/smolagents.md @@ -7,13 +7,11 @@ navLabel: "smolagents" icon: "huggingface" --- -smolagents has no tracing of its own. Its spans come from OpenInference's `openinference-instrumentation-smolagents`, which records every run, step, model call and tool call. Two defaults need changing for Maple: turn on the instrumentor's GenAI attributes, or the session page shows no transcript, and wrap each run in `using_session(...)`, or every message becomes its own session. - -Tested with smolagents 1.26 and `openinference-instrumentation-smolagents` 0.1.40 on Python 3.10+, with `ToolCallingAgent` and `CodeAgent`. +smolagents is traced with OpenInference's `openinference-instrumentation-smolagents`. Turn on its GenAI attributes and wrap each run in `using_session(...)`. Without them, the session page shows no transcript and every message becomes its own session. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-smolagents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-smolagents) skill, which contains every step of this guide. +Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-smolagents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-smolagents) skill and follows it. ```text Set up Maple agent tracing for smolagents in this project. @@ -32,7 +30,7 @@ pip install "smolagents[openai]>=1.26" "openinference-instrumentation-smolagents "opentelemetry-sdk>=1.45" "opentelemetry-exporter-otlp-proto-http>=1.45" ``` -Use `[litellm]` instead of `[openai]` if you run `LiteLLMModel`. Skip the `smolagents[telemetry]` extra, which also installs the Arize Phoenix server. +Use `[litellm]` instead of `[openai]` if you run `LiteLLMModel`. Skip the `smolagents[telemetry]` extra, which installs the Arize Phoenix server. Point the exporter at Maple. For an EU organization, use `https://ingest.eu.maple.dev`. Set the base URL only; the exporter appends `/v1/traces`. @@ -58,28 +56,24 @@ from opentelemetry.sdk.trace.export import BatchSpanProcessor class SmolagentsForMaple(SpanProcessor): - """Fixes what the smolagents instrumentor gets wrong for Maple: agent names, run token totals, tool names and arguments.""" + """Fixes agent names, run token totals, tool names and tool arguments for Maple.""" def on_start(self, span, parent_context=None): if span.instrumentation_scope.name != "openinference.instrumentation.smolagents": return attrs = span.attributes if span.name.endswith(".run"): - # "weather_worker.run" -> gen_ai.agent.name "weather_worker", so sub-agents get lanes span.set_attribute("gen_ai.agent.name", span.name.removesuffix(".run")) - # The run's token totals repeat its model calls (with reset=False, every earlier turn's too) span.set_attribute("gen_ai.usage.input_tokens", 0) span.set_attribute("gen_ai.usage.output_tokens", 0) elif "tool.name" in attrs: - # Every @tool span is named "SimpleTool"; name it after the tool instead span.update_name(f"execute_tool {attrs['tool.name']}") - # The GenAI dual-write copies the tool's input schema here; record the call's arguments if attrs.get("input.value", "").startswith("{"): call = json.loads(attrs["input.value"]) span.set_attribute("gen_ai.tool.call.arguments", json.dumps(call["kwargs"] or call["args"])) -provider = TracerProvider() # reads OTEL_SERVICE_NAME and OTEL_RESOURCE_ATTRIBUTES +provider = TracerProvider() provider.add_span_processor(SmolagentsForMaple()) provider.add_span_processor(BatchSpanProcessor(OTLPSpanExporter())) trace.set_tracer_provider(provider) @@ -90,13 +84,13 @@ SmolagentsInstrumentor().instrument( ) ``` -`enable_genai_semconv=True` makes the instrumentor write the `gen_ai.*` attributes Maple reads for the transcript, tokens and tools. `SmolagentsForMaple` names each agent so managed agents get their own lane, stops run spans from counting tokens a second time, and fixes tool span names and arguments. Keep it as is. +`enable_genai_semconv=True` is required for the transcript, tokens and tool calls. Copy `SmolagentsForMaple` as is. It gives each agent its own lane, stops tokens from being counted twice, and fixes tool names and arguments. If your app already has a `TracerProvider` (from `opentelemetry-instrument`, Logfire or another library), add `SmolagentsForMaple()` and the exporter to that provider and pass it to `instrument()` instead of creating a second one. ## Wrap each run in the conversation id -smolagents has no conversation id, and every `agent.run()` starts a new trace. Maple groups traces by `session.id`, which the instrumentor sets inside OpenInference's `using_session` block: +Wrap every `agent.run()` in OpenInference's `using_session` with the id your app already stores the chat under: ```py from openinference.instrumentation import using_session @@ -117,7 +111,7 @@ def handle_message(conversation_id: str, text: str) -> str: return str(agent.run(text, reset=False)) ``` -Use the id your app already stores the chat under. It must stay the same across the conversation and differ between conversations. Setting `gen_ai.conversation.id` yourself won't group anything, because Maple reads `session.id` for smolagents. +The id must stay the same across the conversation and differ between conversations. Setting `gen_ai.conversation.id` yourself doesn't group smolagents runs. Keep one agent object per conversation, as above. A single shared agent with `reset=False` mixes every user's memory into one conversation. @@ -125,7 +119,7 @@ Give every agent, including managed agents, a `name`. Unnamed agents share one l ## Flush before the process exits -The SDK flushes on a normal interpreter exit. Serverless functions, killed workers and notebooks don't exit normally, so flush after each run: +Serverless functions, killed workers and notebooks don't exit normally, so flush after each run: ```py from tracing import provider @@ -133,14 +127,14 @@ from tracing import provider try: handle_message("conv-42", "What's the weather in Berlin?") finally: - provider.force_flush() # serverless: before returning; notebooks: after each run + provider.force_flush() ``` ## Check that it works -Run two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions**. You should see one session with framework **smolagents**, one turn per `agent.run()`, a transcript, `OpenAIModel.generate` model calls with token counts, and `execute_tool ` tool calls. Each run also ends with an `execute_tool final_answer` call. +Run two or three messages through `handle_message` with the same conversation id, including one that uses a tool, then open **Agent Sessions**. You should see one session with framework **smolagents**, one turn per `agent.run()`, a transcript, `OpenAIModel.generate` model calls with token counts, and `execute_tool ` tool calls. Each run ends with an `execute_tool final_answer` call. -Turns and the session title read `New task:`, because smolagents prefixes every task with that line. Cost shows as unpriced, since smolagents records none. +Turns and the session title start with `New task:`. Cost shows as unpriced. ## Troubleshooting @@ -152,8 +146,5 @@ Turns and the session title read `New task:`, because smolagents prefixes every ## Related -- [Agent Sessions overview](/docs/agent-sessions/overview): what Maple builds from these spans. -- [Trace your AI agent](/docs/agent-tracing): guides for every other framework. -- [Inspecting runs with OpenTelemetry](https://huggingface.co/docs/smolagents/tutorials/inspect_runs): the smolagents docs page on tracing. -- [openinference-instrumentation-smolagents](https://github.com/Arize-ai/openinference/tree/main/python/instrumentation/openinference-instrumentation-smolagents): the instrumentor's source. -- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter): if your models go through either gateway. +- [Agent Sessions overview](/docs/agent-sessions/overview) +- [LiteLLM](/docs/agent-tracing/litellm) and [OpenRouter](/docs/agent-tracing/openrouter), if your models go through either gateway diff --git a/apps/landing/src/content/docs/agent-tracing/spring-ai.md b/apps/landing/src/content/docs/agent-tracing/spring-ai.md index b5016e7895..a3a24d6a8b 100644 --- a/apps/landing/src/content/docs/agent-tracing/spring-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/spring-ai.md @@ -7,7 +7,7 @@ navLabel: "Spring AI" icon: "spring" --- -Spring AI already emits a `spring_ai chat_client` span per `ChatClient` call, a `chat ` span per model call and an `execute_tool ` span per tool call, with model and token counts. You add Spring Boot's OpenTelemetry starter, set sampling to 100%, and add one configuration class that writes the transcript, tool names and failed tools where Maple reads them. The thing to get right is passing the conversation id on every call. +Spring AI already emits spans for `ChatClient`, model and tool calls, with token counts. You add Spring Boot's OpenTelemetry starter, set sampling to 100%, add one configuration class for the transcript, tool names and failed tools, and pass the conversation id on every call. Tested with Spring AI 2.0.1 on Spring Boot 4.1.1 and Java 21. Spring AI 1.1 on Boot 3.5 works too, with different dependencies and property names listed in the [skill](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai). @@ -36,8 +36,6 @@ Next to the `spring-ai-bom` import (2.0.1) and your model starter, such as `spri
``` -It brings the Micrometer tracing bridge, the OpenTelemetry SDK and the OTLP exporter. Boot 4 doesn't need Actuator for tracing. - ## Export to Maple Add to `application.properties`: @@ -60,11 +58,11 @@ maple.ai.capture-content=true For an EU organization, use `https://ingest.eu.maple.dev`. This property takes the full URL, so keep `/v1/traces` on the end. To skip metrics, replace the two metrics lines with `management.otlp.metrics.export.enabled=false`. -Keep `management.tracing.sampling.probability=1.0`. Boot's default is `0.1`, which drops nine turns out of ten without any error. +Keep `management.tracing.sampling.probability=1.0`. Boot's default of `0.1` silently drops nine turns out of ten. ## Add the attributes Maple reads -Add this class in a package your application scans. Boot registers both beans by itself: +Add this class in a package your application scans: ```java package com.example.agent; @@ -92,7 +90,6 @@ import org.springframework.beans.factory.annotation.Value; import org.springframework.context.annotation.Bean; import org.springframework.context.annotation.Configuration; -/** Adds the gen_ai.* attributes Maple reads on top of Spring AI's own observations. */ @Configuration(proxyBeanMethods = false) public class MapleAiObservationConfig { @@ -103,15 +100,13 @@ public class MapleAiObservationConfig { ObservationFilter mapleGenAiAttributes(@Value("${maple.ai.capture-content:false}") boolean captureContent) { return context -> { if (context instanceof ChatClientObservationContext client) { - // One ChatClient call is one agent turn; Spring AI labels it "framework". client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.operation.name", "invoke_agent")); if (client.getRequest().context().get(AGENT_NAME) instanceof String agent) { client.addLowCardinalityKeyValue(KeyValue.of("gen_ai.agent.name", agent)); } } else if (context instanceof AdvisorObservationContext advisor) { - // Advisor span names ("tool _calling ", "message_chat_memory") would be - // counted as extra tool and LLM calls. A neutral name keeps them as plumbing. + // Renamed so Maple doesn't count advisor spans as tool and model calls advisor.setContextualName("spring_ai advisor"); } else if (context instanceof ToolCallingObservationContext tool) { @@ -174,15 +169,13 @@ public class MapleAiObservationConfig { } ``` -The `ObservationFilter` marks each `ChatClient` call as an agent turn, copies tool names and call ids to the `gen_ai.tool.*` keys, and, with `maple.ai.capture-content=true`, writes the messages and tool arguments and results to the spans. Spring AI's own `log-prompt` and `log-completion` settings only write to the application log. The filter also renames the advisor spans, which Maple would otherwise count as extra model and tool calls. - -The `ToolExecutionExceptionProcessor` marks a tool span as failed when the tool throws, then hands the error to the model as before. If your app already defines one, add the `error()` call to it instead of adding a second bean. +If your app already defines a `ToolExecutionExceptionProcessor`, add the `error()` call to it instead of adding a second bean. -Set `maple.ai.capture-content=false` to keep message and tool content out of Maple. Sessions, tokens, tool names and failures still show up, with an empty transcript. +The transcript comes from `maple.ai.capture-content=true`. Spring AI's own `log-prompt` and `log-completion` settings only write to the application log. Set it to `false` to keep message and tool content out of Maple; everything else still shows up, with an empty transcript. ## Group turns into one session -Maple joins the traces of one conversation by the `spring.ai.chat.client.conversation.id` attribute. Spring AI sets it from the `ChatMemory.CONVERSATION_ID` advisor parameter, so pass that parameter on every request: +Pass the `ChatMemory.CONVERSATION_ID` advisor parameter on every request. Spring AI writes it to the `spring.ai.chat.client.conversation.id` attribute, which Maple groups sessions by: ```java import org.springframework.ai.chat.client.ChatClient; @@ -215,11 +208,11 @@ public class ChatService { } ``` -Use the conversation id your app already stores. Set it on the request, as above, and never with `defaultAdvisors` on the builder, which puts every user in one session. The parameter works without a chat memory advisor. Sub-agent calls inside tools don't need it, because they run in the same trace. +Use the conversation id your app already stores. Set it on the request, as above, and never with `defaultAdvisors` on the builder, which puts every user in one session. The parameter works without a chat memory advisor. Sub-agent calls inside tools don't need it. ## Exit command-line apps explicitly -A web app needs nothing extra: Boot flushes pending spans on shutdown. In a `CommandLineRunner` app, exit explicitly, or the OpenAI client's threads keep the JVM (and the unsent spans) waiting about 60 seconds: +A web app needs nothing extra: Boot flushes pending spans on shutdown. In a `CommandLineRunner` app, exit explicitly, or the OpenAI client's threads hold the JVM and the unsent spans for about 60 seconds: ```java public static void main(String[] args) { @@ -231,7 +224,7 @@ On serverless platforms, inject `SdkTracerProvider` and call `tracerProvider.for ## Check that it works -Send two or three messages with the same conversation id, including one that calls a tool. After about a minute, **Agent Sessions** in Maple shows one session for that id with framework **Spring AI**, one turn per `ChatClient` call, the transcript, and tokens on every model call. Cost shows as unpriced, because Spring AI doesn't emit one. +Send two or three messages with the same conversation id, including one that calls a tool. After about a minute, **Agent Sessions** in Maple shows one session for that id with framework **Spring AI**, one turn per `ChatClient` call, the transcript, and tokens on every model call. Cost shows as unpriced, because Spring AI doesn't report it. ## Troubleshooting @@ -244,8 +237,5 @@ Send two or three messages with the same conversation id, including one that cal ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [Agent tracing guides](/docs/agent-tracing) -- [Any language: the OpenTelemetry GenAI conventions](/docs/agent-tracing/opentelemetry), also for LangChain4j - [Spring AI observability reference](https://docs.spring.io/spring-ai/reference/observability/index.html) -- [Spring AI tool calling](https://docs.spring.io/spring-ai/reference/api/tools.html) - [Spring Boot tracing reference](https://docs.spring.io/spring-boot/reference/actuator/tracing.html) diff --git a/apps/landing/src/content/docs/agent-tracing/strands.md b/apps/landing/src/content/docs/agent-tracing/strands.md index 83c81b3c18..757937a92d 100644 --- a/apps/landing/src/content/docs/agent-tracing/strands.md +++ b/apps/landing/src/content/docs/agent-tracing/strands.md @@ -7,15 +7,13 @@ navLabel: "Strands Agents" icon: "strands" --- -Strands Agents ships its own OpenTelemetry tracer: every `agent(...)` call becomes one trace with `invoke_agent`, `chat` and `execute_tool` spans. Maple recognizes them without an extra instrumentation library. +Strands Agents has a built-in OpenTelemetry tracer, and Maple reads its spans without an extra instrumentation library. You set one environment variable so the transcript is recorded, and pass your conversation id as `session.id` on each agent. -By default Strands writes prompts and replies as span events, which Maple doesn't read, so you set one environment variable to move them onto span attributes. You also pass your conversation id as `session.id` on each agent. - -Tested with `strands-agents` 1.57.1 (1.54 or newer required) and the TypeScript SDK `@strands-agents/sdk` 1.19. +You need `strands-agents` 1.54 or newer, or the TypeScript SDK `@strands-agents/sdk` 1.19 or newer. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-strands](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-strands) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-strands](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-strands) skill and follows it. ```text Set up Maple agent tracing for Strands Agents in this project. @@ -29,7 +27,7 @@ Your ingest key is in **Settings → Ingestion**. ## Install and configure the exporter -The `otel` extra adds the OTLP/HTTP exporter. Add the extra for your model provider too (`openai`, `anthropic`, `litellm`; Bedrock needs none). +Replace `openai` with your model provider's extra (`anthropic`, `litellm`; Bedrock needs none). ```bash pip install 'strands-agents[otel,openai]>=1.57' @@ -48,7 +46,7 @@ export OTEL_SEMCONV_STABILITY_OPT_IN="gen_ai_latest_experimental,gen_ai_span_att For an EU organization, use `https://ingest.eu.maple.dev`. The exporter appends `/v1/traces` itself. -`OTEL_SEMCONV_STABILITY_OPT_IN` is the important one. Its three values switch to the current GenAI message format, write messages as span attributes (without this the transcript is empty), and make each `invoke_agent` span report only that call's tokens. Strands reads it once, when the first `Agent` is created, so set it in the environment rather than in code. +Without `OTEL_SEMCONV_STABILITY_OPT_IN` the transcript is empty and token counts are too high. Strands reads it when the first `Agent` is created, so set it in the environment rather than in code. To redact message content, append `gen_ai_unredacted_attributes=` with a `;`-separated allowlist of attributes to keep; everything else becomes `[REDACTED]`. @@ -61,11 +59,11 @@ from strands.telemetry import StrandsTelemetry telemetry = StrandsTelemetry().setup_otlp_exporter() ``` -If your app already sets up a global `TracerProvider` (your web framework, `opentelemetry-instrument`, or the ADOT distro on AgentCore), skip `StrandsTelemetry()`. Add a `BatchSpanProcessor(OTLPSpanExporter(...))` pointing at Maple to that provider instead. +If your app already sets up a global `TracerProvider` (for example `opentelemetry-instrument` or the ADOT distro on AgentCore), skip `StrandsTelemetry()` and add a `BatchSpanProcessor(OTLPSpanExporter(...))` pointing at Maple to that provider. ## Pass the conversation id as session.id -Strands copies `trace_attributes` onto every span of the agent, and Maple groups Strands traces by `session.id`: +Pass the conversation id in `trace_attributes`: ```py from strands import Agent @@ -90,13 +88,13 @@ def handle_message(conversation_id: str, text: str) -> str: The session manager restores history but doesn't put its `session_id` on the spans, so you need both arguments. Use the conversation id your app already has, never a fresh UUID per request. -Create the agent per request, as above. A shared module-level `Agent` would carry one user's id into everyone's traces. Give every agent a `name`: unnamed agents are all called `Strands Agents`, which merges sub-agents into one lane. +Create the agent per request, as above. A shared module-level `Agent` would carry one user's id into everyone's traces. Give every agent a `name`, or sub-agents merge into one lane. -With `agent.as_tool()` sub-agents, only the orchestrator needs `session.id`, because the sub-agents run inside its trace. For a `Swarm`, pass `trace_attributes={"session.id": conversation_id}` to the `Swarm`. For a `Graph`, set `graph.trace_attributes = {"session.id": conversation_id}` after `builder.build()`, which doesn't forward them. +With `agent.as_tool()` sub-agents, only the orchestrator needs `session.id`. For a `Swarm`, pass `trace_attributes={"session.id": conversation_id}` to the `Swarm`. For a `Graph`, set `graph.trace_attributes = {"session.id": conversation_id}` after `builder.build()`. ## Flush in scripts and jobs -Spans are exported in batches every few seconds. A long-running server needs nothing extra. Scripts, CLIs, notebooks and jobs lose their last spans unless they flush: +A long-running server needs nothing extra. Scripts, notebooks and jobs lose their last spans unless they flush: ```py from telemetry import telemetry @@ -108,11 +106,11 @@ finally: telemetry.tracer_provider.shutdown() ``` -On AWS Lambda, call `telemetry.tracer_provider.force_flush()` at the end of each invocation and don't call `shutdown()`, because the warm container reuses the provider. +On AWS Lambda, call `telemetry.tracer_provider.force_flush()` at the end of each invocation and don't call `shutdown()`. ## TypeScript SDK -The TypeScript SDK emits the same spans. Its OpenTelemetry packages are optional peer dependencies, so install them explicitly: +Install the SDK with its OpenTelemetry packages: ```bash npm install @strands-agents/sdk @opentelemetry/api @opentelemetry/sdk-trace-base @opentelemetry/sdk-trace-node @opentelemetry/resources @opentelemetry/exporter-trace-otlp-http @opentelemetry/sdk-metrics @opentelemetry/exporter-metrics-otlp-http @@ -153,13 +151,13 @@ try { } ``` -Set both keys in `traceAttributes`. With a custom service name, Maple can't tell the spans come from Strands and groups them by `gen_ai.conversation.id`, showing the framework as "Unidentified". Create the agent per request here too: a reused TypeScript agent reports its running token total on every turn. +Set both keys in `traceAttributes`. With a custom service name Maple groups these spans by `gen_ai.conversation.id` and shows the framework as "Unidentified". Create the agent per request here too, since a reused TypeScript agent reports its running token total on every turn. ## Check that it works Run a conversation of two or three messages, including one that calls a tool, then open **Agent Sessions** in Maple. You should see one session per conversation id with framework **Strands Agents**, one turn per `agent(...)` call, and a transcript with the user messages, replies and tool calls. -Cost shows as unpriced, because Strands doesn't report it. The sessions list currently shows about twice the real token count for Strands; the session's own page has the correct total. +Cost shows as unpriced because Strands doesn't report it. The sessions list currently shows about twice the real token count; the session's own page has the correct total. ## Troubleshooting @@ -172,6 +170,4 @@ Cost shows as unpriced, because Strands doesn't report it. The sessions list cur ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [Agent tracing guides](/docs/agent-tracing) - [Strands Agents traces documentation](https://strandsagents.com/docs/user-guide/observability-evaluation/traces/) -- [Strands telemetry tracer API reference](https://strandsagents.com/docs/api/python/strands.telemetry.tracer/) diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md index 42e29e30aa..0077731e67 100644 --- a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -7,15 +7,15 @@ navLabel: "Vercel AI SDK" icon: "vercel" --- -The Vercel AI SDK emits OpenTelemetry GenAI spans for every `generateText`, `streamText` and `ToolLoopAgent` call, with the prompts, replies, tool calls and token counts. Maple reads them without an extra instrumentation package. +The Vercel AI SDK emits OpenTelemetry spans for every `generateText`, `streamText` and `ToolLoopAgent` call, and Maple reads them without an extra instrumentation package. -You have to get two things right. In AI SDK 7 nothing is traced until you call `registerTelemetry()` at startup, and the SDK has no conversation id, so you pass one on every call or each message becomes its own session. +In AI SDK 7 nothing is traced until you call `registerTelemetry()` at startup. You also pass a conversation id on every call, or each message becomes its own session. -Tested with `ai` 7.0.118, `@ai-sdk/otel` 1.0.118, OpenTelemetry JS 0.222.0 and `@vercel/otel` 2.1.3 on Node.js 26 and Bun 1.3. You need `ai` 7.0.106 or newer and Node.js 22 or newer. On AI SDK 5 or 6, run `npx @ai-sdk/codemod v7` first; the skill has a fallback setup if you can't upgrade. +You need `ai` 7.0.106 or newer and Node.js 22 or newer (Bun works too). On AI SDK 5 or 6, run `npx @ai-sdk/codemod v7` first. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-vercel-ai-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-vercel-ai-sdk) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-vercel-ai-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-vercel-ai-sdk) skill and follows it. ```text Set up Maple agent tracing for the Vercel AI SDK in this project. @@ -126,7 +126,7 @@ Return AI SDK streams as the response (`toUIMessageStreamResponse()` or `createA ## Pass the conversation id on every call -Maple groups traces into sessions by `gen_ai.conversation.id`. The `enrichSpan` callback above copies it from `runtimeContext.conversationId`, which only reaches telemetry if you also list it in `includeRuntimeContext`: +The `enrichSpan` callback above reads `runtimeContext.conversationId`, which only reaches telemetry if you also list it in `includeRuntimeContext`: ```ts import { streamText } from "ai" @@ -179,21 +179,21 @@ export async function POST(req: Request) { } ``` -Sub-agents called from a tool's `execute` run inside the caller's trace, so they don't need the id. They do need their own `functionId` to get their own lane. +Sub-agents called from a tool's `execute` don't need the id, only their own `functionId`. -Prompts and replies are recorded by default. To keep them out of Maple for a call or agent, set `recordInputs: false` and `recordOutputs: false` in its `telemetry`. +To keep a call's prompts and replies out of Maple, set `recordInputs: false` and `recordOutputs: false` in its `telemetry`. ## Flush in scripts and serverless functions -The SDK exports spans in batches every few seconds, so a short-lived process can exit first. In a script, call `await sdk.shutdown()` in a `finally` block before exiting. In a serverless handler that is reused between invocations, call `await spanProcessor.forceFlush()` in a `finally` instead (see [the variant above](#serverless-or-an-app-that-already-uses-opentelemetry)), so the SDK keeps running. +A short-lived process can exit before its spans are exported. In a script, call `await sdk.shutdown()` in a `finally` block before exiting. In a serverless handler, call `await spanProcessor.forceFlush()` in a `finally` instead (see [the variant above](#serverless-or-an-app-that-already-uses-opentelemetry)). -Read streams to the end (`await result.consumeStream()`) before flushing: a stream's spans only end when it has been read. On Vercel, `@vercel/otel` flushes after each request, but queue consumers and cron jobs need their own `forceFlush()`. +Read streams to the end (`await result.consumeStream()`) before flushing, since a stream's spans only end when it has been read. On Vercel, `@vercel/otel` flushes after each request, but queue consumers and cron jobs need their own `forceFlush()`. ## Check that it works Run a conversation with two messages and a tool call, then open **Agent Sessions** in Maple. You should see one session named after your conversation id with framework **Vercel AI SDK**, one turn per call, and a transcript with the prompts, replies and tool calls. -Two things look off but are expected. Cost shows as unpriced, because the AI SDK doesn't report it. The session's token total is currently twice what the model calls used; the per-model breakdown on the session page is correct. +Cost shows as unpriced because the AI SDK doesn't report it. The session's token total is currently doubled; the per-model breakdown on the session page is correct. ## Troubleshooting @@ -206,8 +206,4 @@ Two things look off but are expected. Cost shows as unpriced, because the AI SDK ## Related - [Agent Sessions overview](/docs/agent-sessions/overview) -- [All agent tracing guides](/docs/agent-tracing) - [AI SDK telemetry](https://ai-sdk.dev/docs/ai-sdk-core/telemetry) -- [Next.js instrumentation](/docs/guides/instrumentation-nextjs) -- [Node.js instrumentation](/docs/guides/instrumentation-nodejs) -- [OpenRouter Broadcast](/docs/agent-tracing/openrouter) From 1b432a7c4dcc9979f7952d44037286dab6908fa2 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 14:44:42 +0200 Subject: [PATCH 10/39] docs(agent-tracing): one quick-setup wording across guides, with the EU region hint --- apps/landing/src/content/docs/agent-tracing.mdx | 4 ++-- apps/landing/src/content/docs/agent-tracing/agno.md | 4 ++-- .../src/content/docs/agent-tracing/claude-agent-sdk.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/crewai.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/dspy.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/google-adk.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/haystack.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/langchain.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/litellm.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/llamaindex.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/mastra.md | 2 +- .../content/docs/agent-tracing/microsoft-agent-framework.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/openai-agents.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/openrouter.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/opentelemetry.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/provider-sdks.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/pydantic-ai.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/smolagents.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/spring-ai.md | 4 ++-- apps/landing/src/content/docs/agent-tracing/strands.md | 2 +- apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md | 2 +- 21 files changed, 39 insertions(+), 39 deletions(-) diff --git a/apps/landing/src/content/docs/agent-tracing.mdx b/apps/landing/src/content/docs/agent-tracing.mdx index 221da1ac6f..891b7bccd8 100644 --- a/apps/landing/src/content/docs/agent-tracing.mdx +++ b/apps/landing/src/content/docs/agent-tracing.mdx @@ -13,7 +13,7 @@ Each guide sets up one framework so every conversation shows up in Maple as one ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) skill, which detects your framework and installs the matching skill. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing) skill, which detects your framework and installs the matching skill. ```text Set up Maple agent tracing in this project. @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Choose your framework diff --git a/apps/landing/src/content/docs/agent-tracing/agno.md b/apps/landing/src/content/docs/agent-tracing/agno.md index 1f1f08662c..5aebadb513 100644 --- a/apps/landing/src/content/docs/agent-tracing/agno.md +++ b/apps/landing/src/content/docs/agent-tracing/agno.md @@ -15,7 +15,7 @@ Tested with Agno 3.0.11 and `openinference-instrumentation-agno` 1.0.10 on Pytho ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-agno](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-agno) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-agno](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-agno) skill and follows it. ```text Set up Maple agent tracing for Agno in this project. @@ -25,7 +25,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is under **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install and point the exporter at Maple diff --git a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md index 9478feefc3..ddb9ea4352 100644 --- a/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/claude-agent-sdk.md @@ -15,7 +15,7 @@ Tested with `@anthropic-ai/claude-agent-sdk` 0.3.283 (TypeScript), `claude-agent ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-claude-agent-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-claude-agent-sdk) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-claude-agent-sdk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-claude-agent-sdk) skill and follows it. ```text Set up Maple agent tracing for the Claude Agent SDK in this project. @@ -25,7 +25,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is under **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Pass the telemetry variables to the CLI diff --git a/apps/landing/src/content/docs/agent-tracing/crewai.md b/apps/landing/src/content/docs/agent-tracing/crewai.md index 3a3383a852..120729145f 100644 --- a/apps/landing/src/content/docs/agent-tracing/crewai.md +++ b/apps/landing/src/content/docs/agent-tracing/crewai.md @@ -11,7 +11,7 @@ CrewAI needs two OpenInference instrumentors: `openinference-instrumentation-cre ## Quick setup with a coding agent -Paste this prompt into Claude Code, Codex, Cursor or another coding agent. It installs the [maple-agent-tracing-crewai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-crewai) skill and follows it. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-crewai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-crewai) skill and follows it. ```text Set up Maple agent tracing for CrewAI in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the instrumentors diff --git a/apps/landing/src/content/docs/agent-tracing/dspy.md b/apps/landing/src/content/docs/agent-tracing/dspy.md index d027186a55..404a01a0c8 100644 --- a/apps/landing/src/content/docs/agent-tracing/dspy.md +++ b/apps/landing/src/content/docs/agent-tracing/dspy.md @@ -13,7 +13,7 @@ Tested with DSPy 3.4 and `openinference-instrumentation-dspy` 0.1.45 on Python 3 ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-dspy](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-dspy) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-dspy](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-dspy) skill and follows it. ```text Set up Maple agent tracing for DSPy in this project. @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is under **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install and point the exporter at Maple diff --git a/apps/landing/src/content/docs/agent-tracing/google-adk.md b/apps/landing/src/content/docs/agent-tracing/google-adk.md index 394fe7e220..654a735b4f 100644 --- a/apps/landing/src/content/docs/agent-tracing/google-adk.md +++ b/apps/landing/src/content/docs/agent-tracing/google-adk.md @@ -13,7 +13,7 @@ This guide covers ADK for Python. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-google-adk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-google-adk) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-google-adk](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-google-adk) skill and follows it. ```text Set up Maple agent tracing for Google ADK in this project. @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is under **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install ADK and the OTLP exporter diff --git a/apps/landing/src/content/docs/agent-tracing/haystack.md b/apps/landing/src/content/docs/agent-tracing/haystack.md index cefe1d9ab4..634989baf9 100644 --- a/apps/landing/src/content/docs/agent-tracing/haystack.md +++ b/apps/landing/src/content/docs/agent-tracing/haystack.md @@ -13,7 +13,7 @@ Tested with `haystack-ai` 3.2 and `opentelemetry-haystack` 1.0 on Python 3.10 or ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-haystack](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-haystack) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-haystack](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-haystack) skill and follows it. ```text Set up Maple agent tracing for Haystack in this project. @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is under **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install and add the Maple tracer diff --git a/apps/landing/src/content/docs/agent-tracing/langchain.md b/apps/landing/src/content/docs/agent-tracing/langchain.md index 18a0bc7e5f..21570205d7 100644 --- a/apps/landing/src/content/docs/agent-tracing/langchain.md +++ b/apps/landing/src/content/docs/agent-tracing/langchain.md @@ -13,7 +13,7 @@ This guide covers Python 3.10 or later. LangChain.js isn't covered yet. ## Quick setup with a coding agent -Paste this prompt into Claude Code, Codex, Cursor or another coding agent. It installs the [maple-agent-tracing-langchain](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-langchain) skill and follows it. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-langchain](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-langchain) skill and follows it. ```text Set up Maple agent tracing for LangChain & LangGraph in this project. @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the instrumentor diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md index 716a61f814..85a57a8984 100644 --- a/apps/landing/src/content/docs/agent-tracing/litellm.md +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -11,7 +11,7 @@ LiteLLM traces each model call. Your code adds the agent and tool spans and pass ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-litellm](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-litellm) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-litellm](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-litellm) skill and follows it. ```text Set up Maple agent tracing for LiteLLM in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Trace in your app or at the proxy diff --git a/apps/landing/src/content/docs/agent-tracing/llamaindex.md b/apps/landing/src/content/docs/agent-tracing/llamaindex.md index ba9cd6c95b..6d3d6c1b70 100644 --- a/apps/landing/src/content/docs/agent-tracing/llamaindex.md +++ b/apps/landing/src/content/docs/agent-tracing/llamaindex.md @@ -11,7 +11,7 @@ OpenInference's `openinference-instrumentation-llama-index` sends LlamaIndex age ## Quick setup with a coding agent -Paste this prompt into Claude Code, Codex, Cursor or another coding agent. It installs the [maple-agent-tracing-llamaindex](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-llamaindex) skill and follows it. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-llamaindex](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-llamaindex) skill and follows it. ```text Set up Maple agent tracing for LlamaIndex in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the instrumentor diff --git a/apps/landing/src/content/docs/agent-tracing/mastra.md b/apps/landing/src/content/docs/agent-tracing/mastra.md index 50c2d9a7a8..4c2ed81846 100644 --- a/apps/landing/src/content/docs/agent-tracing/mastra.md +++ b/apps/landing/src/content/docs/agent-tracing/mastra.md @@ -25,7 +25,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the observability packages diff --git a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md index 22746f9272..46e9422289 100644 --- a/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md +++ b/apps/landing/src/content/docs/agent-tracing/microsoft-agent-framework.md @@ -11,7 +11,7 @@ Microsoft Agent Framework (MAF) emits OpenTelemetry spans in Python and .NET. Yo ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-microsoft-agent-framework](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-microsoft-agent-framework) skill. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-microsoft-agent-framework](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-microsoft-agent-framework) skill and follows it. ```text Set up Maple agent tracing for Microsoft Agent Framework in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Export traces from Python diff --git a/apps/landing/src/content/docs/agent-tracing/openai-agents.md b/apps/landing/src/content/docs/agent-tracing/openai-agents.md index b897ddcf6b..827d267ade 100644 --- a/apps/landing/src/content/docs/agent-tracing/openai-agents.md +++ b/apps/landing/src/content/docs/agent-tracing/openai-agents.md @@ -11,7 +11,7 @@ OpenInference's `openinference-instrumentation-openai-agents` exports OpenAI Age ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-openai-agents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-openai-agents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openai-agents) skill and follows it. ```text Set up Maple agent tracing for the OpenAI Agents SDK in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is under **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the bridge and point it at Maple diff --git a/apps/landing/src/content/docs/agent-tracing/openrouter.md b/apps/landing/src/content/docs/agent-tracing/openrouter.md index 1887e384d5..06ff6f6ec1 100644 --- a/apps/landing/src/content/docs/agent-tracing/openrouter.md +++ b/apps/landing/src/content/docs/agent-tracing/openrouter.md @@ -11,7 +11,7 @@ OpenRouter Broadcast sends a trace of every request on your OpenRouter account t ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-openrouter](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openrouter) skill and follows it. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-openrouter](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-openrouter) skill and follows it. ```text Set up Maple agent tracing for OpenRouter in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. The agent prints the values for the OpenRouter dashboard, which you enter yourself as described in the next section. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. The agent prints the values for the OpenRouter dashboard, which you enter yourself as described in the next section. ## Point Broadcast at Maple diff --git a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md index a8331db33e..678dab9cc8 100644 --- a/apps/landing/src/content/docs/agent-tracing/opentelemetry.md +++ b/apps/landing/src/content/docs/agent-tracing/opentelemetry.md @@ -13,7 +13,7 @@ Tested with the OpenTelemetry JS SDK 2.11 on Node.js 26 and the Python SDK 1.45 ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-opentelemetry](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-opentelemetry) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-opentelemetry](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-opentelemetry) skill and follows it. ```text Set up Maple agent tracing for my hand-rolled agent in this project, using the OpenTelemetry GenAI conventions. @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## The spans and attributes Maple reads diff --git a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md index bf421346fd..0acccad65f 100644 --- a/apps/landing/src/content/docs/agent-tracing/provider-sdks.md +++ b/apps/landing/src/content/docs/agent-tracing/provider-sdks.md @@ -11,7 +11,7 @@ Use this guide when your agent is your own loop around `client.chat.completions. ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-provider-sdks](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-provider-sdks) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-provider-sdks](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-provider-sdks) skill and follows it. ```text Set up Maple agent tracing for the OpenAI, Anthropic or Gemini SDK in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. EU organizations should say EU region. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Configure the exporter diff --git a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md index 1531514546..02f9e465c9 100644 --- a/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/pydantic-ai.md @@ -11,7 +11,7 @@ Pydantic AI already emits OpenTelemetry spans for runs, model calls and tool cal ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-pydantic-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-pydantic-ai) skill and follows it. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-pydantic-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-pydantic-ai) skill and follows it. ```text Set up Maple agent tracing for Pydantic AI in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Export spans to Maple diff --git a/apps/landing/src/content/docs/agent-tracing/smolagents.md b/apps/landing/src/content/docs/agent-tracing/smolagents.md index 17d15344f1..6543c8296e 100644 --- a/apps/landing/src/content/docs/agent-tracing/smolagents.md +++ b/apps/landing/src/content/docs/agent-tracing/smolagents.md @@ -11,7 +11,7 @@ smolagents is traced with OpenInference's `openinference-instrumentation-smolage ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-smolagents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-smolagents) skill and follows it. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-smolagents](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-smolagents) skill and follows it. ```text Set up Maple agent tracing for smolagents in this project. @@ -21,7 +21,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the instrumentor and export to Maple diff --git a/apps/landing/src/content/docs/agent-tracing/spring-ai.md b/apps/landing/src/content/docs/agent-tracing/spring-ai.md index a3a24d6a8b..6501f5c381 100644 --- a/apps/landing/src/content/docs/agent-tracing/spring-ai.md +++ b/apps/landing/src/content/docs/agent-tracing/spring-ai.md @@ -13,7 +13,7 @@ Tested with Spring AI 2.0.1 on Spring Boot 4.1.1 and Java 21. Spring AI 1.1 on B ## Quick setup with a coding agent -Copy this prompt into Claude Code, Codex, Cursor or another agent that can run shell commands. It installs the [maple-agent-tracing-spring-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai) skill, which contains every step of this guide. +Copy this prompt into a coding agent that can run shell commands, such as Claude Code, Codex or Cursor. It installs the [maple-agent-tracing-spring-ai](https://github.com/MapleTechLabs/maple/tree/main/skills/maple-agent-tracing-spring-ai) skill and follows it. ```text Set up Maple agent tracing for Spring AI in this project. @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the OpenTelemetry starter diff --git a/apps/landing/src/content/docs/agent-tracing/strands.md b/apps/landing/src/content/docs/agent-tracing/strands.md index 757937a92d..ed7c45b030 100644 --- a/apps/landing/src/content/docs/agent-tracing/strands.md +++ b/apps/landing/src/content/docs/agent-tracing/strands.md @@ -23,7 +23,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install and configure the exporter diff --git a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md index 0077731e67..2c88e5ad48 100644 --- a/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md +++ b/apps/landing/src/content/docs/agent-tracing/vercel-ai-sdk.md @@ -25,7 +25,7 @@ Install the skill with `npx skills add MapleTechLabs/maple/skills --skill maple- My Maple ingest key is maple_pk_... and my organization is in the US region. ``` -Your ingest key is in **Settings → Ingestion**. +Your ingest key is in **Settings → Ingestion**. If your organization is in the EU region, change `US` to `EU` in the prompt. ## Install the packages From 0e03c81b2a4df58733f2cfd688cc395bc0d0f501 Mon Sep 17 00:00:00 2001 From: JeremyFunk Date: Tue, 29 Sep 2026 15:00:03 +0200 Subject: [PATCH 11/39] feat(docs): render install commands as npm/pnpm/bun and pip/uv tabs --- apps/landing/astro.config.mjs | 4 +- .../src/components/docs/LanguageTabs.astro | 125 +------------- .../src/content/docs/agent-tracing/litellm.md | 3 +- apps/landing/src/layouts/DocsLayout.astro | 1 + apps/landing/src/lib/docs-tabs.ts | 82 +++++++++ apps/landing/src/lib/remark-install-tabs.mjs | 158 ++++++++++++++++++ .../src/lib/remark-install-tabs.test.ts | 54 ++++++ apps/landing/src/styles/global.css | 72 ++++++++ 8 files changed, 377 insertions(+), 122 deletions(-) create mode 100644 apps/landing/src/lib/docs-tabs.ts create mode 100644 apps/landing/src/lib/remark-install-tabs.mjs create mode 100644 apps/landing/src/lib/remark-install-tabs.test.ts diff --git a/apps/landing/astro.config.mjs b/apps/landing/astro.config.mjs index 29694ac703..d68980b05a 100644 --- a/apps/landing/astro.config.mjs +++ b/apps/landing/astro.config.mjs @@ -5,6 +5,7 @@ import { defineConfig } from "astro/config" import { transformerNotationDiff, transformerNotationHighlight } from "@shikijs/transformers" import { unified } from "@astrojs/markdown-remark" import rehypeTableWrap from "./src/lib/rehype-table-wrap.mjs" +import remarkInstallTabs from "./src/lib/remark-install-tabs.mjs" import { codeTheme } from "./src/lib/code-theme.mjs" import mdx from "@astrojs/mdx" import { paraglideVitePlugin } from "@inlang/paraglide-js" @@ -99,7 +100,8 @@ export default defineConfig({ markdown: { // Stay on the remark pipeline: Sätteri doesn't run the Shiki transformers // configured below. Revisit when the transformer story lands there. - processor: unified({ rehypePlugins: [rehypeTableWrap] }), + // MDX inherits these plugins. + processor: unified({ remarkPlugins: [remarkInstallTabs], rehypePlugins: [rehypeTableWrap] }), shikiConfig: { theme: codeTheme, wrap: true, diff --git a/apps/landing/src/components/docs/LanguageTabs.astro b/apps/landing/src/components/docs/LanguageTabs.astro index 9fadd55c1e..a7da1dd248 100644 --- a/apps/landing/src/components/docs/LanguageTabs.astro +++ b/apps/landing/src/components/docs/LanguageTabs.astro @@ -2,7 +2,9 @@ // Language/framework switcher for docs pages. Children are `` // panels holding ordinary markdown, so code blocks keep their highlighting and // copy buttons. The pick is remembered and shared by every group on the site. -// Without JavaScript the first panel shows and the rest stay hidden. +// Without JavaScript the first panel shows and the rest stay hidden. Styles live +// in global.css and behavior in lib/docs-tabs.ts, shared with the package-manager +// switchers remark-install-tabs.mjs renders. import LanguageLogo from "./LanguageLogo.astro"; import { LANGUAGE_IDS, type LanguageId } from "../../lib/docs-languages"; @@ -20,7 +22,7 @@ const { tabs, label = "Language" } = Astro.props; const isLanguage = (id: string): id is LanguageId => (LANGUAGE_IDS as readonly string[]).includes(id); --- -
+
{tabs.map((tab, index) => (
- - diff --git a/apps/landing/src/content/docs/agent-tracing/litellm.md b/apps/landing/src/content/docs/agent-tracing/litellm.md index 85a57a8984..c212e1bb24 100644 --- a/apps/landing/src/content/docs/agent-tracing/litellm.md +++ b/apps/landing/src/content/docs/agent-tracing/litellm.md @@ -213,9 +213,10 @@ The `ghcr.io/berriai/litellm` image works as is. A pip-installed proxy needs the ```bash pip install "litellm[proxy]==1.103.0" "opentelemetry-sdk==1.43.0" \ "opentelemetry-exporter-otlp-proto-http==1.43.0" "opentelemetry-instrumentation-fastapi==0.64b0" -litellm --config config.yaml ``` +Then start it with `litellm --config config.yaml`. + In your app, keep `agent_span` and `run_tool`, drop the LiteLLM logger from `tracing.py`, and send `traceparent` and `x-litellm-session-id` with every request: ```py diff --git a/apps/landing/src/layouts/DocsLayout.astro b/apps/landing/src/layouts/DocsLayout.astro index 11a0a6070a..819c119744 100644 --- a/apps/landing/src/layouts/DocsLayout.astro +++ b/apps/landing/src/layouts/DocsLayout.astro @@ -161,6 +161,7 @@ const hasToc = !wide && headings.some((h) => h.depth >= 2 && h.depth <= 3);