[{"data":1,"prerenderedAt":857},["ShallowReactive",2],{"blog-agent-observability-opentelemetry-en":3},{"slug":4,"published":5,"minutes":6,"category":7,"tags":8,"keywords":14,"about":25,"sources":34,"cover":86,"og":87,"expertise":88,"locales":89,"lang":90,"title":93,"description":94,"coverAlt":95,"metaTitle":96,"takeaways":97,"faq":103,"toc":122,"blocks":159,"others":602},"agent-observability-opentelemetry","2026-10-02",12,"llmops",[9,10,11,12,13],"OpenTelemetry","LLM observability","AI agents","Tracing","Evals",[15,16,17,18,19,20,21,22,23,24],"LLM agent observability OpenTelemetry","OpenTelemetry GenAI semantic conventions","how to trace LLM agents","gen_ai semantic conventions attributes","LLM token usage and cost metrics","LLM tracing sampling","PII in LLM traces","Langfuse vs Arize Phoenix vs Datadog","AI agent tracing evals","OpenTelemetry LLM tracing",[26,28,31],{"name":9,"url":27},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FOpenTelemetry",{"name":29,"url":30},"Observability","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FObservability_(software)",{"name":32,"url":33},"Intelligent agent","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FIntelligent_agent",[35,38,41,44,47,50,53,56,59,62,65,68,71,74,77,80,83],{"title":36,"url":37},"OpenTelemetry: GenAI semantic conventions repository (open-telemetry\u002Fsemantic-conventions-genai)","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai",{"title":39,"url":40},"GenAI conventions: overview (status Development)","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai\u002Fblob\u002Fmain\u002Fdocs\u002Fgen-ai\u002FREADME.md",{"title":42,"url":43},"GenAI conventions: model spans, execute_tool and content capture","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai\u002Fblob\u002Fmain\u002Fdocs\u002Fgen-ai\u002Fgen-ai-spans.md",{"title":45,"url":46},"GenAI conventions: agent spans","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai\u002Fblob\u002Fmain\u002Fdocs\u002Fgen-ai\u002Fgen-ai-agent-spans.md",{"title":48,"url":49},"GenAI conventions: metrics","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai\u002Fblob\u002Fmain\u002Fdocs\u002Fgen-ai\u002Fgen-ai-metrics.md",{"title":51,"url":52},"GenAI conventions: inference token metrics","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai\u002Fblob\u002Fmain\u002Fdocs\u002Fgen-ai\u002Fgen-ai-token-metrics.md",{"title":54,"url":55},"GenAI conventions: events (gen_ai.evaluation.result)","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai\u002Fblob\u002Fmain\u002Fdocs\u002Fgen-ai\u002Fgen-ai-events.md",{"title":57,"url":58},"GenAI conventions: Model Context Protocol","https:\u002F\u002Fgithub.com\u002Fopen-telemetry\u002Fsemantic-conventions-genai\u002Fblob\u002Fmain\u002Fdocs\u002Fgen-ai\u002Fmcp.md",{"title":60,"url":61},"OpenTelemetry docs: GenAI conventions moved notice","https:\u002F\u002Fopentelemetry.io\u002Fdocs\u002Fspecs\u002Fsemconv\u002Fgen-ai\u002F",{"title":63,"url":64},"OpenTelemetry docs: Sampling","https:\u002F\u002Fopentelemetry.io\u002Fdocs\u002Fconcepts\u002Fsampling\u002F",{"title":66,"url":67},"John Hodge: The state of the OpenTelemetry GenAI semantic conventions (July 2026)","https:\u002F\u002Fjohn-hodge.com\u002Fblog\u002Fopentelemetry-genai-semantic-conventions\u002F",{"title":69,"url":70},"Langfuse docs: OpenTelemetry integration","https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fopentelemetry\u002Fget-started",{"title":72,"url":73},"Langfuse repository and licence","https:\u002F\u002Fgithub.com\u002Flangfuse\u002Flangfuse",{"title":75,"url":76},"Arize Phoenix repository","https:\u002F\u002Fgithub.com\u002FArize-ai\u002Fphoenix",{"title":78,"url":79},"Arize OpenInference repository","https:\u002F\u002Fgithub.com\u002FArize-ai\u002Fopeninference",{"title":81,"url":82},"Datadog docs: OpenTelemetry instrumentation for LLM Observability","https:\u002F\u002Fdocs.datadoghq.com\u002Fllm_observability\u002Finstrumentation\u002Fotel_instrumentation\u002F",{"title":84,"url":85},"Honeycomb docs: Send data with OpenTelemetry","https:\u002F\u002Fdocs.honeycomb.io\u002Fsend-data\u002Fopentelemetry\u002F","\u002Fimages\u002Fblog\u002Fagent-observability-opentelemetry\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fagent-observability-opentelemetry\u002Fog.jpg","ai-engineer",[90,91,92],"en","de","hu","Observability for LLM agents with OpenTelemetry: traces, tokens, PII and evals","How to trace LLM agents with OpenTelemetry: GenAI semantic conventions and their status, span tree, token metrics, sampling, PII, evals and tool options.","Diagram: an agent run fans out into OpenTelemetry spans for model calls, tool calls, token metrics and evaluation results, exported to a trace backend.","LLM agent observability with OpenTelemetry · Balázs Csorba",[98,99,100,101,102],"The OpenTelemetry GenAI semantic conventions now live in their own repository and are still at Development status, so pin versions and expect attribute renames.","Model one agent run as one trace: an invoke_agent root span, chat spans for every model call and execute_tool spans for every tool call. That tree is what makes loops and wasted steps visible.","Record token counts, model, finish reason and error type on every span, but keep prompts, tool arguments and results off by default (they are opt-in in the conventions) and store them separately when you need them.","Sample on outcomes, not on a coin flip: keep every error, slow or expensive run and every failed eval, and a small share of the rest. Remember that evals usually finish after the trace does.","Any OTLP backend can take agent traces; the real differences are how well it understands the gen_ai attributes, where it is hosted and what it costs. Start with the standard and keep the exporter swappable.",[104,107,110,113,116,119],{"q":105,"a":106},"What are the OpenTelemetry GenAI semantic conventions?","They are the standard attribute, span, metric and event names for generative AI workloads: gen_ai.operation.name, gen_ai.request.model, gen_ai.usage.input_tokens, execute_tool spans, invoke_agent spans and more. As of October 2026 they are maintained in the open-telemetry\u002Fsemantic-conventions-genai repository and every GenAI-specific item is still marked Development, not Stable.",{"q":108,"a":109},"How do I trace an LLM agent with OpenTelemetry?","Create one trace per agent run. Wrap the run in an invoke_agent span, wrap each model call in a chat span and each tool call in an execute_tool span, and set the gen_ai attributes for model, token usage, finish reason and errors. Use an instrumentation library for your provider or framework, add manual spans for your own tools, and export through OTLP.",{"q":111,"a":112},"Should I log prompts and responses in traces?","Not by default. The conventions treat instructions, inputs and outputs as sensitive and large, tell instrumentations not to capture them unless you opt in, and suggest storing content externally and recording references on the span in production. Capture full content in pre-production, or for a sampled and access-controlled subset.",{"q":114,"a":115},"How do I track LLM token usage and cost with OpenTelemetry?","Record gen_ai.usage.input_tokens and gen_ai.usage.output_tokens on every inference span and use the token usage counters for dashboards. The conventions define token counts, not prices, so you compute cost in your backend or pipeline from a price table keyed by the response model. Cached and reasoning tokens are subsets of the input and output totals.",{"q":117,"a":118},"Which tool is best for LLM observability: Langfuse, Phoenix, Datadog or Honeycomb?","It depends on what you already run. Langfuse and Arize Phoenix are LLM-focused and can be self-hosted, Datadog LLM Observability fits teams already on Datadog and requires gen_ai attributes from OpenTelemetry 1.37 onwards, and Honeycomb is a general OTLP trace backend. Instrument with OpenTelemetry first and the backend stays a replaceable decision.",{"q":120,"a":121},"How do I connect traces to evals?","Attach evaluation results to the span they judge. The GenAI conventions define a gen_ai.evaluation.result event with a name, score, label and explanation that should be parented to the evaluated span. Run offline evals through the same instrumented code, and turn failing production traces into new test cases.",[123,126,129,132,135,138,141,144,147,150,153,156],{"id":124,"title":125},"why-traces","Why agents need traces, not just logs",{"id":127,"title":128},"semantic-conventions","The GenAI semantic conventions: what exists and how stable it is",{"id":130,"title":131},"trace-tree","The trace tree: one run, one trace",{"id":133,"title":134},"what-to-record","What to record on each span",{"id":136,"title":137},"tokens-and-cost","Token and cost metrics",{"id":139,"title":140},"sampling","Sampling: keep the interesting runs",{"id":142,"title":143},"pii","PII and sensitive content in traces",{"id":145,"title":146},"evals","Linking traces to evals",{"id":148,"title":149},"tools","Tool options",{"id":151,"title":152},"checklist","A checklist for the first sprint",{"id":154,"title":155},"closing","Where I would not over-invest yet",{"id":157,"title":158},"sources","Sources",[160,164,167,176,179,195,198,199,202,213,216,223,224,243,252,264,267,268,271,330,333,334,337,349,352,353,364,367,394,397,398,401,404,431,439,440,447,467,468,471,508,511,512,515,536,543,544,547,548],{"type":161,"content":162},"paragraph",[163],"A classic web request is one call, one answer and a stack trace if it breaks. An agent run is a small program the model writes as it goes: it plans, calls a tool, reads the result, calls the model again, retries, and sometimes loops until a budget runs out. When such a run costs four times what it should or quietly gives a wrong answer, logs show you scattered lines, not the shape of what happened.",{"type":161,"content":165},[166],"Distributed tracing is already the right tool for \"what happened, in which order, and how long did each part take\". OpenTelemetry (OTel) is the vendor-neutral way to produce traces, and it now has a dedicated set of GenAI semantic conventions. They are young, they are still marked Development, and they have just moved house, so this article is as much about what to pin and what to wrap as about what to record.",{"type":161,"content":168},[169,170,175],"I cover the conventions and their real status, the span tree for an agent run, what to record on each span, token and cost metrics, sampling, PII, the link to evals, and how the common backends fit. It ends with a checklist. I assume you know what ",{"tag":171,"to":172,"children":173},"link","\u002Fblog\u002Fagent-loop-explained",[174],"an agent loop"," is.",{"type":177,"level":178,"id":124,"text":125},"heading",2,{"type":161,"content":180},[181,182,186,187,190,191,194],"Three failure modes of agents are almost invisible without a trace. ",{"tag":183,"children":184},"strong",[185],"Loops and wasted steps",": the model calls the same search tool five times with slightly different arguments. ",{"tag":183,"children":188},[189],"Silent degradation",": a retrieval step returns nothing, the model answers from memory and the output looks plausible. ",{"tag":183,"children":192},[193],"Cost drift",": a prompt change or a larger tool result inflates the context, and every later model call in the run gets more expensive.",{"type":161,"content":196},[197],"All three are properties of a whole run, not of a single call. A trace gives you the run as a tree with timing, token counts and outcomes per node, and it lets you ask the questions that matter: how many model calls per request at p95, which tool fails most, which step dominates latency. If you only log, you will rebuild a worse tracing system out of correlation IDs.",{"type":177,"level":178,"id":127,"text":128},{"type":161,"content":200},[201],"Semantic conventions are the agreed names for attributes, spans and metrics, so that a backend can understand telemetry from any library. For GenAI they cover model spans (inference, embeddings, retrieval, memory), agent spans (create_agent, invoke_agent, invoke_workflow, plan), tool execution, metrics, events and an MCP convention. Provider-specific pages exist for Anthropic, OpenAI, AWS Bedrock and Azure AI Inference.",{"type":161,"content":203},[204,205,208,209,212],"Two facts matter before you build on them. First, the conventions ",{"tag":183,"children":206},[207],"have moved",": the page on opentelemetry.io now only redirects to the separate open-telemetry\u002Fsemantic-conventions-genai repository. Second, they are ",{"tag":183,"children":210},[211],"not stable",". The overview page is marked Development, and as of its July 2026 review the independent write-up I used found no GenAI-specific attribute, span, metric or event marked Stable (only shared attributes such as error.type are). Expect names to change between releases.",{"type":161,"content":214},[215],"The practical consequence is a version switch. Instrumentations that supported the older conventions keep emitting the frozen v1.36-era output by default and need OTEL_SEMCONV_STABILITY_OPT_IN=gen_ai_latest_experimental to emit the newer shape. Not every framework honours that variable the same way, so check a real exported span instead of trusting the docs. Datadog, for example, requires the 1.37 or newer shape.",{"type":217,"variant":218,"title":219,"body":220},"callout","warn","Treat the conventions as a moving target",[221],[222],"I would pin the instrumentation package versions, put a thin layer of your own helpers between the application and the OTel API (so a rename is a one-line change), and keep a golden-file test that exports one span of each kind and compares attribute names. When a framework upgrade changes the output, you want the test to fail, not your dashboards to go silently empty.",{"type":177,"level":178,"id":130,"text":131},{"type":161,"content":225},[226,227,230,231,234,235,238,239,242],"The conventions define the span kinds you need. The root is an ",{"tag":183,"children":228},[229],"invoke_agent"," span (kind INTERNAL for an in-process agent, CLIENT for a remote agent service). Under it sit one ",{"tag":183,"children":232},[233],"chat"," span per model call (named after the operation and the requested model, kind CLIENT), one ",{"tag":183,"children":236},[237],"execute_tool"," span per tool call (INTERNAL) and, if your agent has an explicit planning phase, a ",{"tag":183,"children":240},[241],"plan"," span. A multi-step pipeline around agents can use an invoke_workflow span. Retrieval has its own span, named after the data source.",{"type":244,"attrs":245,"inner":249,"caption":250},"diagram",{"viewBox":246,"role":247,"aria-labelledby":248},"0 0 730 410","img","d1-otel-t d1-otel-d","\u003Ctitle id=\"d1-otel-t\">Trace tree of one agent run\u003C\u002Ftitle>\u003Cdesc id=\"d1-otel-d\">A tree of eight spans. The root is invoke_agent. Its children are plan, chat, execute_tool search_orders with a retrieval child, a second chat, execute_tool create_refund and a final chat. A dashed box at the bottom shows an evaluation result event attached to the run.\u003C\u002Fdesc>\u003Ctext x=\"20\" y=\"28\" class=\"d-title\">Trace tree of one agent run\u003C\u002Ftext>\u003Ctext x=\"710\" y=\"28\" text-anchor=\"end\" class=\"d-label\">GenAI semantic conventions\u003C\u002Ftext>\u003Crect x=\"20\" y=\"50\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-accent\" \u002F>\u003Ctext x=\"32\" y=\"68.5\" class=\"d-text\">invoke_agent support-agent\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"68.5\" class=\"d-small\">one per user request, the root of the run\u003C\u002Ftext>\u003Cpath d=\"M34 78 V102 H56\" class=\"d-line\" \u002F>\u003Crect x=\"56\" y=\"88\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-box\" \u002F>\u003Ctext x=\"68\" y=\"106.5\" class=\"d-text\">plan support-agent\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"106.5\" class=\"d-small\">optional: task decomposition\u003C\u002Ftext>\u003Cpath d=\"M34 78 V140 H56\" class=\"d-line\" \u002F>\u003Crect x=\"56\" y=\"126\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-sky\" \u002F>\u003Ctext x=\"68\" y=\"144.5\" class=\"d-text\">chat {model}\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"144.5\" class=\"d-small\">tokens in and out, finish reason\u003C\u002Ftext>\u003Cpath d=\"M34 78 V178 H56\" class=\"d-line\" \u002F>\u003Crect x=\"56\" y=\"164\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-mint\" \u002F>\u003Ctext x=\"68\" y=\"182.5\" class=\"d-text\">execute_tool search_orders\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"182.5\" class=\"d-small\">tool arguments and result: opt-in only\u003C\u002Ftext>\u003Cpath d=\"M70 192 V216 H92\" class=\"d-line\" \u002F>\u003Crect x=\"92\" y=\"202\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-box\" \u002F>\u003Ctext x=\"104\" y=\"220.5\" class=\"d-text\">retrieval orders-index\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"220.5\" class=\"d-small\">inside the tool: your own child span\u003C\u002Ftext>\u003Cpath d=\"M34 78 V254 H56\" class=\"d-line\" \u002F>\u003Crect x=\"56\" y=\"240\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-sky\" \u002F>\u003Ctext x=\"68\" y=\"258.5\" class=\"d-text\">chat {model}\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"258.5\" class=\"d-small\">second turn, the context has grown\u003C\u002Ftext>\u003Cpath d=\"M34 78 V292 H56\" class=\"d-line\" \u002F>\u003Crect x=\"56\" y=\"278\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-gold\" \u002F>\u003Ctext x=\"68\" y=\"296.5\" class=\"d-text\">execute_tool create_refund\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"296.5\" class=\"d-small\">write action: log approval and outcome\u003C\u002Ftext>\u003Cpath d=\"M34 78 V330 H56\" class=\"d-line\" \u002F>\u003Crect x=\"56\" y=\"316\" width=\"290\" height=\"28\" rx=\"8\" class=\"d-sky\" \u002F>\u003Ctext x=\"68\" y=\"334.5\" class=\"d-text\">chat {model}\u003C\u002Ftext>\u003Ctext x=\"404\" y=\"334.5\" class=\"d-small\">final answer, finish reason stop\u003C\u002Ftext>\u003Crect x=\"20\" y=\"360\" width=\"690\" height=\"34\" rx=\"8\" class=\"d-box d-dash\" \u002F>\u003Ctext x=\"365\" y=\"381\" text-anchor=\"middle\" class=\"d-small\">gen_ai.evaluation.result: score and label, parented to the span it judges\u003C\u002Ftext>",[251],"One agent run as one trace. Span names follow the conventions; the grey notes are what I would look at in each node.",{"type":161,"content":253},[254,255,258,259,263],"Two details are worth copying. The span name for a model call is the operation plus the model, for example chat plus the model name, which keeps cardinality low and makes the waterfall readable. And the conventions explicitly encourage you to instrument your ",{"tag":183,"children":256},[257],"own"," tools by hand with execute_tool spans, because an auto-instrumentation cannot know about tools that run in your code. For tools that call MCP servers, the MCP convention can carry the trace context in the request metadata (traceparent, tracestate and baggage), so the server side joins the same trace. See ",{"tag":171,"to":260,"children":261},"\u002Fblog\u002Fmcp-tool-design-lessons-jira-server",[262],"MCP tool design lessons"," for why that server-side view matters.",{"type":161,"content":265},[266],"One more point on shape: a retried model call should be one span covering all retries, not several, because the convention defines the span as the logical operation as seen by the caller. If you want to see the retries, record them as events or attributes on that span.",{"type":177,"level":178,"id":133,"text":134},{"type":161,"content":269},[270],"The standard tells you what is available; it does not tell you what is worth the storage. This is the set I would start with. Names in the second column are from the conventions unless marked as custom.",{"type":272,"head":273,"rows":280},"table",[274,276,278],[275],"Unit",[277],"Attributes to set",[279],"Why it pays off",[281,288,295,302,309,316,323],[282,284,286],[283],"Agent run (invoke_agent)",[285],"gen_ai.agent.name, gen_ai.agent.version, gen_ai.conversation.id, error.type, plus custom: release, tenant, final outcome",[287],"Group runs by agent version and session; compare releases; find the runs that ended in a handover or a failure",[289,291,293],[290],"Model call (chat)",[292],"gen_ai.operation.name, gen_ai.provider.name, gen_ai.request.model, gen_ai.response.model, gen_ai.response.finish_reasons, gen_ai.usage.input_tokens, gen_ai.usage.output_tokens, cache read and reasoning token counts, error.type",[294],"Cost per call, truncation (finish reason length), cache hit rate, model fallbacks, which provider errors dominate",[296,298,300],[297],"Request settings",[299],"gen_ai.request.temperature, gen_ai.request.max_tokens, gen_ai.request.reasoning.level, gen_ai.prompt.name, gen_ai.prompt.version",[301],"Explain behaviour changes; tie output quality to a prompt version",[303,305,307],[304],"Tool call (execute_tool)",[306],"gen_ai.tool.name, gen_ai.tool.call.id, gen_ai.tool.type, error.type, plus custom: read or write, approval given",[308],"Failure rate and latency per tool; find write actions and who approved them",[310,312,314],[311],"Retrieval",[313],"gen_ai.data_source.id, plus custom: top-k, number of hits, document IDs without content",[315],"Spot empty or poor retrieval before the model hides it behind a fluent answer",[317,319,321],[318],"Content",[320],"gen_ai.system_instructions, gen_ai.input.messages, gen_ai.output.messages, gen_ai.tool.call.arguments, gen_ai.tool.call.result (all opt-in)",[322],"Debugging and building test cases, at a privacy and storage price (see the PII section)",[324,326,328],[325],"Evaluation",[327],"gen_ai.evaluation.name, gen_ai.evaluation.score.value, gen_ai.evaluation.score.label, gen_ai.evaluation.explanation",[329],"Quality signal attached to the exact span it judges",{"type":161,"content":331},[332],"Set the attributes that a sampler may need at span creation time. The conventions list gen_ai.operation.name, gen_ai.provider.name, gen_ai.request.model and the server address (and gen_ai.agent.name for agent spans) as the ones that SHOULD be available at creation, because a head sampler cannot see attributes you add later.",{"type":177,"level":178,"id":136,"text":137},{"type":161,"content":335},[336],"Spans give you per-run detail; metrics give you cheap, long-lived trends. The conventions define token usage counters per category (input, output, cache read, cache write, reasoning) broken down by modality, which they describe as the primary instruments for consumption and a proxy for cost. Next to them sit histograms for per-operation token distribution, meant for p95 and p99 outlier detection and explicitly not for totals or cost. For agents there are histograms for invocation duration, number of inference calls per invocation and number of tool calls per invocation, plus a tool execution duration. The last two are the ones I would put on a dashboard first: they show a loop before the invoice does.",{"type":161,"content":338},[339,340,343,344,348],"Mind the arithmetic. The input token count SHOULD include cached tokens, and the cache read, cache write and reasoning counts are subsets of the input and output totals. If you add them up as separate line items you double count. The conventions also define ",{"tag":183,"children":341},[342],"no price or cost attribute",", so cost is something you compute: multiply the token categories by a price table keyed by the response model, in your backend or in a pipeline stage, and version that table. The ",{"tag":171,"to":345,"children":346},"\u002Fblog\u002Fllm-cost-latency-prompt-caching-routing",[347],"LLM cost, latency and prompt caching article"," goes into what to do once you can see the numbers.",{"type":161,"content":350},[351],"Keep metric dimensions low-cardinality: model, provider, agent name, operation, outcome. A user ID or conversation ID belongs on a span, not on a metric, or your metrics bill will grow with your user base.",{"type":177,"level":178,"id":139,"text":140},{"type":161,"content":354},[355,356,359,360,363],"Agent traces are bigger than ordinary web traces, and with content capture they can be much bigger. You will sample. OpenTelemetry distinguishes ",{"tag":183,"children":357},[358],"head sampling"," (the decision is made when the trace starts, for example a fixed percentage by trace ID) from ",{"tag":183,"children":361},[362],"tail sampling"," (the decision is made after seeing all or most spans, so you can always keep traces with errors or high latency). The same documentation is frank about the cost: tail sampling is harder to implement and operate, and the component making the decision has to be stateful.",{"type":161,"content":365},[366],"For agents I would use a simple policy, and I would treat the percentages as a starting point to tune:",{"type":368,"ordered":369,"items":370},"list",false,[371,376,380,384,389],[372,375],{"tag":183,"children":373},[374],"Keep 100%"," of runs with an error, a timeout, a handover to a human, or a user complaint.",[377,379],{"tag":183,"children":378},[374]," of runs far above your normal cost, duration or number of model calls. These are your loops.",[381,383],{"tag":183,"children":382},[374]," of runs that failed an eval, and of runs in a canary release.",[385,388],{"tag":183,"children":386},[387],"Sample a small share",", say 5 to 10 percent, of everything else, so that you still have a representative baseline.",[390,393],{"tag":183,"children":391},[392],"Do not rely on sampled traces for rates."," Compute rates and alerts from metrics (the duration, token and tool-call instruments above), not by counting sampled traces.",{"type":161,"content":395},[396],"Two agent-specific traps. A run can last minutes, so the tail sampler must wait long enough before deciding, and it must hold every span of that run in memory meanwhile. And an eval score usually arrives after the trace has been finished and exported, so it cannot drive the tail decision unless you evaluate inline. If you want failed evals to survive sampling, either run a cheap inline check on every run or attach the score later and make sure the sampled-out trace is still available for the runs you flag.",{"type":177,"level":178,"id":142,"text":143},{"type":161,"content":399},[400],"Prompts, tool arguments and tool results contain whatever your users and your systems contain: names, emails, order numbers, contract text, sometimes secrets. The conventions take a clear position. Instructions, inputs and outputs are considered sensitive and often large, so instrumentations SHOULD NOT capture them by default and SHOULD offer an opt-in. They describe three patterns: record nothing (the default), record content on span attributes, or store content externally and put only references on the spans, which is the recommended pattern in production because external storage has its own access controls.",{"type":161,"content":402},[403],"That maps to a practical setup:",{"type":368,"ordered":369,"items":405},[406,411,416,421,426],[407,410],{"tag":183,"children":408},[409],"Production default:"," metadata only. Model, tokens, finish reason, tool names, error types, IDs. Surprisingly much debugging works with this.",[412,415],{"tag":183,"children":413},[414],"Pre-production and tests:"," full content on spans, because there is no real personal data and you want maximum visibility.",[417,420],{"tag":183,"children":418},[419],"Production content:"," only through the external-storage pattern, for a sampled or flagged subset, with a short retention period and access limited to the people who need it. The conventions let an in-process hook modify or redact content before it is recorded, and that hook runs regardless of the sampling decision.",[422,425],{"tag":183,"children":423},[424],"Redact before export",", not in the backend. A processing stage in your telemetry pipeline or an in-process hook can remove patterns such as emails and card numbers; do not rely on the vendor to do it after the data has left your network.",[427,430],{"tag":183,"children":428},[429],"Use pseudonymous IDs"," for users and conversations, never raw emails or names, so you can find a run without a trace becoming a personal-data register.",{"type":161,"content":432},[433,434,438],"Tool arguments deserve special attention, because they are often the most identifying part of a run and are easy to forget. If your traces leave the EU or reach a US vendor, the same rules apply as for the model API itself, see ",{"tag":171,"to":435,"children":436},"\u002Fblog\u002Fgdpr-llm-api-eu-data-residency",[437],"GDPR and LLM APIs",". Prompt content in traces is also a prompt-injection data source: a trace viewer that renders untrusted model output is a target, which is another reason to keep access tight.",{"type":177,"level":178,"id":145,"text":146},{"type":161,"content":441},[442,443,446],"Traces tell you what happened; evals tell you whether it was good. They are most useful when joined. The conventions define a ",{"tag":183,"children":444},[445],"gen_ai.evaluation.result"," event with the evaluation name, a score value, a human-readable label, an explanation and the response ID, and say it SHOULD be parented to the GenAI operation span being evaluated, or carry the response ID when no span is available. An LLM judge or a rule-based check that runs after the fact can therefore write its verdict straight onto the node it judged.",{"type":161,"content":448},[449,450,453,454,457,458,461,462,466],"I use the join in three ways. ",{"tag":183,"children":451},[452],"Offline:"," run the eval dataset through the same instrumented agent, tag the traces with a run or release identifier (a custom attribute), and compare cost, step count and score per release in one place. ",{"tag":183,"children":455},[456],"Online:"," score a sample of production runs and alert on a falling pass rate, with the failing traces one click away. ",{"tag":183,"children":459},[460],"Feedback loop:"," when a production trace fails, copy its input (and expected behaviour) into the test set, which needs content captured, so use the external-storage pattern for the flagged runs. How to build the evals themselves is covered in ",{"tag":171,"to":463,"children":464},"\u002Fblog\u002Fllm-evals-for-product-features",[465],"LLM evals for product features",".",{"type":177,"level":178,"id":148,"text":149},{"type":161,"content":469},[470],"The nice property of OTLP is that the choice of backend comes last. What I checked in the vendors' own documentation:",{"type":272,"head":472,"rows":479},[473,475,477],[474],"Tool",[476],"How it takes OpenTelemetry",[478],"Worth knowing",[480,487,494,501],[481,483,485],[482],"Langfuse",[484],"OTLP endpoint at \u002Fapi\u002Fpublic\u002Fotel with basic auth; HTTP\u002FJSON and HTTP\u002Fprotobuf, no gRPC; maps gen_ai attributes, OpenInference and its own langfuse attributes",[486],"LLM-specific features on top (prompt linking, scoring, cost tracking); self-hostable; the repository is MIT-licensed except for ee directories under a separate licence",[488,490,492],[489],"Arize Phoenix",[491],"Built on OpenTelemetry; uses the OpenInference conventions, which are complementary to OTel; local collector at \u002Fv1\u002Ftraces",[493],"Open source and self-hosted, under the Elastic License 2.0; Arize AX is the managed sibling; strong for experiments and evals",[495,497,499],[496],"Datadog LLM Observability",[498],"OTLP over HTTP\u002Fprotobuf with a dd-api-key header; requires the OpenTelemetry 1.37 or newer gen_ai shape",[500],"Spans without any gen_ai attribute are dropped; the docs mention a delay of a few minutes before traces show up; best if you are already on Datadog",[502,504,506],[503],"Honeycomb",[505],"OTLP over gRPC, HTTP\u002Fprotobuf and HTTP\u002FJSON, with an API-key header and an EU endpoint",[507],"A general trace backend: the ingest docs I read say nothing GenAI-specific, so you build the views yourself",{"type":161,"content":509},[510],"My recommendation: instrument with OpenTelemetry and your own thin helper layer, export through a pipeline stage you control, for example an OpenTelemetry Collector (a natural place for redaction, sampling and cost enrichment), and pick the backend by hosting model and by what your team already operates. If data residency matters, a self-hosted Langfuse or Phoenix is the easiest to defend. If you already pay for a general observability platform, check how well it renders gen_ai spans before adding another tool. I did not verify other vendors, such as the Grafana stack, and leave them out on purpose.",{"type":177,"level":178,"id":151,"text":152},{"type":161,"content":513},[514],"This is the order I would work in:",{"type":368,"ordered":516,"items":517},true,[518,520,522,524,526,528,530,532,534],[519],"Pick the OTLP backend and put a pipeline stage you control in between. Do the redaction and sampling there.",[521],"Install the instrumentation for your provider or framework, pin its version and check one real exported span against the conventions (including the OTEL_SEMCONV_STABILITY_OPT_IN variable).",[523],"Add a root invoke_agent span per run and manual execute_tool spans for your own tools.",[525],"Set model, provider, token usage, finish reason and error type on every model span; add agent version, release and a pseudonymous conversation ID.",[527],"Turn off content capture in production; turn it on in pre-production; decide the external-storage route for flagged runs.",[529],"Build three dashboards: model calls and tool calls per run (p50, p95), tokens and computed cost per agent version, error and timeout rate per tool.",[531],"Add the sampling policy: all errors, outliers and failed evals, a small share of the rest.",[533],"Write eval results as gen_ai.evaluation.result events on the judged spans, and feed failing traces back into the test set.",[535],"Add a golden-file test that fails when instrumentation output changes shape.",{"type":161,"content":537},[538,539,466],"If you are working in a harness, the same instrumentation also pays off there, see ",{"tag":171,"to":540,"children":541},"\u002Fblog\u002Fharness-engineering-coding-agents",[542],"harness engineering for coding agents",{"type":177,"level":178,"id":154,"text":155},{"type":161,"content":545},[546],"Do not build deep, custom analytics on the exact attribute names of a Development-status standard. Build on the concepts, a tree of spans with token counts, outcomes and scores, and keep the mapping from concept to attribute name in one place. The conventions will keep moving, and the backends will keep catching up; a team that traces every run and can answer \"what did this run do and what did it cost\" is ahead of one that waits for the standard to settle.",{"type":177,"level":178,"id":157,"text":158},{"type":368,"ordered":516,"items":549},[550,554,557,560,563,566,569,572,575,578,581,584,587,590,593,596,599],[551],{"tag":552,"href":37,"children":553},"a",[36],[555],{"tag":552,"href":40,"children":556},[39],[558],{"tag":552,"href":43,"children":559},[42],[561],{"tag":552,"href":46,"children":562},[45],[564],{"tag":552,"href":49,"children":565},[48],[567],{"tag":552,"href":52,"children":568},[51],[570],{"tag":552,"href":55,"children":571},[54],[573],{"tag":552,"href":58,"children":574},[57],[576],{"tag":552,"href":61,"children":577},[60],[579],{"tag":552,"href":64,"children":580},[63],[582],{"tag":552,"href":67,"children":583},[66],[585],{"tag":552,"href":70,"children":586},[69],[588],{"tag":552,"href":73,"children":589},[72],[591],{"tag":552,"href":76,"children":592},[75],[594],{"tag":552,"href":79,"children":595},[78],[597],{"tag":552,"href":82,"children":598},[81],[600],{"tag":552,"href":85,"children":601},[84],[603,704,759,813],{"slug":604,"published":5,"minutes":6,"category":7,"tags":605,"keywords":610,"about":621,"sources":631,"cover":698,"og":699,"expertise":88,"locales":700,"lang":90,"title":701,"description":702,"coverAlt":703},"fine-tuning-vs-rag-vs-prompting",[606,607,608,609],"Fine-tuning","RAG","Prompting","Distillation",[611,612,613,614,615,616,617,618,619,620],"fine-tuning vs RAG","prompting vs RAG vs fine-tuning","when to fine-tune an LLM","RAG or fine-tuning","SFT vs DPO vs RFT","reinforcement fine-tuning","LLM distillation","OpenAI fine-tuning shutdown","fine-tuning decision tree","fine-tune for knowledge",[622,625,628],{"name":623,"url":624},"Fine-tuning (deep learning)","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FFine-tuning_(deep_learning)",{"name":626,"url":627},"Retrieval-augmented generation","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FRetrieval-augmented_generation",{"name":629,"url":630},"Knowledge distillation","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FKnowledge_distillation",[632,635,638,641,644,647,650,653,656,659,662,665,668,671,674,677,680,683,686,689,692,695],{"title":633,"url":634},"OpenAI community: OpenAI self-serve fine-tuning availability (wind-down announcement)","https:\u002F\u002Fcommunity.openai.com\u002Ft\u002Fopenai-s-self-serve-fine-tuning-availability\u002F1380481",{"title":636,"url":637},"Tessl: OpenAI is shutting down self-serve fine-tuning","https:\u002F\u002Ftessl.io\u002Fblog\u002Fopenai-shutting-fine-tuning-signals-for-enterprise-ai",{"title":639,"url":640},"OpenAI API docs: Model optimization","https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fmodel-optimization",{"title":642,"url":643},"OpenAI API docs: Supervised fine-tuning","https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fsupervised-fine-tuning",{"title":645,"url":646},"OpenAI API docs: Direct preference optimization","https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fdirect-preference-optimization",{"title":648,"url":649},"OpenAI API docs: Reinforcement fine-tuning","https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Freinforcement-fine-tuning",{"title":651,"url":652},"OpenAI API docs: Fine-tuning best practices","https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Ffine-tuning-best-practices",{"title":654,"url":655},"OpenAI Cookbook: Choosing between SFT, DPO and RFT","https:\u002F\u002Fdevelopers.openai.com\u002Fcookbook\u002Fexamples\u002Ffine_tuning_direct_preference_optimization_guide",{"title":657,"url":658},"Google Cloud: Introduction to tuning (Gemini)","https:\u002F\u002Fdocs.cloud.google.com\u002Fvertex-ai\u002Fgenerative-ai\u002Fdocs\u002Fmodels\u002Ftune-models",{"title":660,"url":661},"Google Cloud: About supervised fine-tuning for Gemini models","https:\u002F\u002Fdocs.cloud.google.com\u002Fvertex-ai\u002Fgenerative-ai\u002Fdocs\u002Fmodels\u002Fgemini-supervised-tuning",{"title":663,"url":664},"Google Cloud: About preference tuning for Gemini models","https:\u002F\u002Fdocs.cloud.google.com\u002Fvertex-ai\u002Fgenerative-ai\u002Fdocs\u002Fmodels\u002Fgemini-preference-tuning",{"title":666,"url":667},"Google Cloud: Reward functions for reinforcement learning fine-tuning","https:\u002F\u002Fdocs.cloud.google.com\u002Fgemini-enterprise-agent-platform\u002Fmodels\u002Ftuning\u002Freinforcement-tuning\u002Freinforcement-tuning-job\u002Freward-functions",{"title":669,"url":670},"Google Cloud: Supervised and distillation fine-tuning for open models","https:\u002F\u002Fdocs.cloud.google.com\u002Fvertex-ai\u002Fgenerative-ai\u002Fdocs\u002Fmodels\u002Fopen-model-tuning",{"title":672,"url":673},"AWS: Fine-tuning for Claude 3 Haiku in Amazon Bedrock is now generally available","https:\u002F\u002Faws.amazon.com\u002Fblogs\u002Faws\u002Ffine-tuning-for-anthropics-claude-3-haiku-model-in-amazon-bedrock-is-now-generally-available\u002F",{"title":675,"url":676},"AWS: Amazon Bedrock now supports reinforcement fine-tuning","https:\u002F\u002Faws.amazon.com\u002Fabout-aws\u002Fwhats-new\u002F2025\u002F12\u002Fbedrock-reinforcement-fine-tuning-66-base-models",{"title":678,"url":679},"AWS: Amazon Bedrock Model Distillation (preview announcement)","https:\u002F\u002Faws.amazon.com\u002Fblogs\u002Faws\u002Fbuild-faster-more-cost-efficient-highly-accurate-models-with-amazon-bedrock-model-distillation-preview\u002F",{"title":681,"url":682},"AWS: Bedrock reinforcement fine-tuning adds open-weight models","https:\u002F\u002Faws.amazon.com\u002Fabout-aws\u002Fwhats-new\u002F2026\u002F02\u002Famazon-bedrock-reinforcement-fine-tuning-openai",{"title":684,"url":685},"Microsoft Foundry blog: What is new in Foundry fine-tuning, April 2026","https:\u002F\u002Fdevblogs.microsoft.com\u002Ffoundry\u002Fwhats-new-in-foundry-finetune-april-2026\u002F",{"title":687,"url":688},"Microsoft: Announcing new fine-tuning models and techniques in Azure AI Foundry","https:\u002F\u002Fazure.microsoft.com\u002Fen-us\u002Fblog\u002Fannouncing-new-fine-tuning-models-and-techniques-in-azure-ai-foundry\u002F",{"title":690,"url":691},"Ovadia et al.: Fine-Tuning or Retrieval? Comparing Knowledge Injection in LLMs","https:\u002F\u002Farxiv.org\u002Fabs\u002F2312.05934",{"title":693,"url":694},"Gekhman et al.: Does Fine-Tuning LLMs on New Knowledge Encourage Hallucinations?","https:\u002F\u002Farxiv.org\u002Fabs\u002F2405.05904",{"title":696,"url":697},"Hu et al.: LoRA: Low-Rank Adaptation of Large Language Models","https:\u002F\u002Farxiv.org\u002Fabs\u002F2106.09685","\u002Fimages\u002Fblog\u002Ffine-tuning-vs-rag-vs-prompting\u002Fcover.webp","\u002Fimages\u002Fblog\u002Ffine-tuning-vs-rag-vs-prompting\u002Fog.jpg",[90,91,92],"Prompting vs RAG vs fine-tuning vs distillation: a decision guide for 2026","Fine-tuning vs RAG vs prompting: what each changes (knowledge, behaviour, format), what it costs, who still offers SFT, DPO and RFT in 2026, and a decision tree.","Diagram: a decision flow from a tested prompt to retrieval for missing facts, supervised fine-tuning for behaviour, and distillation into a smaller model for cost.",{"slug":705,"published":706,"updated":5,"minutes":707,"category":7,"tags":708,"keywords":713,"about":724,"sources":734,"cover":753,"og":754,"expertise":88,"locales":755,"lang":90,"title":756,"description":757,"coverAlt":758},"artificial-analysis-leaderboard-claude-opus-5-5","2026-09-28",8,[709,710,711,712],"Claude Opus 5.5","Artificial Analysis","LLM benchmarks","LLM cost",[714,715,709,716,717,718,719,720,721,722,723],"Claude Opus 5.5 benchmark","Claude Opus 5.5 pricing","Artificial Analysis Intelligence Index","AI model leaderboard","Opus 5.5 benchmark","Opus 5.5 vs GPT-6 Astra","LLM cost per task","reasoning effort setting","Opus 5.5 pricing","best LLM September 2026",[725,728,731],{"name":726,"url":727},"Claude (language model)","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FClaude_(language_model)",{"name":729,"url":730},"Large language model","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FLarge_language_model",{"name":732,"url":733},"Benchmark (computing)","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FBenchmark_(computing)",[735,738,741,744,747,750],{"title":736,"url":737},"Artificial Analysis: Claude Opus 5.5 takes the top spot on the Artificial Analysis Intelligence Index (22 September 2026)","https:\u002F\u002Fartificialanalysis.ai\u002Farticles\u002Fclaude-opus-5-5",{"title":739,"url":740},"Artificial Analysis: Claude Opus 5.5 (max) model page","https:\u002F\u002Fartificialanalysis.ai\u002Fmodels\u002Fclaude-opus-5-5",{"title":742,"url":743},"Artificial Analysis: Claude Opus 5.5 (medium) model page","https:\u002F\u002Fartificialanalysis.ai\u002Fmodels\u002Fclaude-opus-5-5-medium",{"title":745,"url":746},"Artificial Analysis: Claude Opus 5 (max) model page","https:\u002F\u002Fartificialanalysis.ai\u002Fmodels\u002Fclaude-opus-5",{"title":748,"url":749},"OfficeChai: Claude Opus 5.5 creates a 5-point lead over GPT-6 Astra","https:\u002F\u002Fofficechai.com\u002Fai\u002Fclaude-opus-5-5-creates-5-point-lead-over-gpt-6-astra-jumps-to-top-spot-on-artificial-analysis-intelligence-index\u002F",{"title":751,"url":752},"Claude API docs: Models overview, context windows and prices (as of September 2026)","https:\u002F\u002Fplatform.claude.com\u002Fdocs\u002Fen\u002Fabout-claude\u002Fmodels\u002Foverview","\u002Fimages\u002Fblog\u002Fartificial-analysis-leaderboard-claude-opus-5-5\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fartificial-analysis-leaderboard-claude-opus-5-5\u002Fog.jpg",[90,91,92],"Claude Opus 5.5 takes #1 on Artificial Analysis, and medium effort is the real story","Claude Opus 5.5 is #1 of 211 models on Artificial Analysis with 58 points. At medium effort it matches Opus 5 for $1.34 per task instead of $5.86.","Horizontal bars of Intelligence Index scores: Claude Opus 5.5 at max effort 58, GPT-6 Astra and Claude Fable 5.1 53, Opus 5 51, Opus 5.5 at medium effort 51.",{"slug":760,"published":761,"minutes":762,"category":7,"tags":763,"keywords":769,"about":779,"sources":788,"cover":807,"og":808,"expertise":88,"locales":809,"lang":90,"title":810,"description":811,"coverAlt":812},"jev-typed-decisions-llm-routing","2026-09-27",10,[764,765,766,767,768],"LLM routing","Classification","Calibration","OpenRouter","Human in the loop",[770,771,772,773,774,775,776,777,778],"llm routing","llm classification","calibrated confidence llm","typed llm output","jev model","openrouter decisions api","ai triage of code review findings","how to route low-confidence llm answers to a human","llm classifier vs structured output",[780,783,786],{"name":781,"url":782},"Calibration (statistics)","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FCalibration_(statistics)",{"name":784,"url":785},"Statistical classification","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FStatistical_classification",{"name":767,"url":787},"https:\u002F\u002Fopenrouter.ai\u002F",[789,792,795,798,801,804],{"title":790,"url":791},"OpenRouter: Jev documentation","https:\u002F\u002Fopenrouter.ai\u002Fdocs\u002Fguides\u002Fcommunity\u002Fjev",{"title":793,"url":794},"OpenRouter: What is Jev?","https:\u002F\u002Fopenrouter.ai\u002Fblog\u002Finsights\u002Fwhat-is-jev\u002F",{"title":796,"url":797},"OpenRouter: How to use Jev","https:\u002F\u002Fopenrouter.ai\u002Fblog\u002Ftutorials\u002Fhow-to-use-jev\u002F",{"title":799,"url":800},"Guo et al.: On Calibration of Modern Neural Networks","https:\u002F\u002Farxiv.org\u002Fabs\u002F1706.04599",{"title":802,"url":803},"Claude API: Structured outputs","https:\u002F\u002Fplatform.claude.com\u002Fdocs\u002Fen\u002Fbuild-with-claude\u002Fstructured-outputs",{"title":805,"url":806},"AI SDK: Generating structured data","https:\u002F\u002Fai-sdk.dev\u002Fdocs\u002Fai-sdk-core\u002Fgenerating-structured-data","\u002Fimages\u002Fblog\u002Fjev-typed-decisions-llm-routing\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fjev-typed-decisions-llm-routing\u002Fog.jpg",[90,91,92],"Typed decisions for LLM routing and triage: calibrated confidence with Jev","LLM routing with typed decisions: Choice, Score and yes-probability answers with calibrated confidence, thresholds and human hand-off, using Jev.","Fan-out diagram: one diff hunk as state feeds a Choice, a Score and a Noul question in a single decisions call.",{"slug":814,"published":761,"minutes":707,"category":7,"tags":815,"keywords":820,"about":827,"sources":832,"cover":851,"og":852,"expertise":88,"locales":853,"lang":90,"title":854,"description":855,"coverAlt":856},"llm-evals-for-product-features",[816,817,818,819],"LLM evals","LLM-as-judge","Error analysis","CI",[816,821,817,822,823,824,825,826],"AI evals","eval-driven development","pass^k vs pass@k","agent evaluation harness","regression eval suite","error analysis LLM",[828,829],{"name":729,"url":730},{"name":830,"url":831},"Software testing","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FSoftware_testing",[833,836,839,842,845,848],{"title":834,"url":835},"Anthropic: Demystifying evals for AI agents (2026)","https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fdemystifying-evals-for-ai-agents",{"title":837,"url":838},"Hamel Husain: LLM evals FAQ (updated September 2026)","https:\u002F\u002Fhamel.dev\u002Fblog\u002Fposts\u002Fevals-faq\u002F",{"title":840,"url":841},"Shankar et al., Who Validates the Validators? (2024)","https:\u002F\u002Farxiv.org\u002Fabs\u002F2404.12272",{"title":843,"url":844},"Zheng et al., Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena (2023)","https:\u002F\u002Farxiv.org\u002Fabs\u002F2306.05685",{"title":846,"url":847},"Yao et al., tau-bench: A Benchmark for Tool-Agent-User Interaction (2024)","https:\u002F\u002Farxiv.org\u002Fabs\u002F2406.12045",{"title":849,"url":850},"Anthropic: Quantifying infrastructure noise in agentic coding evals (2026)","https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Finfrastructure-noise","\u002Fimages\u002Fblog\u002Fllm-evals-for-product-features\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fllm-evals-for-product-features\u002Fog.jpg",[90,91,92],"LLM evals for product features: from hand-read traces to a CI gate","LLM evals turn a vibe check into a test suite: error analysis on real traces, grader choice, a validated LLM judge, pass^k and a CI gate.","A pipeline from real traces through open coding and a counted failure taxonomy to graders, a validated judge and a CI gate.",1791009037072]