From 5e76628f24726792a9327f4ef93bf39f6b778d28 Mon Sep 17 00:00:00 2001 From: Alexis Georges Date: Tue, 11 Aug 2026 16:12:01 -0400 Subject: [PATCH 1/3] feat(client)!: put conversation content on spans behind an opt-in flag Python writes the text of every prompt and every model answer onto its spans today, unconditionally, with no way to turn it off. That text is PII, and it leaves for whatever collector the SDK points at whether or not anyone asked for it. TypeScript treats it as opt-in. This adds the layer that lets Python do the same. Nothing changes behaviour yet. The handlers still call the old writers; they move over one package at a time, and that is where the default takes effect. BREAKING CHANGE: once the handlers move onto this layer, prompt and completion content will be absent from spans unless the caller passes capture_content=True. Anyone reading gen_ai.prompt.0.content today will need to opt in. The gate is an argument rather than ambient state, so it is visible at every call site. Handlers check it a second time before building the argument, which looks redundant and is not: the guard here makes a forgotten call site harmless, and the guard there avoids walking a conversation and serialising JSON that would then be discarded, once per model turn, inside a loop. Three carriers hold the same content, deliberately. The canonical GenAI attributes are what the semantic conventions make normative. The OpenLLMetry indexed attributes are the only ones LaunchDarkly's trace view reads today, so canonical alone renders an empty transcript. The legacy span events are redundant and deprecated, but every published version has emitted them and removing them would silently break anyone who learned to read them. All three are written from the same messages behind the same gate, so they cannot disagree. The system prompt goes in twice, its own canonical attribute and index 0 of the OpenLLMetry carrier, because that shape has no slot for it and dropping it there hides the system prompt from the only view that renders. The two legacy events are asymmetric: the output side writes nothing at all when there are no messages, the input side still adds its event. That matches the TypeScript source rather than being tidy, and there is a test saying so, because implementing the two symmetrically is the obvious thing to do and would diverge. to_semconv_finish_reason maps Anthropic's and OpenAI's own words onto one enum. Passing them through untranslated made a consumer grouping by finish reason see `stop` and `end_turn` as two different outcomes for the same event. An unmapped word passes through verbatim rather than being coerced, which is the signal to add a row; pause_turn is deliberately absent, because no value in the enum means "did not finish". The two LangChain helpers live here rather than in each LangChain package, because both need exactly the same conversion and a copy in each package is how the span code drifted apart the last time. The narrowing is structural, so the client takes no dependency on LangChain. --- .../src/launchdarkly_ai_server/__init__.py | 24 + .../src/launchdarkly_ai_server/content.py | 484 ++++++++++++++++++ packages/client/tests/test_content.py | 424 +++++++++++++++ 3 files changed, 932 insertions(+) create mode 100644 packages/client/src/launchdarkly_ai_server/content.py create mode 100644 packages/client/tests/test_content.py diff --git a/packages/client/src/launchdarkly_ai_server/__init__.py b/packages/client/src/launchdarkly_ai_server/__init__.py index 039a521..72cba89 100644 --- a/packages/client/src/launchdarkly_ai_server/__init__.py +++ b/packages/client/src/launchdarkly_ai_server/__init__.py @@ -3,6 +3,19 @@ __version__ = "0.1.3" # x-release-please-version from .client import ConfigInstance, config +from .content import ( + SpanMessage, + SpanMessagePart, + ToolDefinitionInput, + lang_chain_finish_reasons, + lang_chain_span_messages, + set_input_content_attributes, + set_output_content_attributes, + set_tool_call_content_attributes, + set_tool_definition_attributes, + text_message, + to_semconv_finish_reason, +) from .graph import GraphInstance, graph, resolve_graph from .judges import build_judge_tasks, run_judge, run_judges from .lifecycle import ( @@ -113,16 +126,27 @@ "TrackData", "InputTokenDetails", "RunUsage", + "SpanMessage", + "SpanMessagePart", "SpanUsage", + "ToolDefinitionInput", "UsageDict", "add_cached_tokens_to_input", "create_run_usage", "end_span_once", + "lang_chain_finish_reasons", + "lang_chain_span_messages", "lang_chain_span_usage", "number_or_zero", + "set_input_content_attributes", "set_model_identity_attributes", + "set_output_content_attributes", + "set_tool_call_content_attributes", + "set_tool_definition_attributes", "set_usage_span_attributes", "to_usage_dict", + "text_message", + "to_semconv_finish_reason", "VariationMeta", # utils "create_handler", diff --git a/packages/client/src/launchdarkly_ai_server/content.py b/packages/client/src/launchdarkly_ai_server/content.py new file mode 100644 index 0000000..394d784 --- /dev/null +++ b/packages/client/src/launchdarkly_ai_server/content.py @@ -0,0 +1,484 @@ +"""Conversation content on spans, per LaunchDarkly's "Richer LLM spans" proposal. + +Two rules drive everything in this module. + +**Attributes, not events.** Canonical content lives on span attributes. OTEP 4430 deprecated the +span-event recording API, and LaunchDarkly's ingest does not normalise content events into the +canonical shape, so a ``gen_ai.content.prompt`` event, which is what these handlers emitted +before, is read by nothing on the LaunchDarkly side. + +**Off by default.** Everything here is conversation content, which is PII. A handler opts in with +``capture_content=True``; every function below takes that decision as its ``capture`` argument and +returns without writing when it is false. Passing the flag rather than reading ambient state keeps +the gate visible at each call site. + +Handlers also test the same flag themselves before building the argument, so the check appears +twice on purpose: the guard here is the safety net that makes a forgotten call site harmless, and +the guard there avoids walking a conversation and serialising JSON that would then be discarded, +once per model turn, inside a loop. + +Two carriers are written for the same content, deliberately: + +* ``gen_ai.input.messages`` / ``gen_ai.output.messages`` / ``gen_ai.system_instructions`` / + ``gen_ai.tool.definitions`` are the canonical JSON shape from the OTel GenAI semantic + conventions, and the shape the proposal makes normative. +* ``gen_ai.prompt.{i}.role|content`` / ``gen_ai.completion.{i}.role|content`` are the OpenLLMetry + shape, one numbered attribute per field. LaunchDarkly's LLM trace view and conversation view + read *only* this one today, so canonical attributes alone would render as an empty transcript. + +A third carrier is written alongside them: the ``gen_ai.content.prompt`` / +``gen_ai.content.completion`` span events. They are redundant today, but they are what every +published version of these handlers has emitted, and removing them would silently break any +consumer that learned to read them. They are written from the same messages and behind the same +gate as everything else here, so the three carriers cannot disagree. +""" + +from __future__ import annotations + +import json +from dataclasses import dataclass, field +from typing import Any, Literal + +# ─── Message shapes ────────────────────────────────────────────────────────── + + +@dataclass +class SpanMessagePart: + """One typed piece of a message, mirroring the ``parts`` union in the GenAI JSON schemas. + + ``type`` selects which of the remaining fields carry meaning: + + * ``text`` and ``reasoning`` use ``content``. + * ``tool_call`` uses ``name``, and optionally ``id`` and ``arguments``. + * ``tool_call_response`` uses ``result``, and optionally ``id``. + + One dataclass rather than a union of four, because mypy's strict mode makes a tagged union of + dataclasses awkward to narrow at the call sites in the handler packages, and the JSON shape is + the contract rather than the Python type. + """ + + type: Literal["text", "reasoning", "tool_call", "tool_call_response"] + content: str | None = None + id: str | None = None + name: str | None = None + arguments: Any = None + result: Any = None + + def to_canonical(self) -> dict[str, Any]: + """The canonical JSON form: only the members this part's type actually uses.""" + out: dict[str, Any] = {"type": self.type} + if self.type in ("text", "reasoning"): + out["content"] = self.content or "" + return out + if self.type == "tool_call": + if self.id is not None: + out["id"] = self.id + out["name"] = self.name or "" + if self.arguments is not None: + out["arguments"] = self.arguments + return out + if self.id is not None: + out["id"] = self.id + if self.result is not None: + out["result"] = self.result + return out + + def to_text(self) -> str: + """Flattens this part to the string the OpenLLMetry carrier holds. + + Text and reasoning contribute their text. Tool traffic becomes JSON, because OpenLLMetry + has no place to put structure. + """ + if self.type in ("text", "reasoning"): + return self.content or "" + if self.type == "tool_call": + return json.dumps({"name": self.name or "", "arguments": self.arguments}) + if isinstance(self.result, str): + return self.result + return json.dumps(self.result) + + +@dataclass +class SpanMessage: + """One turn of the conversation, in the canonical ``{role, parts}`` shape.""" + + role: str + parts: list[SpanMessagePart] = field(default_factory=list) + #: Output messages only. Why the model stopped producing this message. + finish_reason: str | None = None + + def to_canonical(self) -> dict[str, Any]: + """``snake_case`` keys, and no absent members.""" + out: dict[str, Any] = { + "role": self.role, + "parts": [p.to_canonical() for p in self.parts], + } + if self.finish_reason is not None: + out["finish_reason"] = self.finish_reason + return out + + def to_text(self) -> str: + """The parts joined by newlines, with empty parts dropped.""" + return "\n".join(t for t in (p.to_text() for p in self.parts) if t) + + +def text_message(role: str, content: str) -> SpanMessage: + """Convenience for the common case: a whole message that is one block of text.""" + return SpanMessage(role=role, parts=[SpanMessagePart(type="text", content=content)]) + + +@dataclass +class ToolDefinitionInput: + """One entry of the tool catalog, as it was actually offered to the model. + + Deliberately not the AI Config's own tool type: a handler filters the configured tools down to + the ones it has a registered implementation for, and it is that filtered set the model saw. + Recording the unfiltered config would misreport what the model could have called. + """ + + name: str + description: str | None = None + parameters: Any = None + + +# ─── Finish reasons ────────────────────────────────────────────────────────── + +#: Every provider spelling this SDK actually serves, mapped onto semconv's vocabulary. +#: +#: The vocabulary is ``stop``, ``length``, ``content_filter``, ``tool_calls`` and ``error``. +#: Anthropic and OpenAI are the only two vendors behind these six handlers, so they are the only +#: two groups here. Keys are compared lower-cased, which costs one ``lower()`` and means a provider +#: that shouts its enum is mapped rather than passed through as a stray value. +#: +#: ``pause_turn`` is deliberately absent. Anthropic returns it when a long-running server-side tool +#: suspends a turn that has not actually finished, and no semconv value means "did not finish", so +#: it falls through to the passthrough below rather than being flattened into ``stop``. +#: +#: The OpenAI rows are load-bearing only for the two LangChain handlers, which can serve an OpenAI +#: model and do read a ``finish_reason`` string. The two OpenAI handlers use the Responses API, +#: which has no such field, and derive the reason themselves. They must not use this table. +_SEMCONV_FINISH_REASONS: dict[str, str] = { + # Anthropic `stop_reason` + "end_turn": "stop", + "stop_sequence": "stop", + "max_tokens": "length", + "tool_use": "tool_calls", + "refusal": "content_filter", + # OpenAI Chat Completions `finish_reason`, mostly already the vocabulary + "stop": "stop", + "length": "length", + "tool_calls": "tool_calls", + "content_filter": "content_filter", + "function_call": "tool_calls", +} + + +def to_semconv_finish_reason(raw: str | None) -> str | None: + """Maps one provider's finish reason onto semconv's ``gen_ai.response.finish_reasons`` vocabulary. + + This SDK used to pass the provider's string through untranslated, on the argument that + translating ``end_turn`` into ``stop`` loses information. Measuring it settled the argument the + other way: a single run emits ``chat`` spans from more than one handler, so a consumer grouping + by finish reason saw ``stop`` and ``end_turn`` as two different outcomes for the same event. + + Nothing is lost. The provider's own wording is still on the span, because the raw response is + what ``gen_ai.output.messages`` was built from, and an unrecognised reason is passed through + verbatim rather than dropped or coerced, so a new vendor spelling shows up as itself instead of + silently becoming ``stop``. That passthrough is the signal to add a row to the table above. + """ + if not raw or not isinstance(raw, str): + return None + return _SEMCONV_FINISH_REASONS.get(raw.lower(), raw) + + +def lang_chain_finish_reasons(source: Any) -> list[str] | None: + """Reads the finish reasons out of a LangChain ``LLMResult`` or a single ``AIMessage``. + + Both shapes are accepted because both are what the handlers hold: the agents handler finishes a + turn from an ``LLMResult``, while the messages handler finishes one from the ``AIMessage`` that + ``invoke()`` returned. + + LangChain does not normalise the field, so it is read from every place providers put it: + ``generation_info["finish_reason"]`` (OpenAI) and ``response_metadata["finish_reason"]`` or + ``["stop_reason"]`` (Anthropic), then mapped onto the semconv vocabulary. LangChain is the one + place where the same handler can serve either vendor, so it is where an untranslated + passthrough is least defensible. + + Returns ``None`` rather than an empty list when nothing is present, so a caller leaves the + attribute off instead of asserting that the turn finished for no reason. + """ + generations = _get(source, "generations") + candidates: list[Any] + if isinstance(generations, list): + candidates = [item for group in generations for item in _as_list(group)] + else: + candidates = [source] + + reasons: list[str] = [] + for candidate in candidates: + reason = to_semconv_finish_reason(_finish_reason_of(candidate)) + if reason: + reasons.append(reason) + return reasons or None + + +def _as_list(value: Any) -> list[Any]: + return value if isinstance(value, list) else [value] + + +def _get(obj: Any, key: str) -> Any: + """Reads *key* off a mapping or an object, whichever *obj* is. + + LangChain hands back objects in some paths and dicts in others, and both shapes reach these + functions. + """ + if isinstance(obj, dict): + return obj.get(key) + return getattr(obj, key, None) + + +def _finish_reason_of(candidate: Any) -> str | None: + """A ``ChatGeneration`` holds the message under ``message``; an ``AIMessage`` is the message.""" + info = _get(candidate, "generation_info") or {} + message = _get(candidate, "message") + if message is None: + message = candidate + metadata = _get(message, "response_metadata") or {} + reason = ( + (info.get("finish_reason") if isinstance(info, dict) else None) + or (metadata.get("finish_reason") if isinstance(metadata, dict) else None) + or (metadata.get("stop_reason") if isinstance(metadata, dict) else None) + ) + return reason if isinstance(reason, str) and reason else None + + +def lang_chain_span_messages( + messages: list[Any], +) -> tuple[str | None, list[SpanMessage]]: + """Converts LangChain ``BaseMessage`` values into canonical span messages. + + Returns ``(system_instructions, messages)``, with the system prompt lifted out. + + Shared here rather than copied into the two LangChain packages: both need the exact same + conversion, and a copy in each package is how the span code in this SDK drifted apart the last + time. The narrowing is structural, on ``_get_type()``, ``content`` and ``tool_calls``, so the + client takes no dependency on LangChain. + + LangChain names its roles ``human`` and ``ai``; the semconv vocabulary is ``user`` and + ``assistant``. + """ + system: list[str] = [] + converted: list[SpanMessage] = [] + + for raw in messages: + get_type = getattr(raw, "_get_type", None) + msg_type = str(get_type()) if callable(get_type) else "" + text = _lang_chain_content_text(_get(raw, "content")) + + if msg_type in ("system", "developer"): + if text: + system.append(text) + continue + + if msg_type == "tool": + tool_call_id = _get(raw, "tool_call_id") + converted.append( + SpanMessage( + role="tool", + parts=[ + SpanMessagePart( + type="tool_call_response", + id=tool_call_id if isinstance(tool_call_id, str) else None, + result=text, + ) + ], + ) + ) + continue + + parts: list[SpanMessagePart] = [] + if text: + parts.append(SpanMessagePart(type="text", content=text)) + + tool_calls = _get(raw, "tool_calls") + if isinstance(tool_calls, list): + for call in tool_calls: + call_id = _get(call, "id") + parts.append( + SpanMessagePart( + type="tool_call", + id=call_id if isinstance(call_id, str) else None, + name=str(_get(call, "name") or ""), + arguments=_get(call, "args"), + ) + ) + + if msg_type == "human": + role = "user" + elif msg_type == "ai": + role = "assistant" + else: + role = msg_type or "user" + converted.append(SpanMessage(role=role, parts=parts)) + + return ("\n".join(system) if system else None, converted) + + +def _lang_chain_content_text(content: Any) -> str: + """LangChain message content is a string or a list of typed blocks.""" + if isinstance(content, str): + return content + if not isinstance(content, list): + return "" + return "".join( + str(_get(block, "text") or "") + for block in content + if _get(block, "type") == "text" + ) + + +# ─── Writers ───────────────────────────────────────────────────────────────── + + +def _set_openllmetry_messages( + span: Any, prefix: str, messages: list[SpanMessage] +) -> None: + """Writes the OpenLLMetry carrier: one numbered attribute per field. + + ``prefix`` is ``gen_ai.prompt`` or ``gen_ai.completion``, the only difference between the input + and output halves, which is why this is one parameterised function. + + Named for the convention rather than the shape. OpenLLMetry predates the GenAI semantic + conventions and is what LaunchDarkly's LLM trace view parses today, so a reader who needs to + know why ``gen_ai.prompt.0.role`` looks the way it does has somewhere to look it up. + + An empty message list writes nothing rather than a zero-length marker: the reader treats a + missing key and an empty list identically. + """ + for index, message in enumerate(messages): + span.set_attribute(f"{prefix}.{index}.role", message.role) + span.set_attribute(f"{prefix}.{index}.content", message.to_text()) + + +def set_input_content_attributes( + span: Any, + capture: bool, + *, + system_instructions: str | None = None, + messages: list[SpanMessage] | None = None, + tool_definitions: list[ToolDefinitionInput] | None = None, +) -> None: + """Records what the model was given: system instructions, the messages, and the tool catalog. + + ``system_instructions`` is written both to its own attribute and, when present, as message 0 of + the OpenLLMetry carrier. That shape has no separate slot for it, and dropping it there would + hide the system prompt from the only view that renders today. + + The ``gen_ai.content.prompt`` event is written whenever *capture* is true, even with nothing to + say, which is asymmetric with the output side below. That asymmetry matches the TypeScript SDK + and is deliberate rather than tidy. + """ + if not capture: + return + + msgs = messages or [] + + if system_instructions: + span.set_attribute( + "gen_ai.system_instructions", + json.dumps([{"type": "text", "content": system_instructions}]), + ) + + if msgs: + span.set_attribute( + "gen_ai.input.messages", + json.dumps([m.to_canonical() for m in msgs]), + ) + + # The system prompt is a message here, unlike in the canonical carrier where it has its own + # attribute. OpenLLMetry has no separate slot for it, so it goes in at index 0. + openllmetry_messages = ( + [text_message("system", system_instructions), *msgs] + if system_instructions + else msgs + ) + _set_openllmetry_messages(span, "gen_ai.prompt", openllmetry_messages) + span.add_event( + "gen_ai.content.prompt", + { + "gen_ai.prompt": "\n".join( + f"{m.role}: {m.to_text()}" for m in openllmetry_messages + ) + }, + ) + + if tool_definitions: + set_tool_definition_attributes(span, capture, tool_definitions) + + +def set_output_content_attributes( + span: Any, capture: bool, messages: list[SpanMessage] +) -> None: + """Records what the model produced. + + Unlike the input side, this writes nothing at all when there are no messages, event included. + """ + if not capture or not messages: + return + span.set_attribute( + "gen_ai.output.messages", + json.dumps([m.to_canonical() for m in messages]), + ) + _set_openllmetry_messages(span, "gen_ai.completion", messages) + span.add_event( + "gen_ai.content.completion", + {"gen_ai.completion": "\n".join(m.to_text() for m in messages)}, + ) + + +def set_tool_definition_attributes( + span: Any, capture: bool, tools: list[ToolDefinitionInput] +) -> None: + """Records the tools the model was allowed to call. + + The catalog is content because tool descriptions and parameter schemas routinely embed + customer-specific detail, so it sits behind the same gate as the messages. + """ + if not capture or not tools: + return + definitions: list[dict[str, Any]] = [] + for tool in tools: + entry: dict[str, Any] = {"type": "function", "name": tool.name} + if tool.description: + entry["description"] = tool.description + if tool.parameters is not None: + entry["parameters"] = tool.parameters + definitions.append(entry) + span.set_attribute("gen_ai.tool.definitions", json.dumps(definitions)) + + +def set_tool_call_content_attributes( + span: Any, + capture: bool, + *, + arguments: Any = None, + result: Any = None, +) -> None: + """Records one tool call's arguments and result on its ``execute_tool`` span. + + Both are stringified rather than written as native attribute values: an argument bag is an + object, and OTel attributes hold only primitives and sequences of primitives. + """ + if not capture: + return + if arguments is not None: + span.set_attribute( + "gen_ai.tool.call.arguments", _stringify_tool_value(arguments) + ) + if result is not None: + span.set_attribute("gen_ai.tool.call.result", _stringify_tool_value(result)) + + +def _stringify_tool_value(value: Any) -> str: + """A string passes through unchanged; anything else becomes JSON.""" + return value if isinstance(value, str) else json.dumps(value) diff --git a/packages/client/tests/test_content.py b/packages/client/tests/test_content.py new file mode 100644 index 0000000..8a94930 --- /dev/null +++ b/packages/client/tests/test_content.py @@ -0,0 +1,424 @@ +"""Tests for the span content layer. + +Covers TELEMETRY-CONTRACT.md sections 5 (finish reasons) and 7 (content capture). + +The recurring assertion in this file is the gate: with ``capture=False`` nothing at all reaches the +span. Conversation content is PII, so that is the behaviour worth pinning hardest. +""" + +from __future__ import annotations + +import json +from typing import Any + +from launchdarkly_ai_server.content import ( + SpanMessage, + SpanMessagePart, + ToolDefinitionInput, + lang_chain_finish_reasons, + lang_chain_span_messages, + set_input_content_attributes, + set_output_content_attributes, + set_tool_call_content_attributes, + set_tool_definition_attributes, + text_message, + to_semconv_finish_reason, +) + + +class FakeSpan: + """Records what a handler wrote, so a test can assert on the whole span at once.""" + + def __init__(self) -> None: + self.attributes: dict[str, Any] = {} + self.events: list[tuple[str, dict[str, Any]]] = [] + + def set_attribute(self, key: str, value: Any) -> None: + self.attributes[key] = value + + def add_event(self, name: str, attributes: dict[str, Any] | None = None) -> None: + self.events.append((name, attributes or {})) + + +# ─── to_semconv_finish_reason ──────────────────────────────────────────────── + + +class TestToSemconvFinishReason: + def test_maps_every_anthropic_spelling(self) -> None: + assert to_semconv_finish_reason("end_turn") == "stop" + assert to_semconv_finish_reason("stop_sequence") == "stop" + assert to_semconv_finish_reason("max_tokens") == "length" + assert to_semconv_finish_reason("tool_use") == "tool_calls" + assert to_semconv_finish_reason("refusal") == "content_filter" + + def test_maps_every_openai_spelling(self) -> None: + assert to_semconv_finish_reason("stop") == "stop" + assert to_semconv_finish_reason("length") == "length" + assert to_semconv_finish_reason("tool_calls") == "tool_calls" + assert to_semconv_finish_reason("content_filter") == "content_filter" + assert to_semconv_finish_reason("function_call") == "tool_calls" + + def test_compares_lower_cased(self) -> None: + assert to_semconv_finish_reason("END_TURN") == "stop" + assert to_semconv_finish_reason("Max_Tokens") == "length" + + def test_passes_an_unmapped_reason_through_verbatim(self) -> None: + # The passthrough is the signal to add a row to the table, so it must not become 'stop'. + assert to_semconv_finish_reason("brand_new_reason") == "brand_new_reason" + + def test_pause_turn_is_deliberately_unmapped(self) -> None: + # No semconv value means "did not finish", so this must not be flattened into 'stop'. + assert to_semconv_finish_reason("pause_turn") == "pause_turn" + + def test_absent_and_empty_produce_nothing(self) -> None: + assert to_semconv_finish_reason(None) is None + assert to_semconv_finish_reason("") is None + + +# ─── lang_chain_finish_reasons ─────────────────────────────────────────────── + + +class _Message: + def __init__(self, response_metadata: dict[str, Any] | None = None) -> None: + self.response_metadata = response_metadata or {} + + +class TestLangChainFinishReasons: + def test_reads_generation_info_first(self) -> None: + result = {"generations": [[{"generation_info": {"finish_reason": "length"}}]]} + assert lang_chain_finish_reasons(result) == ["length"] + + def test_falls_back_to_response_metadata_finish_reason(self) -> None: + msg = _Message({"finish_reason": "tool_calls"}) + assert lang_chain_finish_reasons(msg) == ["tool_calls"] + + def test_falls_back_to_response_metadata_stop_reason(self) -> None: + # Anthropic through LangChain: the reason lands under a different key and still maps. + msg = _Message({"stop_reason": "end_turn"}) + assert lang_chain_finish_reasons(msg) == ["stop"] + + def test_accepts_a_bare_ai_message(self) -> None: + assert lang_chain_finish_reasons(_Message({"finish_reason": "stop"})) == [ + "stop" + ] + + def test_flattens_nested_generations(self) -> None: + result = { + "generations": [ + [{"generation_info": {"finish_reason": "stop"}}], + [{"generation_info": {"finish_reason": "length"}}], + ] + } + assert lang_chain_finish_reasons(result) == ["stop", "length"] + + def test_returns_none_not_empty_list_when_nothing_is_present(self) -> None: + # None leaves the attribute off; [] would assert the turn finished for no reason. + assert lang_chain_finish_reasons(_Message()) is None + assert lang_chain_finish_reasons({"generations": []}) is None + + +# ─── lang_chain_span_messages ──────────────────────────────────────────────── + + +class _LCMessage: + def __init__( + self, + msg_type: str, + content: Any = "", + tool_calls: list[Any] | None = None, + tool_call_id: str | None = None, + ) -> None: + self._type = msg_type + self.content = content + if tool_calls is not None: + self.tool_calls = tool_calls + if tool_call_id is not None: + self.tool_call_id = tool_call_id + + def _get_type(self) -> str: + return self._type + + +class TestLangChainSpanMessages: + def test_lifts_the_system_prompt_out(self) -> None: + system, messages = lang_chain_span_messages( + [_LCMessage("system", "Be helpful."), _LCMessage("human", "hi")] + ) + assert system == "Be helpful." + assert [m.role for m in messages] == ["user"] + + def test_treats_developer_as_system(self) -> None: + system, messages = lang_chain_span_messages([_LCMessage("developer", "rules")]) + assert system == "rules" + assert messages == [] + + def test_renames_langchain_roles_to_semconv_roles(self) -> None: + _, messages = lang_chain_span_messages( + [_LCMessage("human", "q"), _LCMessage("ai", "a")] + ) + assert [m.role for m in messages] == ["user", "assistant"] + + def test_converts_a_tool_message_to_a_response_part(self) -> None: + _, messages = lang_chain_span_messages( + [_LCMessage("tool", "42", tool_call_id="call_1")] + ) + assert messages[0].role == "tool" + part = messages[0].parts[0] + assert part.type == "tool_call_response" + assert part.id == "call_1" + assert part.result == "42" + + def test_carries_tool_calls_off_an_assistant_message(self) -> None: + _, messages = lang_chain_span_messages( + [ + _LCMessage( + "ai", + "calling", + tool_calls=[ + {"id": "c1", "name": "get_weather", "args": {"city": "NYC"}} + ], + ) + ] + ) + parts = messages[0].parts + assert parts[0].type == "text" + assert parts[1].type == "tool_call" + assert parts[1].name == "get_weather" + assert parts[1].arguments == {"city": "NYC"} + + def test_keeps_only_text_blocks_from_block_list_content(self) -> None: + _, messages = lang_chain_span_messages( + [ + _LCMessage( + "human", + [ + {"type": "text", "text": "look at "}, + {"type": "image_url", "image_url": "http://x"}, + {"type": "text", "text": "this"}, + ], + ) + ] + ) + assert messages[0].parts[0].content == "look at this" + + def test_joins_several_system_messages(self) -> None: + system, _ = lang_chain_span_messages( + [_LCMessage("system", "one"), _LCMessage("system", "two")] + ) + assert system == "one\ntwo" + + def test_returns_none_for_system_when_there_is_none(self) -> None: + system, _ = lang_chain_span_messages([_LCMessage("human", "hi")]) + assert system is None + + +# ─── The capture gate ──────────────────────────────────────────────────────── + + +class TestCaptureGate: + def test_input_writes_nothing_when_capture_is_off(self) -> None: + span = FakeSpan() + set_input_content_attributes( + span, + False, + system_instructions="secret", + messages=[text_message("user", "PII")], + tool_definitions=[ToolDefinitionInput(name="t")], + ) + assert span.attributes == {} + assert span.events == [] + + def test_output_writes_nothing_when_capture_is_off(self) -> None: + span = FakeSpan() + set_output_content_attributes(span, False, [text_message("assistant", "PII")]) + assert span.attributes == {} + assert span.events == [] + + def test_tool_definitions_write_nothing_when_capture_is_off(self) -> None: + span = FakeSpan() + set_tool_definition_attributes(span, False, [ToolDefinitionInput(name="t")]) + assert span.attributes == {} + + def test_tool_call_io_writes_nothing_when_capture_is_off(self) -> None: + span = FakeSpan() + set_tool_call_content_attributes(span, False, arguments={"a": 1}, result="r") + assert span.attributes == {} + + +# ─── Input content ─────────────────────────────────────────────────────────── + + +class TestInputContentAttributes: + def test_writes_all_three_carriers(self) -> None: + span = FakeSpan() + set_input_content_attributes( + span, + True, + system_instructions="Be brief.", + messages=[text_message("user", "hi")], + ) + + assert json.loads(span.attributes["gen_ai.system_instructions"]) == [ + {"type": "text", "content": "Be brief."} + ] + assert json.loads(span.attributes["gen_ai.input.messages"]) == [ + {"role": "user", "parts": [{"type": "text", "content": "hi"}]} + ] + # The system prompt takes index 0 of the OpenLLMetry carrier, which has no slot of its own. + assert span.attributes["gen_ai.prompt.0.role"] == "system" + assert span.attributes["gen_ai.prompt.0.content"] == "Be brief." + assert span.attributes["gen_ai.prompt.1.role"] == "user" + assert span.attributes["gen_ai.prompt.1.content"] == "hi" + assert span.events[0][0] == "gen_ai.content.prompt" + assert span.events[0][1]["gen_ai.prompt"] == "system: Be brief.\nuser: hi" + + def test_omits_the_canonical_keys_when_there_is_nothing_to_say(self) -> None: + span = FakeSpan() + set_input_content_attributes(span, True, messages=[]) + assert "gen_ai.system_instructions" not in span.attributes + assert "gen_ai.input.messages" not in span.attributes + assert "gen_ai.prompt.0.role" not in span.attributes + + def test_still_adds_the_legacy_prompt_event_when_empty(self) -> None: + # Asymmetric with the output side on purpose. This matches the TypeScript SDK; see the + # docstring on set_input_content_attributes. + span = FakeSpan() + set_input_content_attributes(span, True, messages=[]) + assert span.events == [("gen_ai.content.prompt", {"gen_ai.prompt": ""})] + + def test_writes_the_tool_catalog_when_given_one(self) -> None: + span = FakeSpan() + set_input_content_attributes( + span, + True, + messages=[text_message("user", "hi")], + tool_definitions=[ + ToolDefinitionInput( + name="get_weather", description="d", parameters={"type": "object"} + ) + ], + ) + assert json.loads(span.attributes["gen_ai.tool.definitions"]) == [ + { + "type": "function", + "name": "get_weather", + "description": "d", + "parameters": {"type": "object"}, + } + ] + + def test_a_message_with_no_system_prompt_starts_at_index_zero(self) -> None: + span = FakeSpan() + set_input_content_attributes(span, True, messages=[text_message("user", "hi")]) + assert span.attributes["gen_ai.prompt.0.role"] == "user" + + +# ─── Output content ────────────────────────────────────────────────────────── + + +class TestOutputContentAttributes: + def test_writes_all_three_carriers(self) -> None: + span = FakeSpan() + set_output_content_attributes(span, True, [text_message("assistant", "hello")]) + assert json.loads(span.attributes["gen_ai.output.messages"]) == [ + {"role": "assistant", "parts": [{"type": "text", "content": "hello"}]} + ] + assert span.attributes["gen_ai.completion.0.role"] == "assistant" + assert span.attributes["gen_ai.completion.0.content"] == "hello" + assert span.events == [ + ("gen_ai.content.completion", {"gen_ai.completion": "hello"}) + ] + + def test_writes_nothing_at_all_for_an_empty_message_list(self) -> None: + span = FakeSpan() + set_output_content_attributes(span, True, []) + assert span.attributes == {} + assert span.events == [] + + def test_carries_the_finish_reason_into_the_canonical_shape(self) -> None: + span = FakeSpan() + message = SpanMessage( + role="assistant", + parts=[SpanMessagePart(type="text", content="hi")], + finish_reason="stop", + ) + set_output_content_attributes(span, True, [message]) + assert ( + json.loads(span.attributes["gen_ai.output.messages"])[0]["finish_reason"] + == "stop" + ) + + def test_omits_the_finish_reason_key_when_absent(self) -> None: + span = FakeSpan() + set_output_content_attributes(span, True, [text_message("assistant", "hi")]) + assert ( + "finish_reason" + not in json.loads(span.attributes["gen_ai.output.messages"])[0] + ) + + +# ─── Part flattening ───────────────────────────────────────────────────────── + + +class TestPartFlattening: + def test_tool_call_parts_become_json_in_the_flat_carrier(self) -> None: + span = FakeSpan() + message = SpanMessage( + role="assistant", + parts=[SpanMessagePart(type="tool_call", name="f", arguments={"a": 1})], + ) + set_output_content_attributes(span, True, [message]) + assert json.loads(span.attributes["gen_ai.completion.0.content"]) == { + "name": "f", + "arguments": {"a": 1}, + } + + def test_reasoning_parts_contribute_their_text(self) -> None: + span = FakeSpan() + message = SpanMessage( + role="assistant", + parts=[ + SpanMessagePart(type="reasoning", content="thinking"), + SpanMessagePart(type="text", content="answer"), + ], + ) + set_output_content_attributes(span, True, [message]) + assert span.attributes["gen_ai.completion.0.content"] == "thinking\nanswer" + + def test_empty_parts_are_dropped_from_the_join(self) -> None: + span = FakeSpan() + message = SpanMessage( + role="assistant", + parts=[ + SpanMessagePart(type="text", content=""), + SpanMessagePart(type="text", content="only this"), + ], + ) + set_output_content_attributes(span, True, [message]) + assert span.attributes["gen_ai.completion.0.content"] == "only this" + + +# ─── Tool call arguments and results ───────────────────────────────────────── + + +class TestToolCallContentAttributes: + def test_a_string_passes_through_unchanged(self) -> None: + span = FakeSpan() + set_tool_call_content_attributes(span, True, result="72F") + assert span.attributes["gen_ai.tool.call.result"] == "72F" + + def test_anything_else_becomes_json(self) -> None: + span = FakeSpan() + set_tool_call_content_attributes(span, True, arguments={"city": "NYC"}) + assert span.attributes["gen_ai.tool.call.arguments"] == '{"city": "NYC"}' + + def test_writes_only_the_side_it_was_given(self) -> None: + span = FakeSpan() + set_tool_call_content_attributes(span, True, arguments={"a": 1}) + assert "gen_ai.tool.call.arguments" in span.attributes + assert "gen_ai.tool.call.result" not in span.attributes + + def test_an_empty_tool_catalog_writes_nothing(self) -> None: + span = FakeSpan() + set_tool_definition_attributes(span, True, []) + assert span.attributes == {} From 8351105497bccaa17e1630f65a1917ad4199741f Mon Sep 17 00:00:00 2001 From: Alexis Georges Date: Wed, 12 Aug 2026 20:32:24 -0400 Subject: [PATCH 2/3] fix(client): a tool that returned nothing does not return null Every carrier here is written from the same messages so they cannot disagree, and to_canonical omits an absent tool result because the key is simply not there. The OpenLLMetry pair and the legacy content events ran the same part through to_text, which serialised the missing result and rendered the literal text null. That is the pair the LaunchDarkly LLM trace view and conversation view read today, so a tool that returned nothing would be displayed as a tool that returned a null value. Reachable wherever a provider omits the field: an Anthropic tool_result block with no content, or an OpenAI function call result whose output is absent. An absent result now contributes nothing and drops out of the flattened text, which is what to_canonical already reports for it. None is the only spelling of absent here, because to_canonical already reads it that way. The TypeScript SDK can tell an absent result from one explicitly returned as null and reports the second as null; this dataclass cannot hold that distinction, so there is nothing for the two SDKs to disagree about. Two tests: an absent result leaves the transcript empty and agrees with the canonical attribute, and a falsy result that is not absent survives. The first fails without the fix; the second fails if the fix swallows any falsy value. Fixed here rather than after the stack merges, because this layer is what introduces the flattening to Python: a later fix would mean shipping the behaviour first and then changing an attribute's contents on a released package. Matches launchdarkly/js-ai-sdk#21, which fixes the same defect where it is already live. Found by Bugbot on this PR. --- .../src/launchdarkly_ai_server/content.py | 12 ++++++++ packages/client/tests/test_content.py | 30 +++++++++++++++++++ 2 files changed, 42 insertions(+) diff --git a/packages/client/src/launchdarkly_ai_server/content.py b/packages/client/src/launchdarkly_ai_server/content.py index 394d784..0ecaf1e 100644 --- a/packages/client/src/launchdarkly_ai_server/content.py +++ b/packages/client/src/launchdarkly_ai_server/content.py @@ -88,11 +88,23 @@ def to_text(self) -> str: Text and reasoning contribute their text. Tool traffic becomes JSON, because OpenLLMetry has no place to put structure. + + An absent tool result contributes nothing, so :meth:`SpanMessage.to_text` drops it and every + carrier agrees with :meth:`to_canonical`, which omits the key. ``json.dumps(None)`` would + render it as the literal text ``null``, which reads in the trace view as a tool that returned + a null value rather than one that returned nothing. + + ``None`` is the only spelling of absent here, because :meth:`to_canonical` already treats it + that way. The TypeScript SDK can tell an absent result from one explicitly returned as null + and reports the second as ``null``; Python's dataclass cannot hold that distinction, so there + is nothing to disagree about. """ if self.type in ("text", "reasoning"): return self.content or "" if self.type == "tool_call": return json.dumps({"name": self.name or "", "arguments": self.arguments}) + if self.result is None: + return "" if isinstance(self.result, str): return self.result return json.dumps(self.result) diff --git a/packages/client/tests/test_content.py b/packages/client/tests/test_content.py index 8a94930..89e1e26 100644 --- a/packages/client/tests/test_content.py +++ b/packages/client/tests/test_content.py @@ -397,6 +397,36 @@ def test_empty_parts_are_dropped_from_the_join(self) -> None: set_output_content_attributes(span, True, [message]) assert span.attributes["gen_ai.completion.0.content"] == "only this" + def test_a_tool_result_the_provider_never_sent_leaves_the_transcript_empty( + self, + ) -> None: + # Every carrier is written from the same messages so they cannot disagree, and the canonical + # one omits an absent result. Serialising it would render the literal text `null`, so the + # trace view showed a tool that returned nothing as one that returned a null value. + span = FakeSpan() + message = SpanMessage( + role="tool", + parts=[SpanMessagePart(type="tool_call_response", id="c1")], + ) + set_output_content_attributes(span, True, [message]) + + assert span.attributes["gen_ai.completion.0.content"] == "" + # The carrier the LaunchDarkly reader parses, and the canonical one, agree there is no result. + assert json.loads(span.attributes["gen_ai.output.messages"]) == [ + {"role": "tool", "parts": [{"type": "tool_call_response", "id": "c1"}]} + ] + + def test_a_falsy_tool_result_that_is_not_absent_survives(self) -> None: + # 0, False and "" are results. Only None means the provider sent nothing, which is also how + # to_canonical reads it. + span = FakeSpan() + message = SpanMessage( + role="tool", + parts=[SpanMessagePart(type="tool_call_response", id="c1", result=0)], + ) + set_output_content_attributes(span, True, [message]) + assert span.attributes["gen_ai.completion.0.content"] == "0" + # ─── Tool call arguments and results ───────────────────────────────────────── From fb5aae3d37b1d2e7b9bae4e25b8c91f4dfa4eff4 Mon Sep 17 00:00:00 2001 From: Alexis Georges Date: Wed, 12 Aug 2026 21:36:20 -0400 Subject: [PATCH 3/3] fix(client): keep the plain text a LangChain message holds in a list LangChain types message content as str | list[str | dict], so a bare string inside the list is what the library documents rather than a malformed input. _lang_chain_content_text kept only the blocks whose type is text, so those strings were dropped and the span showed less of the conversation than the model was actually given. Reachable from a caller: history content is passed straight into HumanMessage and AIMessage with no conversion, so whatever shape the caller supplies is the shape this function reads. A bare string now contributes its own text, in the order it appears. Blocks that are not text are still ignored, and a plain string content still passes through unchanged, each with its own test so the fix cannot quietly widen. Fixed here rather than after the stack merges, for the same reason as the tool result before it: this layer is what introduces the function to Python, so a later fix would mean shipping the behaviour first and then changing it on a released package. Matches launchdarkly/js-ai-sdk#22, which fixes the same defect where it is already live. I ran both SDKs over five inputs and they agree on all of them. Found by Bugbot on this PR. --- .../src/launchdarkly_ai_server/content.py | 23 ++++++++--- packages/client/tests/test_content.py | 41 +++++++++++++++++++ 2 files changed, 58 insertions(+), 6 deletions(-) diff --git a/packages/client/src/launchdarkly_ai_server/content.py b/packages/client/src/launchdarkly_ai_server/content.py index 0ecaf1e..f21e1cb 100644 --- a/packages/client/src/launchdarkly_ai_server/content.py +++ b/packages/client/src/launchdarkly_ai_server/content.py @@ -337,16 +337,27 @@ def lang_chain_span_messages( def _lang_chain_content_text(content: Any) -> str: - """LangChain message content is a string or a list of typed blocks.""" + """LangChain message content is a string, or a list holding typed blocks and bare strings. + + LangChain types it as ``str | list[str | dict]``, so a bare string inside the list is what the + library documents rather than a malformed input. Keeping only the blocks whose ``type`` is + ``text`` dropped those strings, and the span then showed less of the conversation than the model + was given. + + Reachable from a caller: history content is passed straight into ``HumanMessage`` and + ``AIMessage`` with no conversion, so whatever shape the caller supplies is the shape this reads. + """ if isinstance(content, str): return content if not isinstance(content, list): return "" - return "".join( - str(_get(block, "text") or "") - for block in content - if _get(block, "type") == "text" - ) + texts: list[str] = [] + for block in content: + if isinstance(block, str): + texts.append(block) + elif _get(block, "type") == "text": + texts.append(str(_get(block, "text") or "")) + return "".join(texts) # ─── Writers ───────────────────────────────────────────────────────────────── diff --git a/packages/client/tests/test_content.py b/packages/client/tests/test_content.py index 89e1e26..9589504 100644 --- a/packages/client/tests/test_content.py +++ b/packages/client/tests/test_content.py @@ -147,6 +147,47 @@ def test_lifts_the_system_prompt_out(self) -> None: assert system == "Be helpful." assert [m.role for m in messages] == ["user"] + def test_keeps_a_bare_string_a_list_holds_beside_typed_blocks(self) -> None: + # LangChain types content as `str | list[str | dict]`, so a bare string in the list is what + # the library documents. Keeping only `type: "text"` blocks dropped it, and the span then + # showed less of the conversation than the model was given. + _, messages = lang_chain_span_messages( + [ + _LCMessage( + "human", + ["a bare string", {"type": "text", "text": "a typed block"}], + ) + ] + ) + assert [p.to_canonical() for p in messages[0].parts] == [ + {"type": "text", "content": "a bare stringa typed block"} + ] + + def test_still_ignores_a_block_that_is_not_text(self) -> None: + _, messages = lang_chain_span_messages( + [ + _LCMessage( + "human", + [ + { + "type": "image_url", + "image_url": "https://example.test/x.png", + }, + {"type": "text", "text": "caption"}, + ], + ) + ] + ) + assert [p.to_canonical() for p in messages[0].parts] == [ + {"type": "text", "content": "caption"} + ] + + def test_still_reads_a_plain_string_content_unchanged(self) -> None: + _, messages = lang_chain_span_messages([_LCMessage("human", "just a string")]) + assert [p.to_canonical() for p in messages[0].parts] == [ + {"type": "text", "content": "just a string"} + ] + def test_treats_developer_as_system(self) -> None: system, messages = lang_chain_span_messages([_LCMessage("developer", "rules")]) assert system == "rules"