Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions .claude/skills/llmobs-integrations/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -109,6 +109,23 @@ Note two already-shipped integrations predate this key: bedrock and the claude-a
`metadata["stop_reason"]`. Renaming those is a breaking change and has not been done — follow
`finish_reason` for new work.

### Agent Manifest

Agent spans report the agent's declared configuration as `agent_manifest`, typed by `AgentManifest`
in `ddtrace/llmobs/types.py`. Build it with the helpers in `_integrations/agent_manifest.py`:

- `build_agent_manifest(framework, agent, sections, integration_name)` runs each section in
isolation, drops unset values, and guarantees a JSON-native result. Return `{}` means "do not
annotate".
- Only emit keys declared on `AgentManifest`. Put loop-level knobs in `agent_settings` and inference
params through `filter_model_settings` (an allowlist; widening it is a security decision).
- Read declared configuration only, never per-run values (session ids, run config, interpolated
templates), so the manifest is identical run to run and version diffs stay meaningful.
- Use `instruction_fields` for instructions: a callable ships by name in `extra_instructions` and is
never called or `str()`-ed (its address changes every process).
- Use `normalize_tool` so every tool is `{name, description?, parameters: {p: {type?, required?}}}`.
- A description other agents see when routing to this one goes in `handoff_description`.

## Key Constraints

- **`submit_to_llmobs=True`** must be set on `LlmRequestEvent` for event-based request spans or passed to `integration.trace()` for direct LLMObs spans
Expand Down
104 changes: 87 additions & 17 deletions ddtrace/llmobs/_integrations/agent_manifest.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import math
import types
from typing import Any
from typing import Callable
from typing import Optional
from typing import TypeVar
from typing import Union
Expand Down Expand Up @@ -172,6 +173,88 @@ def wire_value(value: Any, depth: int = 0, ancestors: tuple[int, ...] = (), budg
return None


ManifestSection = tuple[str, Callable[[Any], AgentManifest]]


def build_agent_manifest(
framework: str, agent: Any, sections: tuple[ManifestSection, ...], integration_name: str
) -> AgentManifest:
"""Run each section in isolation, merge them, and drop the fields that mean "not configured".

A section that raises costs only its own fields, so a framework change inside one cannot blank
the rest. The result is passed through wire_value so it always survives JSON encoding.
"""
manifest: AgentManifest = {}
for name, section in sections:
try:
manifest.update(section(agent))
except Exception:
log.debug("failed to build %s agent manifest section %s", integration_name, name, exc_info=True)
try:
wired = wire_value(prune_empty(manifest))
except Exception:
log.debug("failed to finalize %s agent manifest", integration_name, exc_info=True)
return {}
if not wired:
return {}
wired["framework"] = framework
return cast(AgentManifest, wired)


def as_str(value: Any) -> str:
"""str-only: the span encoder reprs what it cannot encode, and a repr can carry anything.

A non-string reports "", which prune_empty drops like any other unset value.
"""
return value if isinstance(value, str) else ""


def config_value(value: Any) -> Any:
"""JSON-native form of a declared config value. Pydantic models are dumped; other objects drop."""
if hasattr(value, "model_dump"):
try:
value = value.model_dump(exclude_none=True)
except Exception:
return None
return wire_value(value)


def filter_model_settings(settings: Any) -> dict[str, Any]:
"""Inference params filtered by ALLOWED_MODEL_SETTINGS_KEYS. Non-mappings report nothing."""
if not isinstance(settings, dict):
return {}
allowed: dict[str, Any] = {}
for key, value in settings.items():
if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value):
continue
# prune_empty drops what wire_value could not encode, so assign it either way.
allowed[key] = wire_value(value)
return allowed


def instruction_fields(value: Any, resolver_type: str = "dynamic_instructions") -> AgentManifest:
"""Static text as instructions. A callable's text is only known at run time, so it ships by name.

The callable is never invoked, and str() of it is avoided because its memory address changes
every process, which would report an instruction change on every deploy.
"""
if isinstance(value, str):
return {"instructions": value}
if callable(value):
return {"extra_instructions": [{"type": resolver_type, "name": callable_name(value)}]}
return {}


def normalize_tool(name: Any, description: Any = None, parameters: Any = None) -> Optional[dict[str, Any]]:
"""One tool as {name, description?, parameters?}, the shape every integration emits.

parameters accepts a JSON Schema object or the {param: {type, required}} mapping.
"""
if not isinstance(name, str) or not name:
return None
return {"name": name, "description": as_str(description), "parameters": tool_parameters(parameters)}


def build_manual_agent_manifest(agent: Any) -> AgentManifest:
"""Build the manifest a caller declared through LLMObs.annotate(agent=...).

Expand Down Expand Up @@ -229,21 +312,8 @@ def _manual_model_name(agent: dict[str, Any]) -> AgentManifest:


def _manual_model_settings(agent: dict[str, Any]) -> AgentManifest:
"""Inference params filtered by ALLOWED_MODEL_SETTINGS_KEYS.

Separate from _manual_model_name so a malformed settings dict does not discard a valid model.
"""
fields: AgentManifest = {}
settings = agent.get("model_settings")
if isinstance(settings, dict):
allowed: dict[str, Any] = {}
for key, value in settings.items():
if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value):
continue
# prune_empty drops what wire_value could not encode, so assign it either way.
allowed[key] = wire_value(value)
fields["model_settings"] = allowed
return fields
"""Separate from _manual_model_name so a malformed settings dict does not discard a valid model."""
return {"model_settings": filter_model_settings(agent.get("model_settings"))}


def _manual_tools(agent: dict[str, Any]) -> AgentManifest:
Expand All @@ -265,7 +335,7 @@ def _manual_tools(agent: dict[str, Any]) -> AgentManifest:
{
"name": name,
"description": description if isinstance(description, str) else None,
"parameters": _manual_tool_parameters(tool.get("parameters")),
"parameters": tool_parameters(tool.get("parameters")),
}
)
wired = wire_value(tools)
Expand All @@ -274,7 +344,7 @@ def _manual_tools(agent: dict[str, Any]) -> AgentManifest:
return fields


def _manual_tool_parameters(parameters: Any) -> dict[str, Any]:
def tool_parameters(parameters: Any) -> dict[str, Any]:
"""{param: {type?, required?}}, matching what the framework integrations extract."""
if not isinstance(parameters, dict):
return {}
Expand Down
114 changes: 96 additions & 18 deletions ddtrace/llmobs/_integrations/claude_agent_sdk.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,10 +10,16 @@
from ddtrace.llmobs._constants import INPUT_TOKENS_METRIC_KEY
from ddtrace.llmobs._constants import OUTPUT_TOKENS_METRIC_KEY
from ddtrace.llmobs._constants import TOTAL_TOKENS_METRIC_KEY
from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest
from ddtrace.llmobs._integrations.agent_manifest import callable_name
from ddtrace.llmobs._integrations.agent_manifest import config_value
from ddtrace.llmobs._integrations.agent_manifest import as_str
from ddtrace.llmobs._integrations.base import BaseLLMIntegration
from ddtrace.llmobs._utils import _annotate_llmobs_span_data
from ddtrace.llmobs._utils import _get_attr
from ddtrace.llmobs._utils import safe_json
from ddtrace.llmobs.types import AgentCapability
from ddtrace.llmobs.types import AgentManifest
from ddtrace.llmobs.types import Message
from ddtrace.llmobs.types import ToolCall
from ddtrace.llmobs.types import ToolResult
Expand Down Expand Up @@ -139,7 +145,7 @@ def _llmobs_set_agent_tags(
span.set_tag(ERROR_TYPE, error_type)
span.set_tag(ERROR_MSG, error_message)

agent_manifest = self._build_agent_manifest(model, metadata, init_system_message)
agent_manifest = self._build_agent_manifest(model, kwargs.get("options"), init_system_message)

_annotate_llmobs_span_data(
span,
Expand All @@ -149,25 +155,25 @@ def _llmobs_set_agent_tags(
metadata=metadata,
output_value=output_messages,
metrics=metrics,
agent_manifest=agent_manifest,
agent_manifest=agent_manifest or None,
)

def _build_agent_manifest(
self, model: str, metadata: dict[str, Any], init_system_message: dict[str, Any]
) -> dict[str, Any]:
manifest: dict[str, Any] = {}
manifest["framework"] = "Claude Agent SDK"
if model:
manifest["model"] = model
if init_system_message:
tools = init_system_message.get("tools", []) or []
manifest["tools"] = [{"name": tool} for tool in tools]
if init_system_message:
mcp_servers = init_system_message.get("mcp_servers", []) or []
manifest["dependencies"] = {"mcp_servers": mcp_servers}
if "max_turns" in metadata:
manifest["max_iterations"] = metadata["max_turns"]
return manifest
def _build_agent_manifest(self, model: str, options: Any, init_system_message: dict[str, Any]) -> dict[str, Any]:
declared = {"model": model, "options": options, "init": init_system_message or {}}
manifest = build_agent_manifest(
FRAMEWORK_NAME,
declared,
(
("model", lambda d: {"model": as_str(d["model"])}),
("instructions", lambda d: _manifest_instructions(d["options"])),
("tools", _manifest_tools),
("handoffs", lambda d: _manifest_handoffs(d["options"])),
("guardrails", lambda d: _manifest_guardrails(d["options"])),
("agent_settings", lambda d: _manifest_agent_settings(d["options"])),
),
self._integration_name,
)
return dict(manifest)

def _extract_input_messages(self, prompt: Any, span: Span) -> list[Message]:
prompt_wrapper = span._get_ctx_item("_dd_prompt_wrapper") if span else None
Expand Down Expand Up @@ -479,3 +485,75 @@ def _parse_tok(self, s: str) -> int:
if s.lower().endswith("m"):
return round(float(s[:-1]) * 1_000_000)
return int(float(s))


FRAMEWORK_NAME = "Claude Agent SDK"

_AGENT_SETTINGS_OPTIONS = (
"max_turns",
"max_budget_usd",
"max_thinking_tokens",
"permission_mode",
"allowed_tools",
"disallowed_tools",
)


def _manifest_instructions(options: Any) -> AgentManifest:
system_prompt = getattr(options, "system_prompt", None)
if isinstance(system_prompt, str):
return {"instructions": system_prompt}
if isinstance(system_prompt, dict):
# A preset is Claude Code's own prompt, resolved by the CLI, plus optional appended text.
fields: AgentManifest = {"instructions": as_str(system_prompt.get("append"))}
preset = as_str(system_prompt.get("preset"))
if preset:
fields["extra_instructions"] = [{"type": "preset", "name": preset}]
return fields
return {}


def _manifest_tools(declared: dict[str, Any]) -> AgentManifest:
# The init message lists every tool the session can call, built-ins included; allowed_tools only
# lists the ones that skip the permission prompt, so it is not the tool set.
init = declared["init"]
tools = [{"name": tool} for tool in init.get("tools") or [] if isinstance(tool, str) and tool]
servers = getattr(declared["options"], "mcp_servers", None)
if isinstance(servers, dict):
names = [name for name in servers if isinstance(name, str)]
else:
# The init entries also carry a connection status, which is per run and so not reported.
names = [as_str(_get_attr(server, "name", None)) for server in init.get("mcp_servers") or []]
capabilities: list[AgentCapability] = [{"name": name, "type": "mcp"} for name in names if name]
return {"tools": tools, "capabilities": capabilities}


def _manifest_handoffs(options: Any) -> AgentManifest:
agents = getattr(options, "agents", None)
if not isinstance(agents, dict):
return {}
return {
"handoffs": [
{"agent_name": name, "handoff_description": as_str(getattr(definition, "description", None))}
for name, definition in agents.items()
if isinstance(name, str) and name
]
}


def _manifest_guardrails(options: Any) -> AgentManifest:
"""The permission callback and PreToolUse hooks, which can deny a tool call before it runs."""
guardrails: list[str] = []
can_use_tool = getattr(options, "can_use_tool", None)
if callable(can_use_tool):
guardrails.append(callable_name(can_use_tool))
hooks = getattr(options, "hooks", None)
if isinstance(hooks, dict):
for matcher in hooks.get("PreToolUse") or []:
guardrails.extend(callable_name(fn) for fn in getattr(matcher, "hooks", None) or [] if callable(fn))
return {"guardrails": guardrails}


def _manifest_agent_settings(options: Any) -> AgentManifest:
settings = {key: config_value(getattr(options, key, None)) for key in _AGENT_SETTINGS_OPTIONS}
return {"agent_settings": settings}
Loading
Loading