From 08b38cb9f223bc81befc1ffcd35ac4ff88c91008 Mon Sep 17 00:00:00 2001 From: Max Zhang Date: Thu, 1 Oct 2026 16:56:20 -0400 Subject: [PATCH] feat(llmobs): standardize agent manifest across integrations Move CrewAI, Google ADK, LangGraph, OpenAI Agents and Claude Agent SDK onto shared manifest helpers so every integration emits only AgentManifest keys, with instructions, tools and model populated where the framework declares them. Co-Authored-By: Claude Opus 5.5 --- .claude/skills/llmobs-integrations/SKILL.md | 17 + .../llmobs/_integrations/agent_manifest.py | 104 ++++- .../llmobs/_integrations/claude_agent_sdk.py | 114 +++++- ddtrace/llmobs/_integrations/crewai.py | 170 ++++++--- ddtrace/llmobs/_integrations/google_adk.py | 230 +++++++++-- ddtrace/llmobs/_integrations/langgraph.py | 196 +++++----- ddtrace/llmobs/_integrations/openai_agents.py | 302 ++++++++------- ddtrace/llmobs/_integrations/pydantic_ai.py | 50 +-- ddtrace/llmobs/types.py | 4 +- ...rdize-agent-manifest-df86615c800061b2.yaml | 17 + .../test_claude_agent_sdk_llmobs.py | 4 +- tests/contrib/claude_agent_sdk/utils.py | 7 +- tests/contrib/crewai/test_crewai_llmobs.py | 39 +- .../google_adk/test_google_adk_llmobs.py | 24 +- .../langgraph/test_langgraph_llmobs.py | 36 +- .../test_openai_agents_llmobs.py | 33 +- .../test_agent_manifest_integrations.py | 357 ++++++++++++++++++ 17 files changed, 1229 insertions(+), 475 deletions(-) create mode 100644 releasenotes/notes/llmobs-standardize-agent-manifest-df86615c800061b2.yaml create mode 100644 tests/llmobs/test_agent_manifest_integrations.py diff --git a/.claude/skills/llmobs-integrations/SKILL.md b/.claude/skills/llmobs-integrations/SKILL.md index ecb1570c5de..98e163d9156 100644 --- a/.claude/skills/llmobs-integrations/SKILL.md +++ b/.claude/skills/llmobs-integrations/SKILL.md @@ -109,6 +109,23 @@ Note two already-shipped integrations predate this key: bedrock and the claude-a `metadata["stop_reason"]`. Renaming those is a breaking change and has not been done — follow `finish_reason` for new work. +### Agent Manifest + +Agent spans report the agent's declared configuration as `agent_manifest`, typed by `AgentManifest` +in `ddtrace/llmobs/types.py`. Build it with the helpers in `_integrations/agent_manifest.py`: + +- `build_agent_manifest(framework, agent, sections, integration_name)` runs each section in + isolation, drops unset values, and guarantees a JSON-native result. Return `{}` means "do not + annotate". +- Only emit keys declared on `AgentManifest`. Put loop-level knobs in `agent_settings` and inference + params through `filter_model_settings` (an allowlist; widening it is a security decision). +- Read declared configuration only, never per-run values (session ids, run config, interpolated + templates), so the manifest is identical run to run and version diffs stay meaningful. +- Use `instruction_fields` for instructions: a callable ships by name in `extra_instructions` and is + never called or `str()`-ed (its address changes every process). +- Use `normalize_tool` so every tool is `{name, description?, parameters: {p: {type?, required?}}}`. +- A description other agents see when routing to this one goes in `handoff_description`. + ## Key Constraints - **`submit_to_llmobs=True`** must be set on `LlmRequestEvent` for event-based request spans or passed to `integration.trace()` for direct LLMObs spans diff --git a/ddtrace/llmobs/_integrations/agent_manifest.py b/ddtrace/llmobs/_integrations/agent_manifest.py index f58284680fe..eea9b5c3e2e 100644 --- a/ddtrace/llmobs/_integrations/agent_manifest.py +++ b/ddtrace/llmobs/_integrations/agent_manifest.py @@ -3,6 +3,7 @@ import math import types from typing import Any +from typing import Callable from typing import Optional from typing import TypeVar from typing import Union @@ -172,6 +173,88 @@ def wire_value(value: Any, depth: int = 0, ancestors: tuple[int, ...] = (), budg return None +ManifestSection = tuple[str, Callable[[Any], AgentManifest]] + + +def build_agent_manifest( + framework: str, agent: Any, sections: tuple[ManifestSection, ...], integration_name: str +) -> AgentManifest: + """Run each section in isolation, merge them, and drop the fields that mean "not configured". + + A section that raises costs only its own fields, so a framework change inside one cannot blank + the rest. The result is passed through wire_value so it always survives JSON encoding. + """ + manifest: AgentManifest = {} + for name, section in sections: + try: + manifest.update(section(agent)) + except Exception: + log.debug("failed to build %s agent manifest section %s", integration_name, name, exc_info=True) + try: + wired = wire_value(prune_empty(manifest)) + except Exception: + log.debug("failed to finalize %s agent manifest", integration_name, exc_info=True) + return {} + if not wired: + return {} + wired["framework"] = framework + return cast(AgentManifest, wired) + + +def as_str(value: Any) -> str: + """str-only: the span encoder reprs what it cannot encode, and a repr can carry anything. + + A non-string reports "", which prune_empty drops like any other unset value. + """ + return value if isinstance(value, str) else "" + + +def config_value(value: Any) -> Any: + """JSON-native form of a declared config value. Pydantic models are dumped; other objects drop.""" + if hasattr(value, "model_dump"): + try: + value = value.model_dump(exclude_none=True) + except Exception: + return None + return wire_value(value) + + +def filter_model_settings(settings: Any) -> dict[str, Any]: + """Inference params filtered by ALLOWED_MODEL_SETTINGS_KEYS. Non-mappings report nothing.""" + if not isinstance(settings, dict): + return {} + allowed: dict[str, Any] = {} + for key, value in settings.items(): + if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value): + continue + # prune_empty drops what wire_value could not encode, so assign it either way. + allowed[key] = wire_value(value) + return allowed + + +def instruction_fields(value: Any, resolver_type: str = "dynamic_instructions") -> AgentManifest: + """Static text as instructions. A callable's text is only known at run time, so it ships by name. + + The callable is never invoked, and str() of it is avoided because its memory address changes + every process, which would report an instruction change on every deploy. + """ + if isinstance(value, str): + return {"instructions": value} + if callable(value): + return {"extra_instructions": [{"type": resolver_type, "name": callable_name(value)}]} + return {} + + +def normalize_tool(name: Any, description: Any = None, parameters: Any = None) -> Optional[dict[str, Any]]: + """One tool as {name, description?, parameters?}, the shape every integration emits. + + parameters accepts a JSON Schema object or the {param: {type, required}} mapping. + """ + if not isinstance(name, str) or not name: + return None + return {"name": name, "description": as_str(description), "parameters": tool_parameters(parameters)} + + def build_manual_agent_manifest(agent: Any) -> AgentManifest: """Build the manifest a caller declared through LLMObs.annotate(agent=...). @@ -229,21 +312,8 @@ def _manual_model_name(agent: dict[str, Any]) -> AgentManifest: def _manual_model_settings(agent: dict[str, Any]) -> AgentManifest: - """Inference params filtered by ALLOWED_MODEL_SETTINGS_KEYS. - - Separate from _manual_model_name so a malformed settings dict does not discard a valid model. - """ - fields: AgentManifest = {} - settings = agent.get("model_settings") - if isinstance(settings, dict): - allowed: dict[str, Any] = {} - for key, value in settings.items(): - if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value): - continue - # prune_empty drops what wire_value could not encode, so assign it either way. - allowed[key] = wire_value(value) - fields["model_settings"] = allowed - return fields + """Separate from _manual_model_name so a malformed settings dict does not discard a valid model.""" + return {"model_settings": filter_model_settings(agent.get("model_settings"))} def _manual_tools(agent: dict[str, Any]) -> AgentManifest: @@ -265,7 +335,7 @@ def _manual_tools(agent: dict[str, Any]) -> AgentManifest: { "name": name, "description": description if isinstance(description, str) else None, - "parameters": _manual_tool_parameters(tool.get("parameters")), + "parameters": tool_parameters(tool.get("parameters")), } ) wired = wire_value(tools) @@ -274,7 +344,7 @@ def _manual_tools(agent: dict[str, Any]) -> AgentManifest: return fields -def _manual_tool_parameters(parameters: Any) -> dict[str, Any]: +def tool_parameters(parameters: Any) -> dict[str, Any]: """{param: {type?, required?}}, matching what the framework integrations extract.""" if not isinstance(parameters, dict): return {} diff --git a/ddtrace/llmobs/_integrations/claude_agent_sdk.py b/ddtrace/llmobs/_integrations/claude_agent_sdk.py index bd402a22cd2..ae9d08c6a42 100644 --- a/ddtrace/llmobs/_integrations/claude_agent_sdk.py +++ b/ddtrace/llmobs/_integrations/claude_agent_sdk.py @@ -10,10 +10,16 @@ from ddtrace.llmobs._constants import INPUT_TOKENS_METRIC_KEY from ddtrace.llmobs._constants import OUTPUT_TOKENS_METRIC_KEY from ddtrace.llmobs._constants import TOTAL_TOKENS_METRIC_KEY +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import config_value +from ddtrace.llmobs._integrations.agent_manifest import as_str from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._utils import _annotate_llmobs_span_data from ddtrace.llmobs._utils import _get_attr from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentManifest from ddtrace.llmobs.types import Message from ddtrace.llmobs.types import ToolCall from ddtrace.llmobs.types import ToolResult @@ -139,7 +145,7 @@ def _llmobs_set_agent_tags( span.set_tag(ERROR_TYPE, error_type) span.set_tag(ERROR_MSG, error_message) - agent_manifest = self._build_agent_manifest(model, metadata, init_system_message) + agent_manifest = self._build_agent_manifest(model, kwargs.get("options"), init_system_message) _annotate_llmobs_span_data( span, @@ -149,25 +155,25 @@ def _llmobs_set_agent_tags( metadata=metadata, output_value=output_messages, metrics=metrics, - agent_manifest=agent_manifest, + agent_manifest=agent_manifest or None, ) - def _build_agent_manifest( - self, model: str, metadata: dict[str, Any], init_system_message: dict[str, Any] - ) -> dict[str, Any]: - manifest: dict[str, Any] = {} - manifest["framework"] = "Claude Agent SDK" - if model: - manifest["model"] = model - if init_system_message: - tools = init_system_message.get("tools", []) or [] - manifest["tools"] = [{"name": tool} for tool in tools] - if init_system_message: - mcp_servers = init_system_message.get("mcp_servers", []) or [] - manifest["dependencies"] = {"mcp_servers": mcp_servers} - if "max_turns" in metadata: - manifest["max_iterations"] = metadata["max_turns"] - return manifest + def _build_agent_manifest(self, model: str, options: Any, init_system_message: dict[str, Any]) -> dict[str, Any]: + declared = {"model": model, "options": options, "init": init_system_message or {}} + manifest = build_agent_manifest( + FRAMEWORK_NAME, + declared, + ( + ("model", lambda d: {"model": as_str(d["model"])}), + ("instructions", lambda d: _manifest_instructions(d["options"])), + ("tools", _manifest_tools), + ("handoffs", lambda d: _manifest_handoffs(d["options"])), + ("guardrails", lambda d: _manifest_guardrails(d["options"])), + ("agent_settings", lambda d: _manifest_agent_settings(d["options"])), + ), + self._integration_name, + ) + return dict(manifest) def _extract_input_messages(self, prompt: Any, span: Span) -> list[Message]: prompt_wrapper = span._get_ctx_item("_dd_prompt_wrapper") if span else None @@ -479,3 +485,75 @@ def _parse_tok(self, s: str) -> int: if s.lower().endswith("m"): return round(float(s[:-1]) * 1_000_000) return int(float(s)) + + +FRAMEWORK_NAME = "Claude Agent SDK" + +_AGENT_SETTINGS_OPTIONS = ( + "max_turns", + "max_budget_usd", + "max_thinking_tokens", + "permission_mode", + "allowed_tools", + "disallowed_tools", +) + + +def _manifest_instructions(options: Any) -> AgentManifest: + system_prompt = getattr(options, "system_prompt", None) + if isinstance(system_prompt, str): + return {"instructions": system_prompt} + if isinstance(system_prompt, dict): + # A preset is Claude Code's own prompt, resolved by the CLI, plus optional appended text. + fields: AgentManifest = {"instructions": as_str(system_prompt.get("append"))} + preset = as_str(system_prompt.get("preset")) + if preset: + fields["extra_instructions"] = [{"type": "preset", "name": preset}] + return fields + return {} + + +def _manifest_tools(declared: dict[str, Any]) -> AgentManifest: + # The init message lists every tool the session can call, built-ins included; allowed_tools only + # lists the ones that skip the permission prompt, so it is not the tool set. + init = declared["init"] + tools = [{"name": tool} for tool in init.get("tools") or [] if isinstance(tool, str) and tool] + servers = getattr(declared["options"], "mcp_servers", None) + if isinstance(servers, dict): + names = [name for name in servers if isinstance(name, str)] + else: + # The init entries also carry a connection status, which is per run and so not reported. + names = [as_str(_get_attr(server, "name", None)) for server in init.get("mcp_servers") or []] + capabilities: list[AgentCapability] = [{"name": name, "type": "mcp"} for name in names if name] + return {"tools": tools, "capabilities": capabilities} + + +def _manifest_handoffs(options: Any) -> AgentManifest: + agents = getattr(options, "agents", None) + if not isinstance(agents, dict): + return {} + return { + "handoffs": [ + {"agent_name": name, "handoff_description": as_str(getattr(definition, "description", None))} + for name, definition in agents.items() + if isinstance(name, str) and name + ] + } + + +def _manifest_guardrails(options: Any) -> AgentManifest: + """The permission callback and PreToolUse hooks, which can deny a tool call before it runs.""" + guardrails: list[str] = [] + can_use_tool = getattr(options, "can_use_tool", None) + if callable(can_use_tool): + guardrails.append(callable_name(can_use_tool)) + hooks = getattr(options, "hooks", None) + if isinstance(hooks, dict): + for matcher in hooks.get("PreToolUse") or []: + guardrails.extend(callable_name(fn) for fn in getattr(matcher, "hooks", None) or [] if callable(fn)) + return {"guardrails": guardrails} + + +def _manifest_agent_settings(options: Any) -> AgentManifest: + settings = {key: config_value(getattr(options, key, None)) for key in _AGENT_SETTINGS_OPTIONS} + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/crewai.py b/ddtrace/llmobs/_integrations/crewai.py index 610924a4c25..2878310daaa 100644 --- a/ddtrace/llmobs/_integrations/crewai.py +++ b/ddtrace/llmobs/_integrations/crewai.py @@ -10,12 +10,20 @@ from ddtrace.internal.utils.formats import format_trace_id from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL from ddtrace.llmobs._constants import ROOT_PARENT_ID +from ddtrace.llmobs._integrations.agent_manifest import as_str +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import is_number +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._utils import _annotate_llmobs_span_data from ddtrace.llmobs._utils import _get_nearest_llmobs_ancestor from ddtrace.llmobs._utils import get_llmobs_parent_id from ddtrace.llmobs._utils import get_llmobs_span_links from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentManifest from ddtrace.llmobs.types import _SpanLink from ddtrace.trace import Span from ddtrace.trace import tracer @@ -246,52 +254,23 @@ def _llmobs_set_tags_tool(self, span, args, kwargs, response): def _tag_agent_manifest(self, span, agent): if not agent: return - - manifest = {} - manifest["framework"] = "CrewAI" - manifest["name"] = agent.role if hasattr(agent, "role") and agent.role else "CrewAI Agent" - if hasattr(agent, "goal"): - manifest["goal"] = agent.goal - if hasattr(agent, "backstory"): - manifest["backstory"] = agent.backstory - if hasattr(agent, "llm"): - if hasattr(agent.llm, "model"): - manifest["model"] = agent.llm.model - model_settings = {} - if hasattr(agent.llm, "max_tokens"): - model_settings["max_tokens"] = agent.llm.max_tokens - if hasattr(agent.llm, "temperature"): - model_settings["temperature"] = agent.llm.temperature - if model_settings: - manifest["model_settings"] = model_settings - if hasattr(agent, "allow_delegation"): - manifest["handoffs"] = {"allow_delegation": agent.allow_delegation} - code_execution_permissions = {} - if hasattr(agent, "allow_code_execution"): - manifest["code_execution_permissions"] = {"allow_code_execution": agent.allow_code_execution} - if hasattr(agent, "code_execution_mode"): - manifest["code_execution_permissions"] = {"code_execution_mode": agent.code_execution_mode} - if code_execution_permissions: - manifest["code_execution_permissions"] = code_execution_permissions - if hasattr(agent, "max_iter"): - manifest["max_iterations"] = agent.max_iter - if hasattr(agent, "tools"): - manifest["tools"] = self._get_agent_tools(agent.tools) - - _annotate_llmobs_span_data(span, agent_manifest=manifest) - - def _get_agent_tools(self, tools): - if not tools or not isinstance(tools, list): - return [] - formatted_tools = [] - for tool in tools: - tool_dict = {} - if hasattr(tool, "name"): - tool_dict["name"] = tool.name - if hasattr(tool, "description"): - tool_dict["description"] = tool.description - formatted_tools.append(tool_dict) - return formatted_tools + manifest = build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", _manifest_labels), + ("instructions", _manifest_instructions), + ("model", _manifest_model), + ("tools", _manifest_tools), + ("capabilities", _manifest_capabilities), + ("handoffs", _manifest_handoffs), + ("guardrails", _manifest_guardrails), + ("agent_settings", _manifest_agent_settings), + ), + self._integration_name, + ) + if manifest: + _annotate_llmobs_span_data(span, agent_manifest=dict(manifest)) def _llmobs_set_tags_flow(self, span, args, kwargs, response): inputs = get_argument_value(args, kwargs, 0, "inputs", optional=True) or {} @@ -520,3 +499,102 @@ def _get_crew_id(span, operation): parent_id = span.parent_id return f"crew_{span.trace_id}_{parent_id}" return f"{span.trace_id}" + + +FRAMEWORK_NAME = "CrewAI" + +# CrewAI LLM attribute to its generic model_settings key. stop is left out: CrewAI sets its own stop +# words on the LLM at run time. +_LLM_SETTINGS_KEYS = { + "temperature": "temperature", + "max_tokens": "max_tokens", + "top_p": "top_p", + "seed": "seed", + "presence_penalty": "presence_penalty", + "frequency_penalty": "frequency_penalty", + "timeout": "timeout", +} + + +def _manifest_labels(agent: Any) -> AgentManifest: + role = getattr(agent, "role", None) + return {"name": role if isinstance(role, str) and role else "CrewAI Agent"} + + +def _manifest_instructions(agent: Any) -> AgentManifest: + # CrewAI builds the agent's system prompt from its goal and backstory. + texts = [as_str(getattr(agent, "goal", None)), as_str(getattr(agent, "backstory", None))] + templates = [as_str(getattr(agent, attr, None)) for attr in ("system_template", "prompt_template")] + return { + "instructions": "\n\n".join(text for text in texts if text), + "system_prompts": [template for template in templates if template], + } + + +def _manifest_model(agent: Any) -> AgentManifest: + llm = getattr(agent, "llm", None) + model = llm if isinstance(llm, str) else as_str(getattr(llm, "model", None)) + settings = {key: getattr(llm, attr, None) for attr, key in _LLM_SETTINGS_KEYS.items()} + return {"model": model, "model_settings": filter_model_settings(settings)} + + +def _manifest_tools(agent: Any) -> AgentManifest: + tools: list[dict[str, Any]] = [] + for tool in getattr(agent, "tools", None) or []: + description = as_str(getattr(tool, "description", None)) + entry = normalize_tool( + getattr(tool, "name", None), + _extract_tool_description_field(description) if description else None, + _tool_json_schema(tool), + ) + if entry: + tools.append(entry) + return {"tools": tools} + + +def _tool_json_schema(tool: Any) -> Optional[dict[str, Any]]: + model_json_schema = getattr(getattr(tool, "args_schema", None), "model_json_schema", None) + if not callable(model_json_schema): + return None + try: + schema = model_json_schema() + except Exception: + return None + return schema if isinstance(schema, dict) else None + + +def _manifest_capabilities(agent: Any) -> AgentManifest: + capabilities: list[AgentCapability] = [ + {"name": type(source).__name__, "type": "knowledge"} + for source in getattr(agent, "knowledge_sources", None) or [] + ] + return {"capabilities": capabilities} + + +def _manifest_handoffs(agent: Any) -> AgentManifest: + # CrewAI delegates to any crew member rather than declaring targets, so only the flag is known. + allow_delegation = getattr(agent, "allow_delegation", None) + return {"handoffs": {"allow_delegation": allow_delegation}} if isinstance(allow_delegation, bool) else {} + + +def _manifest_guardrails(agent: Any) -> AgentManifest: + guardrail = getattr(agent, "guardrail", None) + if isinstance(guardrail, str): + return {"guardrails": [guardrail]} + if callable(guardrail): + return {"guardrails": [callable_name(guardrail)]} + return {} + + +def _manifest_agent_settings(agent: Any) -> AgentManifest: + settings: dict[str, Any] = {} + for attr in ("max_iter", "max_rpm", "max_execution_time", "max_retry_limit"): + value = getattr(agent, attr, None) + if is_number(value): + settings[attr] = value + if getattr(agent, "allow_code_execution", None) is True: + settings["allow_code_execution"] = True + settings["code_execution_mode"] = as_str(getattr(agent, "code_execution_mode", None)) + if getattr(agent, "reasoning", None) is True: + settings["reasoning"] = True + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/google_adk.py b/ddtrace/llmobs/_integrations/google_adk.py index 486ab3676a3..435b23194b5 100644 --- a/ddtrace/llmobs/_integrations/google_adk.py +++ b/ddtrace/llmobs/_integrations/google_adk.py @@ -1,3 +1,4 @@ +import inspect from inspect import isfunction from typing import Any from typing import Optional @@ -6,12 +7,21 @@ from ddtrace.internal.constants import COMPONENT from ddtrace.internal.utils import get_argument_value from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL +from ddtrace.llmobs._integrations.agent_manifest import as_str +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import is_number +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.agent_manifest import type_name from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._integrations.google_utils import extract_message_from_part_google_genai from ddtrace.llmobs._integrations.google_utils import extract_messages_from_adk_events from ddtrace.llmobs._utils import _annotate_llmobs_span_data -from ddtrace.llmobs._utils import _get_attr from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentManifest from ddtrace.trace import Span @@ -135,23 +145,23 @@ def _llmobs_set_tags_tool( def _tag_agent_manifest(self, span: Span, kwargs: dict[str, Any], agent: Any) -> None: if not agent: return - - manifest: dict[str, Any] = {} - - manifest["framework"] = "Google ADK" - manifest["name"] = getattr(agent, "name", "") - manifest["model"] = getattr(getattr(agent, "model", ""), "model", "") - manifest["description"] = getattr(agent, "description", "") - manifest["instructions"] = getattr(agent, "instruction", "") - manifest["model_configuration"] = safe_json(getattr(agent, "model_config", "")) - manifest["session_management"] = { - "session_id": kwargs.get("session_id", ""), - "user_id": kwargs.get("user_id", ""), - "app_name": kwargs.get("app_name", ""), - } - manifest["tools"] = self._get_agent_tools(getattr(agent, "tools", [])) - - _annotate_llmobs_span_data(span, agent_manifest=manifest) + manifest = build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", _manifest_labels), + ("instructions", _manifest_instructions), + ("model", _manifest_model), + ("tools", _manifest_tools), + ("data_contracts", _manifest_data_contracts), + ("handoffs", _manifest_handoffs), + ("guardrails", _manifest_guardrails), + ("agent_settings", _manifest_agent_settings), + ), + self._integration_name, + ) + if manifest: + _annotate_llmobs_span_data(span, agent_manifest=dict(manifest)) def _llmobs_set_tags_code_execute( self, span: Span, args: list[Any], kwargs: dict[str, Any], response: Optional[Any] = None @@ -172,18 +182,172 @@ def _llmobs_set_tags_code_execute( output_value=output, ) - def _get_agent_tools(self, tools): - if not tools or not isinstance(tools, list): - return [] - - agent_tools = [] - for tool in tools: - if isfunction(tool): - tool_name = tool.__name__ - tool_description = tool.__doc__ or "" - else: - tool_name = _get_attr(tool, "name", "Agent Tool") - tool_description = _get_attr(tool, "description", "") - agent_tools.append({"name": tool_name, "description": tool_description}) - - return agent_tools + +FRAMEWORK_NAME = "Google ADK" + +# GenerateContentConfig field to its generic model_settings key. +_GENERATE_CONTENT_CONFIG_KEYS = { + "temperature": "temperature", + "top_p": "top_p", + "top_k": "top_k", + "max_output_tokens": "max_tokens", + "stop_sequences": "stop_sequences", + "seed": "seed", + "presence_penalty": "presence_penalty", + "frequency_penalty": "frequency_penalty", +} + +# JSON Schema names, so a signature-derived tool reads the same as one from a schema. +_JSON_SCHEMA_TYPES = { + "str": "string", + "int": "integer", + "float": "number", + "bool": "boolean", + "list": "array", + "dict": "object", +} + +# Framework-injected arguments, not ones the model fills in. +_IGNORED_TOOL_PARAMETERS = frozenset({"self", "cls", "tool_context"}) + + +def _manifest_labels(agent: Any) -> AgentManifest: + # ADK shows a sub-agent's description to its parent's model when choosing a transfer target, + # which is the role handoff_description plays in the other integrations. + return { + "name": as_str(getattr(agent, "name", None)), + "handoff_description": as_str(getattr(agent, "description", None)), + } + + +def _manifest_instructions(agent: Any) -> AgentManifest: + fields = instruction_fields(getattr(agent, "instruction", None)) + system_prompts: list[str] = [] + global_instruction = getattr(agent, "global_instruction", None) + if isinstance(global_instruction, str) and global_instruction: + system_prompts.append(global_instruction) + elif callable(global_instruction): + fields["extra_instructions"] = fields.get("extra_instructions", []) + [ + {"type": "dynamic_global_instruction", "name": callable_name(global_instruction)} + ] + static_text = _content_text(getattr(agent, "static_instruction", None)) + if static_text: + system_prompts.append(static_text) + fields["system_prompts"] = system_prompts + return fields + + +def _content_text(content: Any) -> str: + """Text of a genai ContentUnion: a str, a Part, a Content, or a list of those.""" + if isinstance(content, str): + return content + if isinstance(content, list): + texts = [_content_text(item) for item in content] + return "\n".join(text for text in texts if text) + parts = getattr(content, "parts", None) + if isinstance(parts, list): + return _content_text(parts) + return as_str(getattr(content, "text", None)) + + +def _manifest_model(agent: Any) -> AgentManifest: + model = getattr(agent, "model", None) + model_name = model if isinstance(model, str) else as_str(getattr(model, "model", None)) + config = getattr(agent, "generate_content_config", None) + settings = {key: getattr(config, field, None) for field, key in _GENERATE_CONTENT_CONFIG_KEYS.items()} + return {"model": model_name, "model_settings": filter_model_settings(settings)} + + +def _manifest_tools(agent: Any) -> AgentManifest: + tools: list[dict[str, Any]] = [] + capabilities: list[AgentCapability] = [] + for tool in getattr(agent, "tools", None) or []: + if isfunction(tool): + entry = normalize_tool(tool.__name__, tool.__doc__, _function_parameters(tool)) + elif hasattr(tool, "get_tools") and not hasattr(tool, "name"): + # A toolset resolves its tools at run time, so only its presence is declared. + kind = "mcp" if "mcp" in type(tool).__name__.lower() else "toolset" + capabilities.append({"name": type(tool).__name__, "type": kind}) + continue + else: + func = getattr(tool, "func", None) + entry = normalize_tool( + getattr(tool, "name", None), + getattr(tool, "description", None), + _function_parameters(func) if callable(func) else None, + ) + if entry: + tools.append(entry) + return {"tools": tools, "capabilities": capabilities} + + +def _function_parameters(fn: Any) -> dict[str, Any]: + """{param: {type?, required?}} read off the signature. The function is never called.""" + try: + signature = inspect.signature(fn) + except (TypeError, ValueError): + return {} + parameters: dict[str, Any] = {} + for name, param in signature.parameters.items(): + if name in _IGNORED_TOOL_PARAMETERS or param.kind in (param.VAR_POSITIONAL, param.VAR_KEYWORD): + continue + spec: dict[str, Any] = {} + if param.annotation is not param.empty: + annotation = param.annotation + # A string annotation comes from a module using postponed evaluation. + annotation_name = annotation if isinstance(annotation, str) else type_name(annotation) + spec["type"] = _JSON_SCHEMA_TYPES.get(annotation_name, annotation_name) + if param.default is param.empty: + spec["required"] = True + parameters[name] = spec + return parameters + + +def _manifest_data_contracts(agent: Any) -> AgentManifest: + contracts: dict[str, Any] = {} + for key, attr in (("input", "input_schema"), ("output", "output_schema")): + schema = getattr(agent, attr, None) + if isinstance(schema, type): + contracts[key] = {"name": type_name(schema)} + return {"data_contracts": contracts} + + +def _manifest_handoffs(agent: Any) -> AgentManifest: + handoffs: list[dict[str, Any]] = [] + for sub_agent in getattr(agent, "sub_agents", None) or []: + name = as_str(getattr(sub_agent, "name", None)) + if name: + handoffs.append( + {"agent_name": name, "handoff_description": as_str(getattr(sub_agent, "description", None))} + ) + return {"handoffs": handoffs} + + +def _manifest_guardrails(agent: Any) -> AgentManifest: + """Callbacks that can veto a model or tool call before it runs, which is how ADK does guardrails.""" + guardrails: list[str] = [] + for attr in ("before_model_callback", "before_tool_callback"): + callbacks = getattr(agent, attr, None) + for callback in callbacks if isinstance(callbacks, list) else [callbacks]: + if callable(callback): + guardrails.append(callable_name(callback)) + return {"guardrails": guardrails} + + +def _manifest_agent_settings(agent: Any) -> AgentManifest: + settings: dict[str, Any] = {} + for attr in ("disallow_transfer_to_parent", "disallow_transfer_to_peers"): + if getattr(agent, attr, None) is True: + settings[attr] = True + include_contents = getattr(agent, "include_contents", None) + if isinstance(include_contents, str) and include_contents != "default": + settings["include_contents"] = include_contents + settings["output_key"] = as_str(getattr(agent, "output_key", None)) + max_iterations = getattr(agent, "max_iterations", None) + if is_number(max_iterations): + settings["max_iterations"] = max_iterations + for attr in ("planner", "code_executor"): + value = getattr(agent, attr, None) + if value is not None: + settings[attr] = type(value).__name__ + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/langgraph.py b/ddtrace/llmobs/_integrations/langgraph.py index 226ec96d1a9..83c66500aad 100644 --- a/ddtrace/llmobs/_integrations/langgraph.py +++ b/ddtrace/llmobs/_integrations/langgraph.py @@ -10,6 +10,14 @@ from ddtrace.internal.utils.formats import format_trace_id from ddtrace.llmobs import LLMObs from ddtrace.llmobs._constants import ROOT_PARENT_ID +from ddtrace.llmobs._integrations.agent_manifest import ALLOWED_MODEL_SETTINGS_KEYS +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import is_number +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.agent_manifest import as_str from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._integrations.constants import LANGGRAPH_ASTREAM_OUTPUT from ddtrace.llmobs._integrations.utils import format_langchain_io @@ -18,6 +26,7 @@ from ddtrace.llmobs._utils import _get_nearest_llmobs_ancestor from ddtrace.llmobs._utils import get_llmobs_parent_id from ddtrace.llmobs._utils import get_llmobs_span_links +from ddtrace.llmobs.types import AgentManifest from ddtrace.llmobs.types import _SpanLink from ddtrace.trace import Span @@ -28,25 +37,13 @@ PREGEL_PUSH = "__pregel_push" # represents a task queued up by a `Send` command PREGEL_TASKS = "__pregel_tasks" # name of ephemeral channel that pregel `Send` commands write to -ALLOWED_MODEL_SETTINGS_KEYS = [ - "max_tokens", - "temperature", - "top_p", - "top_k", - "frequency_penalty", - "presence_penalty", - "stop", - "n", - "logprobs", - "echo", - "logit_bias", -] +FRAMEWORK_NAME = "LangGraph" class LangGraphIntegration(BaseLLMIntegration): _integration_name = "langgraph" _graph_nodes_for_graph_by_task_id: WeakKeyDictionary[Span, dict[str, Any]] = WeakKeyDictionary() - _agent_manifests: WeakKeyDictionary[Any, dict[str, Any]] = WeakKeyDictionary() + _agent_manifests: WeakKeyDictionary[Any, AgentManifest] = WeakKeyDictionary() _graph_spans_to_graph_instances: WeakKeyDictionary[Span, Any] = WeakKeyDictionary() def trace( @@ -126,26 +123,22 @@ def _get_agent_manifest(self, agent, args, config: dict[str, Any]) -> Optional[d if agent is None: return None - agent_manifest = self._agent_manifests.get(agent) - if agent_manifest is None: - tools = _get_tools_from_graph(agent) - agent_manifest = {"name": agent.name or "LangGraph", "tools": tools} - self._agent_manifests[agent] = agent_manifest - - if "framework" not in agent_manifest: - agent_manifest["framework"] = "LangGraph" - if "max_iterations" not in agent_manifest: - agent_manifest["max_iterations"] = _get_attr(config, "recursion_limit", 25) - - if ( - "dependencies" not in agent_manifest - and isinstance(args, tuple) - and len(args) > 0 - and isinstance(args[0], dict) - ): - agent_manifest["dependencies"] = list(args[0].keys()) + declared = self._agent_manifests.get(agent) + if declared is None: + declared = build_agent_manifest( + FRAMEWORK_NAME, + agent, + (("labels", _manifest_labels), ("tools", _manifest_graph_tools)), + self._integration_name, + ) + self._agent_manifests[agent] = declared - return agent_manifest + # Copied so a run's config never leaks into the manifest cached for the next run. + manifest: dict[str, Any] = dict(declared) + recursion_limit = _get_attr(config, "recursion_limit", None) + if is_number(recursion_limit): + manifest["agent_settings"] = {**manifest.get("agent_settings", {}), "recursion_limit": recursion_limit} + return manifest def _get_node_metadata_from_span(self, span: Span, instance_id: str) -> dict[str, Any]: """ @@ -170,35 +163,24 @@ def llmobs_handle_agent_manifest(self, agent, args: tuple, kwargs: dict): if not self.llmobs_enabled: return - model = get_argument_value( - args, kwargs, 0, "model", True - ) # required parameter on the langgraph side, but optional should that ever change - model_name, model_provider, model_settings = _get_model_info(model) - - agent_tools: list[Any] = ( - get_argument_value(args, kwargs, 1, "tools", True) or [] - ) # required parameter on the langgraph side, but optional should that ever change - tools = _get_tools_from_react_agent(agent_tools) - - system_prompt: Optional[str] = _get_system_prompt_from_react_agent(kwargs.get("prompt")) - name: Optional[str] = kwargs.get("name") - - agent_manifest: dict[str, Any] = {} - - if model_name: - agent_manifest["model"] = model_name - if model_provider: - agent_manifest["model_provider"] = model_provider - if model_settings: - agent_manifest["model_settings"] = model_settings - if tools: - agent_manifest["tools"] = tools - if system_prompt: - agent_manifest["instructions"] = system_prompt - if name: - agent_manifest["name"] = name - - self._agent_manifests[agent] = agent_manifest + # model and tools are required parameters on the langgraph side, but optional should that ever change. + declared = { + "name": kwargs.get("name"), + "model": get_argument_value(args, kwargs, 0, "model", True), + "tools": get_argument_value(args, kwargs, 1, "tools", True) or [], + "prompt": kwargs.get("prompt"), + } + self._agent_manifests[agent] = build_agent_manifest( + FRAMEWORK_NAME, + declared, + ( + ("labels", lambda d: {"name": as_str(d["name"]) or "LangGraph"}), + ("model", lambda d: _manifest_model(d["model"])), + ("tools", lambda d: {"tools": _get_tools_from_react_agent(d["tools"]) or []}), + ("instructions", lambda d: _manifest_react_prompt(d["prompt"])), + ), + self._integration_name, + ) def llmobs_handle_pregel_loop_tick( self, finished_tasks: dict, next_tasks: dict, more_tasks: bool, is_subgraph_node: bool = False @@ -333,17 +315,24 @@ def _link_standalone_terminal_tasks( _annotate_llmobs_span_data(graph_span, span_links=graph_span_links) -def _get_model_info(model) -> tuple[Optional[str], Optional[str], dict[str, Any]]: - """Get the model name, provider, and settings from a langchain llm""" - if isinstance(model, str): - # something like "openai:gpt-4" - model_provider_str, model_name_str = model.split(":", maxsplit=1) - return model_name_str, model_provider_str, {} +def _manifest_labels(agent: Any) -> AgentManifest: + return {"name": as_str(_get_attr(agent, "name", None)) or "LangGraph"} + + +def _manifest_graph_tools(agent: Any) -> AgentManifest: + return {"tools": _get_tools_from_graph(agent)} - model_name = _get_attr(model, "model_name", None) - model_provider = _get_model_provider(model) - model_settings = _get_model_settings(model) - return model_name, model_provider, model_settings + +def _manifest_model(model: Any) -> AgentManifest: + """The model name, provider and settings from a langchain chat model or a "provider:model" string.""" + if isinstance(model, str): + provider, sep, name = model.partition(":") + return {"model": name, "model_provider": provider} if sep else {"model": model} + return { + "model": as_str(_get_attr(model, "model_name", None)), + "model_provider": as_str(_get_model_provider(model)), + "model_settings": _get_model_settings(model), + } def _get_model_provider(model) -> Optional[str]: @@ -358,32 +347,20 @@ def _get_model_provider(model) -> Optional[str]: def _get_model_settings(model) -> dict[str, Any]: """Get the model settings from a langchain llm""" - model_dict_fn = _get_attr(model, "dict", None) - if model_dict_fn is None or not callable(model_dict_fn): - return {} - - model_dict: dict = model.dict() - return {key: value for key, value in model_dict.items() if key in ALLOWED_MODEL_SETTINGS_KEYS and value} - - -def _get_system_prompt_from_react_agent(system_prompt) -> Optional[str]: - """ - Get the system prompt from a react agent. - - The system prompt can be: - - a string - - a dict with a "content" key - - a Callable that returns a string or dict - - In the case of a Callable (which is dynamic as a function of state and config), we end up returning None. - """ - if system_prompt is None: - return None + settings = {key: _get_attr(model, key, None) for key in ALLOWED_MODEL_SETTINGS_KEYS} + settings["stop_sequences"] = _get_attr(model, "stop", None) + return filter_model_settings(settings) - if isinstance(system_prompt, str): - return system_prompt - return _get_attr(system_prompt, "content", None) +def _manifest_react_prompt(prompt: Any) -> AgentManifest: + """A react agent's prompt: a string, a SystemMessage, or a callable resolved per run.""" + if prompt is None or isinstance(prompt, str): + return instruction_fields(prompt) + content = _get_attr(prompt, "content", None) + if isinstance(content, str): + return {"instructions": content} + # A Runnable or callable prompt decides the text from the run's state. + return {"extra_instructions": [{"type": "dynamic_prompt", "name": callable_name(prompt)}]} def _get_tools_from_react_agent(tools: Any) -> Optional[list[dict[str, Any]]]: @@ -409,12 +386,27 @@ def _get_tool_repr_from_langchain_base_tool(tool) -> Optional[dict[str, Any]]: """Get the tool representation from a langchain base tool""" if tool is None or isinstance(tool, dict): return None - - return { - "name": _get_attr(tool, "name", ""), - "description": _get_attr(tool, "description", ""), - "parameters": _get_attr(tool, "args", {}), - } + if _get_attr(tool, "name", None) is None and callable(tool): + # A plain function, which langgraph wraps into a tool itself. + return normalize_tool(callable_name(tool), getattr(tool, "__doc__", None)) + return normalize_tool(_get_attr(tool, "name", None), _get_attr(tool, "description", None), _tool_json_schema(tool)) + + +def _tool_json_schema(tool) -> Optional[dict[str, Any]]: + """The tool's JSON Schema, which unlike tool.args also says which parameters are required.""" + args_schema = _get_attr(tool, "args_schema", None) + if isinstance(args_schema, dict): + return args_schema + model_json_schema = getattr(args_schema, "model_json_schema", None) + if callable(model_json_schema): + try: + schema = model_json_schema() + except Exception: + schema = None + if isinstance(schema, dict): + return schema + args = _get_attr(tool, "args", None) + return {"type": "object", "properties": args} if isinstance(args, dict) else None def _is_tool_node(maybe_tool_node): diff --git a/ddtrace/llmobs/_integrations/openai_agents.py b/ddtrace/llmobs/_integrations/openai_agents.py index 69a51a6989a..75c40b47c33 100644 --- a/ddtrace/llmobs/_integrations/openai_agents.py +++ b/ddtrace/llmobs/_integrations/openai_agents.py @@ -2,6 +2,7 @@ from typing import Any from typing import Optional from typing import Union +from typing import get_origin import weakref from ddtrace.internal import core @@ -14,6 +15,15 @@ from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL_OUTPUT_USED from ddtrace.llmobs._constants import OAI_HANDOFF_TOOL_ARG from ddtrace.llmobs._constants import ROOT_PARENT_ID +from ddtrace.llmobs._integrations.agent_manifest import ALLOWED_MODEL_SETTINGS_KEYS +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import config_value +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.agent_manifest import as_str +from ddtrace.llmobs._integrations.agent_manifest import type_name from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._integrations.utils import LLMObsTraceInfo from ddtrace.llmobs._integrations.utils import OaiSpanAdapter @@ -23,8 +33,10 @@ from ddtrace.llmobs._utils import get_llmobs_parent_id from ddtrace.llmobs._utils import get_llmobs_span_name from ddtrace.llmobs._utils import get_tool_version_from_llm_span -from ddtrace.llmobs._utils import load_data_value from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentInstructionResolver +from ddtrace.llmobs.types import AgentManifest from ddtrace.trace import Span @@ -325,140 +337,158 @@ def tag_agent_manifest(self, span: Span, args: list[Any], kwargs: dict[str, Any] def _tag_agent_manifest_from_agent(self, span: Span, agent: Any) -> None: if not agent or not self.llmobs_enabled: return + manifest = build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", _manifest_labels), + ("instructions", _manifest_instructions), + ("model", _manifest_model), + ("tools", _manifest_tools), + ("capabilities", _manifest_capabilities), + ("data_contracts", _manifest_data_contracts), + ("handoffs", _manifest_handoffs), + ("guardrails", _manifest_guardrails), + ("agent_settings", _manifest_agent_settings), + ), + self._integration_name, + ) + if manifest: + _annotate_llmobs_span_data(span, agent_manifest=dict(manifest)) + + +FRAMEWORK_NAME = "OpenAI" + +# Declared config worth diffing on each hosted tool. Everything else on these tools is a callable, +# a client object, or a credential (HostedMCPTool.tool_config carries headers and authorization). +_HOSTED_TOOL_FIELDS = { + "file_search": ("vector_store_ids", "max_num_results", "include_search_results", "ranking_options", "filters"), + "web_search": ("user_location", "search_context_size", "filters"), + "web_search_preview": ("user_location", "search_context_size", "filters"), +} +_HOSTED_MCP_CONFIG_KEYS = ("server_label", "allowed_tools", "require_approval") + + +def _manifest_labels(agent: Any) -> AgentManifest: + return { + "name": as_str(getattr(agent, "name", None)), + "handoff_description": as_str(getattr(agent, "handoff_description", None)), + } + + +def _manifest_instructions(agent: Any) -> AgentManifest: + fields = instruction_fields(getattr(agent, "instructions", None)) + prompt = getattr(agent, "prompt", None) + # A stored prompt is resolved by the Responses API at run time, so it is recorded by id. + resolver: Optional[AgentInstructionResolver] = None + if isinstance(prompt, dict) and isinstance(prompt.get("id"), str): + resolver = {"type": "prompt", "name": prompt["id"]} + elif callable(prompt): + resolver = {"type": "dynamic_prompt", "name": callable_name(prompt)} + if resolver: + fields["extra_instructions"] = fields.get("extra_instructions", []) + [resolver] + return fields + + +def _manifest_model(agent: Any) -> AgentManifest: + model = getattr(agent, "model", None) + model_name = model if isinstance(model, str) else as_str(getattr(model, "model", None)) + settings = getattr(agent, "model_settings", None) + # Read field by field rather than dumped, so extra_headers and extra_body are never touched. + if settings is not None and not isinstance(settings, dict): + settings = {key: getattr(settings, key, None) for key in ALLOWED_MODEL_SETTINGS_KEYS} + return {"model": model_name, "model_settings": filter_model_settings(settings)} + + +def _manifest_tools(agent: Any) -> AgentManifest: + tools: list[dict[str, Any]] = [] + for tool in getattr(agent, "tools", None) or []: + name = getattr(tool, "name", None) + if hasattr(tool, "params_json_schema"): + entry = normalize_tool(name, getattr(tool, "description", None), tool.params_json_schema) + else: + entry = _hosted_tool(tool, name) + if entry: + tools.append(entry) + return {"tools": tools} - manifest = {} - manifest["framework"] = "OpenAI" - if hasattr(agent, "name"): - manifest["name"] = agent.name - if hasattr(agent, "instructions"): - manifest["instructions"] = agent.instructions - if hasattr(agent, "handoff_description"): - manifest["handoff_description"] = agent.handoff_description - if hasattr(agent, "model"): - model = agent.model - manifest["model"] = model if isinstance(model, str) else getattr(model, "model", "") - - model_settings = self._extract_model_settings_from_agent(agent) - if model_settings: - manifest["model_settings"] = model_settings - - tools = self._extract_tools_from_agent(agent) - if tools: - manifest["tools"] = tools - - handoffs = self._extract_handoffs_from_agent(agent) - if handoffs: - manifest["handoffs"] = handoffs - - guardrails = self._extract_guardrails_from_agent(agent) - if guardrails: - manifest["guardrails"] = guardrails - - _annotate_llmobs_span_data(span, agent_manifest=manifest) - - def _extract_model_settings_from_agent(self, agent): - if not hasattr(agent, "model_settings"): - return None - - # convert model_settings to dict if it's not already - model_settings = agent.model_settings - if not isinstance(model_settings, dict): - model_settings = getattr(model_settings, "__dict__", None) - - return load_data_value(model_settings) - - def _extract_tools_from_agent(self, agent): - if not hasattr(agent, "tools") or not agent.tools: - return None - - tools = [] - for tool in agent.tools: - tool_dict = {} - tool_name = getattr(tool, "name", None) - if tool_name: - tool_dict["name"] = tool_name - if tool_name == "web_search_preview": - if hasattr(tool, "user_location"): - tool_dict["user_location"] = tool.user_location - if hasattr(tool, "search_context_size"): - tool_dict["search_context_size"] = tool.search_context_size - elif tool_name == "file_search": - if hasattr(tool, "vector_store_ids"): - tool_dict["vector_store_ids"] = tool.vector_store_ids - if hasattr(tool, "max_num_results"): - tool_dict["max_num_results"] = tool.max_num_results - if hasattr(tool, "include_search_results"): - tool_dict["include_search_results"] = tool.include_search_results - if hasattr(tool, "ranking_options"): - tool_dict["ranking_options"] = tool.ranking_options - if hasattr(tool, "filters"): - tool_dict["filters"] = tool.filters - elif tool_name == "computer_use_preview": - if hasattr(tool, "computer"): - tool_dict["computer"] = tool.computer - if hasattr(tool, "on_safety_check"): - tool_dict["on_safety_check"] = tool.on_safety_check - elif tool_name == "code_interpreter": - if hasattr(tool, "tool_config"): - tool_dict["tool_config"] = tool.tool_config - elif tool_name == "hosted_mcp": - if hasattr(tool, "tool_config"): - tool_dict["tool_config"] = tool.tool_config - if hasattr(tool, "on_approval_request"): - tool_dict["on_approval_request"] = tool.on_approval_request - elif tool_name == "image_generation": - if hasattr(tool, "tool_config"): - tool_dict["tool_config"] = tool.tool_config - elif tool_name == "local_shell": - if hasattr(tool, "executor"): - tool_dict["executor"] = tool.executor - else: - if hasattr(tool, "description"): - tool_dict["description"] = tool.description - if hasattr(tool, "strict_json_schema"): - tool_dict["strict_json_schema"] = tool.strict_json_schema - if hasattr(tool, "params_json_schema"): - parameter_schema = tool.params_json_schema - required_params = {param: True for param in parameter_schema.get("required", [])} - parameters = {} - for param, schema in parameter_schema.get("properties", {}).items(): - param_dict = {} - if "type" in schema: - param_dict["type"] = schema["type"] - if "title" in schema: - param_dict["title"] = schema["title"] - if param in required_params: - param_dict["required"] = True - parameters[param] = param_dict - tool_dict["parameters"] = parameters - tools.append(tool_dict) - - return tools - - def _extract_handoffs_from_agent(self, agent): - if not hasattr(agent, "handoffs") or not agent.handoffs: - return None - handoffs = [] - for handoff in agent.handoffs: - handoff_dict = {} - if hasattr(handoff, "handoff_description") or hasattr(handoff, "tool_description"): - handoff_dict["handoff_description"] = getattr(handoff, "handoff_description", None) or getattr( - handoff, "tool_description", None - ) - if hasattr(handoff, "name") or hasattr(handoff, "agent_name"): - handoff_dict["agent_name"] = getattr(handoff, "name", None) or getattr(handoff, "agent_name", None) - if hasattr(handoff, "tool_name"): - handoff_dict["tool_name"] = handoff.tool_name - if handoff_dict: - handoffs.append(handoff_dict) - - return handoffs - - def _extract_guardrails_from_agent(self, agent): - guardrails = [] - if hasattr(agent, "input_guardrails"): - guardrails.extend([getattr(guardrail, "name", "") for guardrail in agent.input_guardrails]) - if hasattr(agent, "output_guardrails"): - guardrails.extend([getattr(guardrail, "name", "") for guardrail in agent.output_guardrails]) - return guardrails +def _hosted_tool(tool: Any, name: Any) -> Optional[dict[str, Any]]: + if not isinstance(name, str) or not name: + return None + entry: dict[str, Any] = {"name": name} + for field in _HOSTED_TOOL_FIELDS.get(name, ()): + entry[field] = config_value(getattr(tool, field, None)) + if name == "hosted_mcp": + tool_config = getattr(tool, "tool_config", None) + if isinstance(tool_config, dict): + for key in _HOSTED_MCP_CONFIG_KEYS: + entry[key] = config_value(tool_config.get(key)) + return entry + + +def _manifest_capabilities(agent: Any) -> AgentManifest: + capabilities: list[AgentCapability] = [] + for server in getattr(agent, "mcp_servers", None) or []: + name = as_str(getattr(server, "name", None)) + if name: + capabilities.append({"name": name, "type": "mcp"}) + return {"capabilities": capabilities} + + +def _manifest_data_contracts(agent: Any) -> AgentManifest: + output_type = getattr(agent, "output_type", None) + if output_type is None or output_type is str: + return {} + if isinstance(output_type, type) or get_origin(output_type) is not None: + name = type_name(output_type) + else: + # An AgentOutputSchemaBase wraps the declared type; its class name is all that is stable. + name = type(output_type).__name__ + return {"data_contracts": {"output": {"name": name}}} + + +def _manifest_handoffs(agent: Any) -> AgentManifest: + handoffs: list[dict[str, Any]] = [] + for handoff in getattr(agent, "handoffs", None) or []: + # A bare Agent describes itself through handoff_description; a Handoff carries the tool + # description the calling model actually sees. + entry = { + "agent_name": as_str(getattr(handoff, "agent_name", None) or getattr(handoff, "name", None)), + "tool_name": as_str(getattr(handoff, "tool_name", None)), + "handoff_description": as_str( + getattr(handoff, "handoff_description", None) or getattr(handoff, "tool_description", None) + ), + } + if entry["agent_name"]: + handoffs.append(entry) + return {"handoffs": handoffs} + + +def _manifest_guardrails(agent: Any) -> AgentManifest: + guardrails: list[str] = [] + for guardrail in list(getattr(agent, "input_guardrails", None) or []) + list( + getattr(agent, "output_guardrails", None) or [] + ): + name = getattr(guardrail, "name", None) + fn = getattr(guardrail, "guardrail_function", None) + if not name and callable(fn): + name = callable_name(fn) + if isinstance(name, str) and name: + guardrails.append(name) + return {"guardrails": guardrails} + + +def _manifest_agent_settings(agent: Any) -> AgentManifest: + settings: dict[str, Any] = {} + behavior = getattr(agent, "tool_use_behavior", None) + if isinstance(behavior, str): + settings["tool_use_behavior"] = behavior + elif isinstance(behavior, dict): + settings["tool_use_behavior"] = config_value(behavior) + elif callable(behavior): + settings["tool_use_behavior"] = callable_name(behavior) + reset_tool_choice = getattr(agent, "reset_tool_choice", None) + if isinstance(reset_tool_choice, bool): + settings["reset_tool_choice"] = reset_tool_choice + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/pydantic_ai.py b/ddtrace/llmobs/_integrations/pydantic_ai.py index 360516d799e..34610db7f81 100644 --- a/ddtrace/llmobs/_integrations/pydantic_ai.py +++ b/ddtrace/llmobs/_integrations/pydantic_ai.py @@ -8,11 +8,10 @@ from ddtrace.internal.logger import get_logger from ddtrace.internal.utils import get_argument_value from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL -from ddtrace.llmobs._integrations.agent_manifest import ALLOWED_MODEL_SETTINGS_KEYS +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest from ddtrace.llmobs._integrations.agent_manifest import callable_name -from ddtrace.llmobs._integrations.agent_manifest import is_flat_scalar_value +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings from ddtrace.llmobs._integrations.agent_manifest import is_number -from ddtrace.llmobs._integrations.agent_manifest import prune_empty from ddtrace.llmobs._integrations.agent_manifest import type_name from ddtrace.llmobs._integrations.agent_manifest import wire_value from ddtrace.llmobs._integrations.base import BaseLLMIntegration @@ -270,28 +269,25 @@ def _build_agent_manifest(self, agent: Any) -> AgentManifest: declared configuration is read, so the manifest is identical run to run, and a field pydantic-ai does not expose is omitted rather than invented. """ - manifest: AgentManifest = {} - for name, section in ( - ("labels", self._manifest_labels), - ("instructions", self._manifest_instructions), - ("model", self._manifest_model), - ("capabilities", self._manifest_capabilities), - ("data_contracts", self._manifest_data_contracts), - ("memory_policies", self._manifest_memory_policies), - ("guardrails", self._manifest_guardrails), - ("agent_settings", self._manifest_agent_settings), - ): - try: - manifest.update(section(agent)) - except Exception: - log.debug("failed to build pydantic_ai agent manifest section %s", name, exc_info=True) - # Sections assign unconditionally so mypy can check every key name against the type; one - # prune here is what drops the fields that mean "not configured". - return prune_empty(manifest) + return build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", self._manifest_labels), + ("instructions", self._manifest_instructions), + ("model", self._manifest_model), + ("capabilities", self._manifest_capabilities), + ("data_contracts", self._manifest_data_contracts), + ("memory_policies", self._manifest_memory_policies), + ("guardrails", self._manifest_guardrails), + ("agent_settings", self._manifest_agent_settings), + ), + self._integration_name, + ) def _manifest_labels(self, agent: Any) -> AgentManifest: """Labels that name the agent. Grouped for failure isolation only; the manifest is flat.""" - fields: AgentManifest = {"framework": FRAMEWORK_NAME} + fields: AgentManifest = {} # placeholder per review, matching the span name fallback. Two unnamed agents # therefore share it, so name is not an identity. agent_name = getattr(agent, "name", None) @@ -336,15 +332,7 @@ def _manifest_model(self, agent: Any) -> AgentManifest: # reprs what it cannot encode, which can carry a connection string. if isinstance(model_name, str): fields["model"] = model_name - settings = getattr(agent, "model_settings", None) - if isinstance(settings, dict): - allowed: dict[str, Any] = {} - for key, value in settings.items(): - if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value): - continue - # prune_empty drops what wire_value could not encode, so assign it either way. - allowed[key] = wire_value(value) - fields["model_settings"] = allowed + fields["model_settings"] = filter_model_settings(getattr(agent, "model_settings", None)) return fields def _manifest_capabilities(self, agent: Any) -> AgentManifest: diff --git a/ddtrace/llmobs/types.py b/ddtrace/llmobs/types.py index d7e61075ac6..6a8298ce6b4 100644 --- a/ddtrace/llmobs/types.py +++ b/ddtrace/llmobs/types.py @@ -97,13 +97,15 @@ class AgentManifest(TypedDict, total=False): system_prompts: list[str] extra_instructions: list[AgentInstructionResolver] model: str + model_provider: str model_settings: dict[str, Any] agent_settings: dict[str, Any] tools: list[dict[str, Any]] capabilities: list[AgentCapability] data_contracts: dict[str, Any] guardrails: list[str] - handoffs: list[Any] + # A list of targets, or {"allow_delegation": bool} for frameworks that only report a flag. + handoffs: Union[list[dict[str, Any]], dict[str, Any]] handoff_description: str memory_policies: list[str] metadata: dict[str, Any] diff --git a/releasenotes/notes/llmobs-standardize-agent-manifest-df86615c800061b2.yaml b/releasenotes/notes/llmobs-standardize-agent-manifest-df86615c800061b2.yaml new file mode 100644 index 00000000000..906db0c95c5 --- /dev/null +++ b/releasenotes/notes/llmobs-standardize-agent-manifest-df86615c800061b2.yaml @@ -0,0 +1,17 @@ +--- +features: + - | + LLM Observability: CrewAI, Google ADK, LangGraph, OpenAI Agents, and Claude Agent SDK agent spans + now report agent configuration with a consistent set of fields, including instructions, tool + parameters, handoffs, and settings where the framework declares them. +fixes: + - | + LLM Observability: Fixes an issue where traces fail to be sent in agentless mode with + ``TypeError: Object of type function is not JSON serializable`` when a Google ADK agent uses a + callable instruction provider. + - | + LLM Observability: Fixes an issue where the agent configuration reported for OpenAI Agents could + include ``extra_headers``, ``extra_body``, and hosted MCP tool headers. + - | + LLM Observability: Fixes an issue where the model was missing from the Google ADK agent + configuration when the agent's model is set as a string. diff --git a/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py b/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py index 565f08ff3bf..0538ed592ef 100644 --- a/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py +++ b/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py @@ -141,7 +141,7 @@ async def test_llmobs_query_with_options( metadata={ "max_turns": 3, "stop_reason": "end_turn", - "_dd": {"agent_manifest": expected_agent_manifest(max_iterations=3)}, + "_dd": {"agent_manifest": expected_agent_manifest(max_turns=3)}, }, metrics=EXPECTED_QUERY_USAGE, tags=COMMON_TAGS, @@ -236,7 +236,7 @@ async def test_llmobs_query_error_no_output( span_kind="agent", input_value=safe_json(input_msgs), output_value=safe_json([{"content": ""}]), - metadata={"_dd": {"agent_manifest": {"framework": "Claude Agent SDK"}}}, + metadata={}, metrics={}, tags=COMMON_TAGS, error={"type": "builtins.ValueError", "message": "Connection failed", "stack": ANY}, diff --git a/tests/contrib/claude_agent_sdk/utils.py b/tests/contrib/claude_agent_sdk/utils.py index b6089dbe794..46b75806237 100644 --- a/tests/contrib/claude_agent_sdk/utils.py +++ b/tests/contrib/claude_agent_sdk/utils.py @@ -40,7 +40,7 @@ } -def expected_agent_manifest(max_iterations=None): +def expected_agent_manifest(max_turns=None): """Helper to build expected agent manifest.""" manifest = { "framework": "Claude Agent SDK", @@ -52,10 +52,9 @@ def expected_agent_manifest(max_iterations=None): {"name": "Write"}, {"name": "Grep"}, ], - "dependencies": {"mcp_servers": []}, } - if max_iterations is not None: - manifest["max_iterations"] = max_iterations + if max_turns is not None: + manifest["agent_settings"] = {"max_turns": max_turns} return manifest diff --git a/tests/contrib/crewai/test_crewai_llmobs.py b/tests/contrib/crewai/test_crewai_llmobs.py index 376dc74141d..400c08d0fd9 100644 --- a/tests/contrib/crewai/test_crewai_llmobs.py +++ b/tests/contrib/crewai/test_crewai_llmobs.py @@ -21,60 +21,49 @@ "Senior Research Scientist": { "framework": "CrewAI", "name": "Senior Research Scientist", - "goal": "Uncover cutting-edge developments in AI", - "backstory": "You're a seasoned researcher with a knack for uncovering the latest developments in AI. " + "instructions": "Uncover cutting-edge developments in AI\n\n" + "You're a seasoned researcher with a knack for uncovering the latest developments in AI. " "Known for your ability to find the most relevant information and present it in a clear " "and concise manner.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, - "tools": [], + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, }, "AI Reporting Analyst": { "framework": "CrewAI", "name": "AI Reporting Analyst", - "goal": "Create detailed reports based on AI data analysis and research findings", - "backstory": "You're a meticulous analyst with a keen eye for detail. You're known for your ability to turn " + "instructions": "Create detailed reports based on AI data analysis and research findings\n\n" + "You're a meticulous analyst with a keen eye for detail. You're known for your ability to turn " "complex data into clear and concise reports, making it easy for others to understand and act on the " "information you provide.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, - "tools": [], + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, }, "Python Data Analyst": { "framework": "CrewAI", "name": "Python Data Analyst", - "goal": "Analyze data and provide insights using Python", - "backstory": "You are an experienced data analyst with strong Python skills.", + "instructions": "Analyze data and provide insights using Python\n\n" + "You are an experienced data analyst with strong Python skills.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, "tools": [ { "name": "Average Calculator", - "description": "Tool Name: Average Calculator\nTool Arguments: {'entries': {'description': None, " - "'type': 'list'}}\nTool Description: This tool returns the average of a list of numbers.", + "description": "This tool returns the average of a list of numbers.", + "parameters": {"entries": {"type": "array", "required": True}}, } ], }, "Tour Guide": { "framework": "CrewAI", "name": "Tour Guide", - "goal": "Recommend fun activities for a group of humans.", - "backstory": "You are a tour guide with a passion for finding the best activities for groups of people.", + "instructions": "Recommend fun activities for a group of humans.\n\n" + "You are a tour guide with a passion for finding the best activities for groups of people.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, - "tools": [], + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, }, } diff --git a/tests/contrib/google_adk/test_google_adk_llmobs.py b/tests/contrib/google_adk/test_google_adk_llmobs.py index 7847a8d7ee4..7835c2391ab 100644 --- a/tests/contrib/google_adk/test_google_adk_llmobs.py +++ b/tests/contrib/google_adk/test_google_adk_llmobs.py @@ -18,7 +18,7 @@ AGENT_MANIFEST_METADATA = { "_dd": { "agent_manifest": { - "description": "Test agent for ADK integration testing", + "handoff_description": "Test agent for ADK integration testing", "framework": "Google ADK", "instructions": "You are a helpful test agent. You can: " "(1) call tools using the provided " @@ -29,17 +29,23 @@ "capability. Always be helpful and use " "your available capabilities.", "model": "gemini-2.5-pro", - "model_configuration": '{"arbitrary_types_allowed": true, "extra": "forbid"}', "name": "test_agent", - "session_management": { - "session_id": "test-session", - "user_id": "test-user", - "app_name": "TestADKApp", - }, "tools": [ - {"description": "A tiny search tool stub.", "name": "search_docs"}, - {"description": "Simple arithmetic tool.", "name": "multiply"}, + { + "description": "A tiny search tool stub.", + "name": "search_docs", + "parameters": {"query": {"type": "string", "required": True}}, + }, + { + "description": "Simple arithmetic tool.", + "name": "multiply", + "parameters": { + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, + }, + }, ], + "agent_settings": {"code_executor": "UnsafeLocalCodeExecutor"}, } } } diff --git a/tests/contrib/langgraph/test_langgraph_llmobs.py b/tests/contrib/langgraph/test_langgraph_llmobs.py index a16f73ac185..3153f82872d 100644 --- a/tests/contrib/langgraph/test_langgraph_llmobs.py +++ b/tests/contrib/langgraph/test_langgraph_llmobs.py @@ -378,26 +378,17 @@ def test_agent_manifest_simple_graph( expected_agent_a_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list", "which"], "name": "agent_a", - "tools": [], } expected_conditional_agent_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list", "which"], "name": conditional_agent_name, - "tools": [], } expected_agent_d_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list", "which"], "name": "agent_d", - "tools": [], } agent_a_metadata = get_llmobs_metadata(agent_a_span) or {} @@ -419,22 +410,14 @@ def test_agent_manifest_from_create_react_agent(self, langgraph_llmobs, test_spa expected_agent_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["messages"], "name": "not_your_average_bostonian", "tools": [ { "name": "add", "description": "Adds two numbers together", "parameters": { - "a": { - "title": "A", - "type": "integer", - }, - "b": { - "title": "B", - "type": "integer", - }, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -460,22 +443,14 @@ def test_agent_manifest_populates_tools_from_tool_node( expected_agent_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list"], "name": "custom_agent_with_tool_node", "tools": [ { "name": "add", "description": "Adds two numbers together", "parameters": { - "a": { - "title": "A", - "type": "integer", - }, - "b": { - "title": "B", - "type": "integer", - }, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -496,7 +471,8 @@ def test_agent_manifest_different_recursion_limit( agent_span = _find_span_by_name(spans, "agent") agent_metadata = get_llmobs_metadata(agent_span) or {} - assert agent_metadata.get("_dd", {}).get("agent_manifest", {}).get("max_iterations") == 100 + manifest = agent_metadata.get("_dd", {}).get("agent_manifest", {}) + assert manifest.get("agent_settings", {}).get("recursion_limit") == 100 @pytest.mark.skipif(LANGGRAPH_VERSION < (0, 3, 22), reason="Agent names are only supported in LangGraph 0.3.22+") def test_agent_with_tool_calls_integrations_enabled( diff --git a/tests/contrib/openai_agents/test_openai_agents_llmobs.py b/tests/contrib/openai_agents/test_openai_agents_llmobs.py index c728196def3..43e66b3c0a0 100644 --- a/tests/contrib/openai_agents/test_openai_agents_llmobs.py +++ b/tests/contrib/openai_agents/test_openai_agents_llmobs.py @@ -28,25 +28,22 @@ "framework": "OpenAI", "name": "Simple Agent", "instructions": "You are a helpful assistant who answers questions concisely and accurately.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, # different versions of the library have different model settings + "agent_settings": mock.ANY, }, "Simple Agent with Guardrails": { "framework": "OpenAI", "name": "Simple Agent", "instructions": "You are a helpful assistant specialized in addition calculations.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, # different versions of the library have different model settings + "agent_settings": mock.ANY, "tools": [ { "name": "add", "description": "Add two numbers together", - "strict_json_schema": True, "parameters": { - "a": {"type": "integer", "title": "A", "required": True}, - "b": {"type": "integer", "title": "B", "required": True}, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -56,17 +53,15 @@ "framework": "OpenAI", "name": "Addition Agent", "instructions": mock.ANY, - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, "tools": [ { "name": "add", "description": "Add two numbers together", - "strict_json_schema": True, "parameters": { - "a": {"type": "integer", "title": "A", "required": True}, - "b": {"type": "integer", "title": "B", "required": True}, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -76,36 +71,32 @@ "name": "Researcher", "instructions": "You are a helpful assistant that can research a topic using your research tool. " "Always research the topic before summarizing.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, "tools": [ { "name": "research", "description": "Research the internet on a topic.", - "strict_json_schema": True, - "parameters": {"query": {"type": "string", "title": "Query", "required": True}}, + "parameters": {"query": {"type": "string", "required": True}}, } ], "handoffs": [ - {"handoff_description": None, "agent_name": "Summarizer"}, + {"agent_name": "Summarizer"}, ], }, "Summarizer": { "framework": "OpenAI", "name": "Summarizer", "instructions": "You are a helpful assistant that can summarize a research results.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, }, "Weather Agent": { "framework": "OpenAI", "name": "Weather Agent", "instructions": "You are a helpful assistant specialized in searching the web for weather information.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, "tools": [ { "name": "web_search_preview", diff --git a/tests/llmobs/test_agent_manifest_integrations.py b/tests/llmobs/test_agent_manifest_integrations.py new file mode 100644 index 00000000000..891212c8c74 --- /dev/null +++ b/tests/llmobs/test_agent_manifest_integrations.py @@ -0,0 +1,357 @@ +"""Agent manifest contract across integrations, exercised with stand-in framework objects. + +Each builder reads attributes off a framework object, so a SimpleNamespace stands in for one and +these tests need no framework installed. +""" + +import json +from types import SimpleNamespace +from unittest import mock + +import pytest + +from ddtrace.llmobs._constants import LLMOBS_STRUCT +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.claude_agent_sdk import ClaudeAgentSdkIntegration +from ddtrace.llmobs._integrations.crewai import CrewAIIntegration +from ddtrace.llmobs._integrations.google_adk import GoogleAdkIntegration +from ddtrace.llmobs._integrations.langgraph import LangGraphIntegration +from ddtrace.llmobs._integrations.openai_agents import OpenAIAgentsIntegration +from ddtrace.llmobs._utils import _get_llmobs_data_metastruct +from ddtrace.llmobs.types import AgentManifest +from ddtrace.trace import Span + + +CANONICAL_KEYS = frozenset(AgentManifest.__annotations__) + + +def _never_called(*args, **kwargs): + raise AssertionError("a declared callable must not be invoked while building the manifest") + + +def _search(query: str, limit: int = 5, tool_context=None) -> list: + """Search the docs.""" + + +class _Answer: + pass + + +class _Graph: + """Stands in for a compiled graph, which the integration keys weakly.""" + + def __init__(self, name): + self.name = name + self.builder = None + + +def _manifest_from_span(span): + meta = _get_llmobs_data_metastruct(span).get(LLMOBS_STRUCT.META, {}) + return meta.get(LLMOBS_STRUCT.METADATA, {}).get(LLMOBS_STRUCT.METADATA_DD, {}).get(LLMOBS_STRUCT.AGENT_MANIFEST) + + +def _adk_agent(**overrides): + agent = dict( + name="billing", + description="Handles billing questions", + model="gemini-2.0-flash", + instruction=_never_called, + global_instruction="Be polite.", + static_instruction=None, + tools=[_search], + generate_content_config=SimpleNamespace(temperature=0.2, max_output_tokens=256), + output_schema=_Answer, + sub_agents=[SimpleNamespace(name="refunds", description="Issues refunds")], + before_model_callback=None, + before_tool_callback=None, + model_config={"arbitrary_types_allowed": True, "extra": "forbid"}, + ) + agent.update(overrides) + return SimpleNamespace(**agent) + + +def _openai_agent(**overrides): + agent = dict( + name="triage", + instructions="Route the request.", + prompt=None, + handoff_description="Routes requests", + model="gpt-4o", + model_settings=SimpleNamespace(temperature=0.1, extra_headers={"Authorization": "Bearer secret"}), + tools=[ + SimpleNamespace( + name="lookup", + description="Look up an order", + params_json_schema={ + "type": "object", + "properties": {"order_id": {"type": "string", "title": "Order Id"}}, + "required": ["order_id"], + }, + ), + SimpleNamespace( + name="hosted_mcp", + tool_config={ + "server_label": "github", + "server_url": "https://mcp.example.com?token=secret", + "headers": {"Authorization": "Bearer secret"}, + "allowed_tools": ["search"], + }, + on_approval_request=_never_called, + ), + SimpleNamespace(name="computer_use_preview", computer=object(), on_safety_check=_never_called), + ], + mcp_servers=[], + output_type=None, + handoffs=[SimpleNamespace(name="refunds", handoff_description="Issues refunds", tools=[], handoffs=[])], + input_guardrails=[SimpleNamespace(name=None, guardrail_function=_never_called)], + output_guardrails=[], + tool_use_behavior="run_llm_again", + reset_tool_choice=True, + ) + agent.update(overrides) + return SimpleNamespace(**agent) + + +def _crewai_agent(**overrides): + agent = dict( + role="AI Researcher", + goal="Research AI", + backstory="An expert in AI.", + llm=SimpleNamespace(model="gpt-4o-mini", temperature=0.0, max_tokens=None), + tools=[], + allow_delegation=False, + max_iter=25, + allow_code_execution=False, + code_execution_mode="safe", + ) + agent.update(overrides) + return SimpleNamespace(**agent) + + +def _claude_options(**overrides): + options = dict( + system_prompt="You are a code reviewer.", + mcp_servers={"fs": {"command": "fs-server", "env": {"TOKEN": "secret"}}}, + agents={"tester": SimpleNamespace(description="Writes tests")}, + can_use_tool=_never_called, + hooks=None, + max_turns=3, + ) + options.update(overrides) + return SimpleNamespace(**options) + + +def _build(integration_name, *args): + """Build a manifest the way each integration does on its agent span.""" + span = Span("agent") + if integration_name == "google_adk": + GoogleAdkIntegration(integration_config=mock.MagicMock())._tag_agent_manifest(span, {}, *args) + elif integration_name == "openai_agents": + integration = OpenAIAgentsIntegration(integration_config=mock.MagicMock()) + with mock.patch.object( + OpenAIAgentsIntegration, "llmobs_enabled", new_callable=mock.PropertyMock, return_value=True + ): + integration._tag_agent_manifest_from_agent(span, *args) + elif integration_name == "crewai": + CrewAIIntegration(integration_config=mock.MagicMock())._tag_agent_manifest(span, *args) + elif integration_name == "claude_agent_sdk": + return ClaudeAgentSdkIntegration(integration_config=mock.MagicMock())._build_agent_manifest(*args) + return _manifest_from_span(span) + + +CONTRACT_CASES = [ + ("google_adk", lambda: (_adk_agent(),)), + ("openai_agents", lambda: (_openai_agent(),)), + ("crewai", lambda: (_crewai_agent(),)), + ("claude_agent_sdk", lambda: ("claude-sonnet", _claude_options(), {"tools": ["Read"], "mcp_servers": []})), +] + + +@pytest.mark.parametrize("integration_name,make_args", CONTRACT_CASES, ids=[c[0] for c in CONTRACT_CASES]) +def test_manifest_contract(integration_name, make_args): + manifest = _build(integration_name, *make_args()) + + assert manifest, "the stand-in agent declares enough to report a manifest" + assert set(manifest) <= CANONICAL_KEYS, set(manifest) - CANONICAL_KEYS + assert manifest["framework"] + # Round-trips through JSON unchanged, so nothing relies on the encoder's repr fallback. + assert json.loads(json.dumps(manifest)) == manifest + # Rebuilt from a fresh but identical declaration, so nothing per process (an address) or per run leaks in. + assert _build(integration_name, *make_args()) == manifest + + +class TestSharedHelpers: + def test_callable_instructions_ship_by_name(self): + assert instruction_fields(_never_called) == { + "extra_instructions": [{"type": "dynamic_instructions", "name": "_never_called"}] + } + assert instruction_fields("Be brief.") == {"instructions": "Be brief."} + assert instruction_fields(object()) == {} + + def test_model_settings_allowlist(self): + assert filter_model_settings( + {"temperature": 0, "extra_headers": {"Authorization": "x"}, "extra_body": {"a": 1}} + ) == {"temperature": 0} + assert filter_model_settings(None) == {} + + def test_normalize_tool_flattens_json_schema(self): + assert normalize_tool( + "add", "Adds", {"type": "object", "properties": {"a": {"type": "integer", "title": "A"}}, "required": ["a"]} + ) == {"name": "add", "description": "Adds", "parameters": {"a": {"type": "integer", "required": True}}} + assert normalize_tool("", "unnamed") is None + assert normalize_tool("t", object())["description"] == "" + + def test_failing_section_costs_only_its_fields(self): + def broken(_): + raise RuntimeError("framework changed") + + manifest = build_agent_manifest("X", None, (("labels", lambda _: {"name": "a"}), ("tools", broken)), "test") + assert manifest == {"name": "a", "framework": "X"} + + def test_empty_manifest_is_not_reported(self): + assert build_agent_manifest("X", None, (("labels", lambda _: {"name": ""}),), "test") == {} + + +class TestGoogleAdk: + def test_manifest(self): + assert _build("google_adk", _adk_agent()) == { + "framework": "Google ADK", + "name": "billing", + "handoff_description": "Handles billing questions", + "extra_instructions": [{"type": "dynamic_instructions", "name": "_never_called"}], + "system_prompts": ["Be polite."], + "model": "gemini-2.0-flash", + "model_settings": {"temperature": 0.2, "max_tokens": 256}, + "tools": [ + { + "name": "_search", + "description": "Search the docs.", + "parameters": {"query": {"type": "string", "required": True}, "limit": {"type": "integer"}}, + } + ], + "data_contracts": {"output": {"name": "_Answer"}}, + "handoffs": [{"agent_name": "refunds", "handoff_description": "Issues refunds"}], + } + + def test_model_object(self): + manifest = _build("google_adk", _adk_agent(model=SimpleNamespace(model="gemini-2.5-pro"))) + assert manifest["model"] == "gemini-2.5-pro" + + def test_static_instruction_content(self): + content = SimpleNamespace(parts=[SimpleNamespace(text="Cached preamble.")]) + manifest = _build("google_adk", _adk_agent(global_instruction="", static_instruction=content)) + assert manifest["system_prompts"] == ["Cached preamble."] + + +class TestOpenAIAgents: + def test_manifest(self): + assert _build("openai_agents", _openai_agent()) == { + "framework": "OpenAI", + "name": "triage", + "handoff_description": "Routes requests", + "instructions": "Route the request.", + "model": "gpt-4o", + "model_settings": {"temperature": 0.1}, + "tools": [ + { + "name": "lookup", + "description": "Look up an order", + "parameters": {"order_id": {"type": "string", "required": True}}, + }, + {"name": "hosted_mcp", "server_label": "github", "allowed_tools": ["search"]}, + {"name": "computer_use_preview"}, + ], + "handoffs": [{"agent_name": "refunds", "handoff_description": "Issues refunds"}], + "guardrails": ["_never_called"], + "agent_settings": {"tool_use_behavior": "run_llm_again", "reset_tool_choice": True}, + } + + def test_callable_instructions_and_stored_prompt(self): + manifest = _build("openai_agents", _openai_agent(instructions=_never_called, prompt={"id": "pmpt_123"})) + assert "instructions" not in manifest + assert manifest["extra_instructions"] == [ + {"type": "dynamic_instructions", "name": "_never_called"}, + {"type": "prompt", "name": "pmpt_123"}, + ] + + +class TestCrewAI: + def test_manifest(self): + assert _build("crewai", _crewai_agent()) == { + "framework": "CrewAI", + "name": "AI Researcher", + "instructions": "Research AI\n\nAn expert in AI.", + "model": "gpt-4o-mini", + "model_settings": {"temperature": 0.0}, + "handoffs": {"allow_delegation": False}, + "agent_settings": {"max_iter": 25}, + } + + def test_code_execution_settings(self): + manifest = _build("crewai", _crewai_agent(allow_code_execution=True, code_execution_mode="unsafe")) + assert manifest["agent_settings"] == { + "max_iter": 25, + "allow_code_execution": True, + "code_execution_mode": "unsafe", + } + + +class TestLangGraph: + @pytest.mark.parametrize( + "model,expected", + [ + ("gpt-4o", {"model": "gpt-4o"}), + ("openai:gpt-4o", {"model": "gpt-4o", "model_provider": "openai"}), + ], + ) + def test_react_agent_model_string(self, model, expected): + integration = LangGraphIntegration(integration_config=mock.MagicMock()) + agent = _Graph("react") + with mock.patch.object( + LangGraphIntegration, "llmobs_enabled", new_callable=mock.PropertyMock, return_value=True + ): + integration.llmobs_handle_agent_manifest(agent, (model, []), {"name": "react", "prompt": _never_called}) + manifest = integration._get_agent_manifest(agent, (), {}) + assert manifest == { + "framework": "LangGraph", + "name": "react", + "extra_instructions": [{"type": "dynamic_prompt", "name": "_never_called"}], + **expected, + } + + def test_run_config_does_not_leak_into_cache(self): + integration = LangGraphIntegration(integration_config=mock.MagicMock()) + graph = _Graph("graph") + first = integration._get_agent_manifest(graph, (), {"recursion_limit": 100}) + second = integration._get_agent_manifest(graph, (), {}) + assert first["agent_settings"] == {"recursion_limit": 100} + assert second == {"framework": "LangGraph", "name": "graph"} + + +class TestClaudeAgentSdk: + def test_manifest(self): + manifest = _build( + "claude_agent_sdk", + "claude-sonnet", + _claude_options(), + {"tools": ["Read", "Bash"], "mcp_servers": [{"name": "fs", "status": "connected"}]}, + ) + assert manifest == { + "framework": "Claude Agent SDK", + "model": "claude-sonnet", + "instructions": "You are a code reviewer.", + "tools": [{"name": "Read"}, {"name": "Bash"}], + "capabilities": [{"name": "fs", "type": "mcp"}], + "handoffs": [{"agent_name": "tester", "handoff_description": "Writes tests"}], + "guardrails": ["_never_called"], + "agent_settings": {"max_turns": 3}, + } + + def test_preset_system_prompt(self): + options = _claude_options(system_prompt={"type": "preset", "preset": "claude_code", "append": "Be terse."}) + manifest = _build("claude_agent_sdk", "claude-sonnet", options, {}) + assert manifest["instructions"] == "Be terse." + assert manifest["extra_instructions"] == [{"type": "preset", "name": "claude_code"}]