diff --git a/.claude/skills/llmobs-integrations/SKILL.md b/.claude/skills/llmobs-integrations/SKILL.md index ecb1570c5de..98e163d9156 100644 --- a/.claude/skills/llmobs-integrations/SKILL.md +++ b/.claude/skills/llmobs-integrations/SKILL.md @@ -109,6 +109,23 @@ Note two already-shipped integrations predate this key: bedrock and the claude-a `metadata["stop_reason"]`. Renaming those is a breaking change and has not been done — follow `finish_reason` for new work. +### Agent Manifest + +Agent spans report the agent's declared configuration as `agent_manifest`, typed by `AgentManifest` +in `ddtrace/llmobs/types.py`. Build it with the helpers in `_integrations/agent_manifest.py`: + +- `build_agent_manifest(framework, agent, sections, integration_name)` runs each section in + isolation, drops unset values, and guarantees a JSON-native result. Return `{}` means "do not + annotate". +- Only emit keys declared on `AgentManifest`. Put loop-level knobs in `agent_settings` and inference + params through `filter_model_settings` (an allowlist; widening it is a security decision). +- Read declared configuration only, never per-run values (session ids, run config, interpolated + templates), so the manifest is identical run to run and version diffs stay meaningful. +- Use `instruction_fields` for instructions: a callable ships by name in `extra_instructions` and is + never called or `str()`-ed (its address changes every process). +- Use `normalize_tool` so every tool is `{name, description?, parameters: {p: {type?, required?}}}`. +- A description other agents see when routing to this one goes in `handoff_description`. + ## Key Constraints - **`submit_to_llmobs=True`** must be set on `LlmRequestEvent` for event-based request spans or passed to `integration.trace()` for direct LLMObs spans diff --git a/ddtrace/llmobs/_integrations/agent_manifest.py b/ddtrace/llmobs/_integrations/agent_manifest.py index f58284680fe..eea9b5c3e2e 100644 --- a/ddtrace/llmobs/_integrations/agent_manifest.py +++ b/ddtrace/llmobs/_integrations/agent_manifest.py @@ -3,6 +3,7 @@ import math import types from typing import Any +from typing import Callable from typing import Optional from typing import TypeVar from typing import Union @@ -172,6 +173,88 @@ def wire_value(value: Any, depth: int = 0, ancestors: tuple[int, ...] = (), budg return None +ManifestSection = tuple[str, Callable[[Any], AgentManifest]] + + +def build_agent_manifest( + framework: str, agent: Any, sections: tuple[ManifestSection, ...], integration_name: str +) -> AgentManifest: + """Run each section in isolation, merge them, and drop the fields that mean "not configured". + + A section that raises costs only its own fields, so a framework change inside one cannot blank + the rest. The result is passed through wire_value so it always survives JSON encoding. + """ + manifest: AgentManifest = {} + for name, section in sections: + try: + manifest.update(section(agent)) + except Exception: + log.debug("failed to build %s agent manifest section %s", integration_name, name, exc_info=True) + try: + wired = wire_value(prune_empty(manifest)) + except Exception: + log.debug("failed to finalize %s agent manifest", integration_name, exc_info=True) + return {} + if not wired: + return {} + wired["framework"] = framework + return cast(AgentManifest, wired) + + +def as_str(value: Any) -> str: + """str-only: the span encoder reprs what it cannot encode, and a repr can carry anything. + + A non-string reports "", which prune_empty drops like any other unset value. + """ + return value if isinstance(value, str) else "" + + +def config_value(value: Any) -> Any: + """JSON-native form of a declared config value. Pydantic models are dumped; other objects drop.""" + if hasattr(value, "model_dump"): + try: + value = value.model_dump(exclude_none=True) + except Exception: + return None + return wire_value(value) + + +def filter_model_settings(settings: Any) -> dict[str, Any]: + """Inference params filtered by ALLOWED_MODEL_SETTINGS_KEYS. Non-mappings report nothing.""" + if not isinstance(settings, dict): + return {} + allowed: dict[str, Any] = {} + for key, value in settings.items(): + if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value): + continue + # prune_empty drops what wire_value could not encode, so assign it either way. + allowed[key] = wire_value(value) + return allowed + + +def instruction_fields(value: Any, resolver_type: str = "dynamic_instructions") -> AgentManifest: + """Static text as instructions. A callable's text is only known at run time, so it ships by name. + + The callable is never invoked, and str() of it is avoided because its memory address changes + every process, which would report an instruction change on every deploy. + """ + if isinstance(value, str): + return {"instructions": value} + if callable(value): + return {"extra_instructions": [{"type": resolver_type, "name": callable_name(value)}]} + return {} + + +def normalize_tool(name: Any, description: Any = None, parameters: Any = None) -> Optional[dict[str, Any]]: + """One tool as {name, description?, parameters?}, the shape every integration emits. + + parameters accepts a JSON Schema object or the {param: {type, required}} mapping. + """ + if not isinstance(name, str) or not name: + return None + return {"name": name, "description": as_str(description), "parameters": tool_parameters(parameters)} + + def build_manual_agent_manifest(agent: Any) -> AgentManifest: """Build the manifest a caller declared through LLMObs.annotate(agent=...). @@ -229,21 +312,8 @@ def _manual_model_name(agent: dict[str, Any]) -> AgentManifest: def _manual_model_settings(agent: dict[str, Any]) -> AgentManifest: - """Inference params filtered by ALLOWED_MODEL_SETTINGS_KEYS. - - Separate from _manual_model_name so a malformed settings dict does not discard a valid model. - """ - fields: AgentManifest = {} - settings = agent.get("model_settings") - if isinstance(settings, dict): - allowed: dict[str, Any] = {} - for key, value in settings.items(): - if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value): - continue - # prune_empty drops what wire_value could not encode, so assign it either way. - allowed[key] = wire_value(value) - fields["model_settings"] = allowed - return fields + """Separate from _manual_model_name so a malformed settings dict does not discard a valid model.""" + return {"model_settings": filter_model_settings(agent.get("model_settings"))} def _manual_tools(agent: dict[str, Any]) -> AgentManifest: @@ -265,7 +335,7 @@ def _manual_tools(agent: dict[str, Any]) -> AgentManifest: { "name": name, "description": description if isinstance(description, str) else None, - "parameters": _manual_tool_parameters(tool.get("parameters")), + "parameters": tool_parameters(tool.get("parameters")), } ) wired = wire_value(tools) @@ -274,7 +344,7 @@ def _manual_tools(agent: dict[str, Any]) -> AgentManifest: return fields -def _manual_tool_parameters(parameters: Any) -> dict[str, Any]: +def tool_parameters(parameters: Any) -> dict[str, Any]: """{param: {type?, required?}}, matching what the framework integrations extract.""" if not isinstance(parameters, dict): return {} diff --git a/ddtrace/llmobs/_integrations/claude_agent_sdk.py b/ddtrace/llmobs/_integrations/claude_agent_sdk.py index bd402a22cd2..ae9d08c6a42 100644 --- a/ddtrace/llmobs/_integrations/claude_agent_sdk.py +++ b/ddtrace/llmobs/_integrations/claude_agent_sdk.py @@ -10,10 +10,16 @@ from ddtrace.llmobs._constants import INPUT_TOKENS_METRIC_KEY from ddtrace.llmobs._constants import OUTPUT_TOKENS_METRIC_KEY from ddtrace.llmobs._constants import TOTAL_TOKENS_METRIC_KEY +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import config_value +from ddtrace.llmobs._integrations.agent_manifest import as_str from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._utils import _annotate_llmobs_span_data from ddtrace.llmobs._utils import _get_attr from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentManifest from ddtrace.llmobs.types import Message from ddtrace.llmobs.types import ToolCall from ddtrace.llmobs.types import ToolResult @@ -139,7 +145,7 @@ def _llmobs_set_agent_tags( span.set_tag(ERROR_TYPE, error_type) span.set_tag(ERROR_MSG, error_message) - agent_manifest = self._build_agent_manifest(model, metadata, init_system_message) + agent_manifest = self._build_agent_manifest(model, kwargs.get("options"), init_system_message) _annotate_llmobs_span_data( span, @@ -149,25 +155,25 @@ def _llmobs_set_agent_tags( metadata=metadata, output_value=output_messages, metrics=metrics, - agent_manifest=agent_manifest, + agent_manifest=agent_manifest or None, ) - def _build_agent_manifest( - self, model: str, metadata: dict[str, Any], init_system_message: dict[str, Any] - ) -> dict[str, Any]: - manifest: dict[str, Any] = {} - manifest["framework"] = "Claude Agent SDK" - if model: - manifest["model"] = model - if init_system_message: - tools = init_system_message.get("tools", []) or [] - manifest["tools"] = [{"name": tool} for tool in tools] - if init_system_message: - mcp_servers = init_system_message.get("mcp_servers", []) or [] - manifest["dependencies"] = {"mcp_servers": mcp_servers} - if "max_turns" in metadata: - manifest["max_iterations"] = metadata["max_turns"] - return manifest + def _build_agent_manifest(self, model: str, options: Any, init_system_message: dict[str, Any]) -> dict[str, Any]: + declared = {"model": model, "options": options, "init": init_system_message or {}} + manifest = build_agent_manifest( + FRAMEWORK_NAME, + declared, + ( + ("model", lambda d: {"model": as_str(d["model"])}), + ("instructions", lambda d: _manifest_instructions(d["options"])), + ("tools", _manifest_tools), + ("handoffs", lambda d: _manifest_handoffs(d["options"])), + ("guardrails", lambda d: _manifest_guardrails(d["options"])), + ("agent_settings", lambda d: _manifest_agent_settings(d["options"])), + ), + self._integration_name, + ) + return dict(manifest) def _extract_input_messages(self, prompt: Any, span: Span) -> list[Message]: prompt_wrapper = span._get_ctx_item("_dd_prompt_wrapper") if span else None @@ -479,3 +485,75 @@ def _parse_tok(self, s: str) -> int: if s.lower().endswith("m"): return round(float(s[:-1]) * 1_000_000) return int(float(s)) + + +FRAMEWORK_NAME = "Claude Agent SDK" + +_AGENT_SETTINGS_OPTIONS = ( + "max_turns", + "max_budget_usd", + "max_thinking_tokens", + "permission_mode", + "allowed_tools", + "disallowed_tools", +) + + +def _manifest_instructions(options: Any) -> AgentManifest: + system_prompt = getattr(options, "system_prompt", None) + if isinstance(system_prompt, str): + return {"instructions": system_prompt} + if isinstance(system_prompt, dict): + # A preset is Claude Code's own prompt, resolved by the CLI, plus optional appended text. + fields: AgentManifest = {"instructions": as_str(system_prompt.get("append"))} + preset = as_str(system_prompt.get("preset")) + if preset: + fields["extra_instructions"] = [{"type": "preset", "name": preset}] + return fields + return {} + + +def _manifest_tools(declared: dict[str, Any]) -> AgentManifest: + # The init message lists every tool the session can call, built-ins included; allowed_tools only + # lists the ones that skip the permission prompt, so it is not the tool set. + init = declared["init"] + tools = [{"name": tool} for tool in init.get("tools") or [] if isinstance(tool, str) and tool] + servers = getattr(declared["options"], "mcp_servers", None) + if isinstance(servers, dict): + names = [name for name in servers if isinstance(name, str)] + else: + # The init entries also carry a connection status, which is per run and so not reported. + names = [as_str(_get_attr(server, "name", None)) for server in init.get("mcp_servers") or []] + capabilities: list[AgentCapability] = [{"name": name, "type": "mcp"} for name in names if name] + return {"tools": tools, "capabilities": capabilities} + + +def _manifest_handoffs(options: Any) -> AgentManifest: + agents = getattr(options, "agents", None) + if not isinstance(agents, dict): + return {} + return { + "handoffs": [ + {"agent_name": name, "handoff_description": as_str(getattr(definition, "description", None))} + for name, definition in agents.items() + if isinstance(name, str) and name + ] + } + + +def _manifest_guardrails(options: Any) -> AgentManifest: + """The permission callback and PreToolUse hooks, which can deny a tool call before it runs.""" + guardrails: list[str] = [] + can_use_tool = getattr(options, "can_use_tool", None) + if callable(can_use_tool): + guardrails.append(callable_name(can_use_tool)) + hooks = getattr(options, "hooks", None) + if isinstance(hooks, dict): + for matcher in hooks.get("PreToolUse") or []: + guardrails.extend(callable_name(fn) for fn in getattr(matcher, "hooks", None) or [] if callable(fn)) + return {"guardrails": guardrails} + + +def _manifest_agent_settings(options: Any) -> AgentManifest: + settings = {key: config_value(getattr(options, key, None)) for key in _AGENT_SETTINGS_OPTIONS} + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/crewai.py b/ddtrace/llmobs/_integrations/crewai.py index 610924a4c25..2878310daaa 100644 --- a/ddtrace/llmobs/_integrations/crewai.py +++ b/ddtrace/llmobs/_integrations/crewai.py @@ -10,12 +10,20 @@ from ddtrace.internal.utils.formats import format_trace_id from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL from ddtrace.llmobs._constants import ROOT_PARENT_ID +from ddtrace.llmobs._integrations.agent_manifest import as_str +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import is_number +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._utils import _annotate_llmobs_span_data from ddtrace.llmobs._utils import _get_nearest_llmobs_ancestor from ddtrace.llmobs._utils import get_llmobs_parent_id from ddtrace.llmobs._utils import get_llmobs_span_links from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentManifest from ddtrace.llmobs.types import _SpanLink from ddtrace.trace import Span from ddtrace.trace import tracer @@ -246,52 +254,23 @@ def _llmobs_set_tags_tool(self, span, args, kwargs, response): def _tag_agent_manifest(self, span, agent): if not agent: return - - manifest = {} - manifest["framework"] = "CrewAI" - manifest["name"] = agent.role if hasattr(agent, "role") and agent.role else "CrewAI Agent" - if hasattr(agent, "goal"): - manifest["goal"] = agent.goal - if hasattr(agent, "backstory"): - manifest["backstory"] = agent.backstory - if hasattr(agent, "llm"): - if hasattr(agent.llm, "model"): - manifest["model"] = agent.llm.model - model_settings = {} - if hasattr(agent.llm, "max_tokens"): - model_settings["max_tokens"] = agent.llm.max_tokens - if hasattr(agent.llm, "temperature"): - model_settings["temperature"] = agent.llm.temperature - if model_settings: - manifest["model_settings"] = model_settings - if hasattr(agent, "allow_delegation"): - manifest["handoffs"] = {"allow_delegation": agent.allow_delegation} - code_execution_permissions = {} - if hasattr(agent, "allow_code_execution"): - manifest["code_execution_permissions"] = {"allow_code_execution": agent.allow_code_execution} - if hasattr(agent, "code_execution_mode"): - manifest["code_execution_permissions"] = {"code_execution_mode": agent.code_execution_mode} - if code_execution_permissions: - manifest["code_execution_permissions"] = code_execution_permissions - if hasattr(agent, "max_iter"): - manifest["max_iterations"] = agent.max_iter - if hasattr(agent, "tools"): - manifest["tools"] = self._get_agent_tools(agent.tools) - - _annotate_llmobs_span_data(span, agent_manifest=manifest) - - def _get_agent_tools(self, tools): - if not tools or not isinstance(tools, list): - return [] - formatted_tools = [] - for tool in tools: - tool_dict = {} - if hasattr(tool, "name"): - tool_dict["name"] = tool.name - if hasattr(tool, "description"): - tool_dict["description"] = tool.description - formatted_tools.append(tool_dict) - return formatted_tools + manifest = build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", _manifest_labels), + ("instructions", _manifest_instructions), + ("model", _manifest_model), + ("tools", _manifest_tools), + ("capabilities", _manifest_capabilities), + ("handoffs", _manifest_handoffs), + ("guardrails", _manifest_guardrails), + ("agent_settings", _manifest_agent_settings), + ), + self._integration_name, + ) + if manifest: + _annotate_llmobs_span_data(span, agent_manifest=dict(manifest)) def _llmobs_set_tags_flow(self, span, args, kwargs, response): inputs = get_argument_value(args, kwargs, 0, "inputs", optional=True) or {} @@ -520,3 +499,102 @@ def _get_crew_id(span, operation): parent_id = span.parent_id return f"crew_{span.trace_id}_{parent_id}" return f"{span.trace_id}" + + +FRAMEWORK_NAME = "CrewAI" + +# CrewAI LLM attribute to its generic model_settings key. stop is left out: CrewAI sets its own stop +# words on the LLM at run time. +_LLM_SETTINGS_KEYS = { + "temperature": "temperature", + "max_tokens": "max_tokens", + "top_p": "top_p", + "seed": "seed", + "presence_penalty": "presence_penalty", + "frequency_penalty": "frequency_penalty", + "timeout": "timeout", +} + + +def _manifest_labels(agent: Any) -> AgentManifest: + role = getattr(agent, "role", None) + return {"name": role if isinstance(role, str) and role else "CrewAI Agent"} + + +def _manifest_instructions(agent: Any) -> AgentManifest: + # CrewAI builds the agent's system prompt from its goal and backstory. + texts = [as_str(getattr(agent, "goal", None)), as_str(getattr(agent, "backstory", None))] + templates = [as_str(getattr(agent, attr, None)) for attr in ("system_template", "prompt_template")] + return { + "instructions": "\n\n".join(text for text in texts if text), + "system_prompts": [template for template in templates if template], + } + + +def _manifest_model(agent: Any) -> AgentManifest: + llm = getattr(agent, "llm", None) + model = llm if isinstance(llm, str) else as_str(getattr(llm, "model", None)) + settings = {key: getattr(llm, attr, None) for attr, key in _LLM_SETTINGS_KEYS.items()} + return {"model": model, "model_settings": filter_model_settings(settings)} + + +def _manifest_tools(agent: Any) -> AgentManifest: + tools: list[dict[str, Any]] = [] + for tool in getattr(agent, "tools", None) or []: + description = as_str(getattr(tool, "description", None)) + entry = normalize_tool( + getattr(tool, "name", None), + _extract_tool_description_field(description) if description else None, + _tool_json_schema(tool), + ) + if entry: + tools.append(entry) + return {"tools": tools} + + +def _tool_json_schema(tool: Any) -> Optional[dict[str, Any]]: + model_json_schema = getattr(getattr(tool, "args_schema", None), "model_json_schema", None) + if not callable(model_json_schema): + return None + try: + schema = model_json_schema() + except Exception: + return None + return schema if isinstance(schema, dict) else None + + +def _manifest_capabilities(agent: Any) -> AgentManifest: + capabilities: list[AgentCapability] = [ + {"name": type(source).__name__, "type": "knowledge"} + for source in getattr(agent, "knowledge_sources", None) or [] + ] + return {"capabilities": capabilities} + + +def _manifest_handoffs(agent: Any) -> AgentManifest: + # CrewAI delegates to any crew member rather than declaring targets, so only the flag is known. + allow_delegation = getattr(agent, "allow_delegation", None) + return {"handoffs": {"allow_delegation": allow_delegation}} if isinstance(allow_delegation, bool) else {} + + +def _manifest_guardrails(agent: Any) -> AgentManifest: + guardrail = getattr(agent, "guardrail", None) + if isinstance(guardrail, str): + return {"guardrails": [guardrail]} + if callable(guardrail): + return {"guardrails": [callable_name(guardrail)]} + return {} + + +def _manifest_agent_settings(agent: Any) -> AgentManifest: + settings: dict[str, Any] = {} + for attr in ("max_iter", "max_rpm", "max_execution_time", "max_retry_limit"): + value = getattr(agent, attr, None) + if is_number(value): + settings[attr] = value + if getattr(agent, "allow_code_execution", None) is True: + settings["allow_code_execution"] = True + settings["code_execution_mode"] = as_str(getattr(agent, "code_execution_mode", None)) + if getattr(agent, "reasoning", None) is True: + settings["reasoning"] = True + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/google_adk.py b/ddtrace/llmobs/_integrations/google_adk.py index 486ab3676a3..435b23194b5 100644 --- a/ddtrace/llmobs/_integrations/google_adk.py +++ b/ddtrace/llmobs/_integrations/google_adk.py @@ -1,3 +1,4 @@ +import inspect from inspect import isfunction from typing import Any from typing import Optional @@ -6,12 +7,21 @@ from ddtrace.internal.constants import COMPONENT from ddtrace.internal.utils import get_argument_value from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL +from ddtrace.llmobs._integrations.agent_manifest import as_str +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import is_number +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.agent_manifest import type_name from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._integrations.google_utils import extract_message_from_part_google_genai from ddtrace.llmobs._integrations.google_utils import extract_messages_from_adk_events from ddtrace.llmobs._utils import _annotate_llmobs_span_data -from ddtrace.llmobs._utils import _get_attr from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentManifest from ddtrace.trace import Span @@ -135,23 +145,23 @@ def _llmobs_set_tags_tool( def _tag_agent_manifest(self, span: Span, kwargs: dict[str, Any], agent: Any) -> None: if not agent: return - - manifest: dict[str, Any] = {} - - manifest["framework"] = "Google ADK" - manifest["name"] = getattr(agent, "name", "") - manifest["model"] = getattr(getattr(agent, "model", ""), "model", "") - manifest["description"] = getattr(agent, "description", "") - manifest["instructions"] = getattr(agent, "instruction", "") - manifest["model_configuration"] = safe_json(getattr(agent, "model_config", "")) - manifest["session_management"] = { - "session_id": kwargs.get("session_id", ""), - "user_id": kwargs.get("user_id", ""), - "app_name": kwargs.get("app_name", ""), - } - manifest["tools"] = self._get_agent_tools(getattr(agent, "tools", [])) - - _annotate_llmobs_span_data(span, agent_manifest=manifest) + manifest = build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", _manifest_labels), + ("instructions", _manifest_instructions), + ("model", _manifest_model), + ("tools", _manifest_tools), + ("data_contracts", _manifest_data_contracts), + ("handoffs", _manifest_handoffs), + ("guardrails", _manifest_guardrails), + ("agent_settings", _manifest_agent_settings), + ), + self._integration_name, + ) + if manifest: + _annotate_llmobs_span_data(span, agent_manifest=dict(manifest)) def _llmobs_set_tags_code_execute( self, span: Span, args: list[Any], kwargs: dict[str, Any], response: Optional[Any] = None @@ -172,18 +182,172 @@ def _llmobs_set_tags_code_execute( output_value=output, ) - def _get_agent_tools(self, tools): - if not tools or not isinstance(tools, list): - return [] - - agent_tools = [] - for tool in tools: - if isfunction(tool): - tool_name = tool.__name__ - tool_description = tool.__doc__ or "" - else: - tool_name = _get_attr(tool, "name", "Agent Tool") - tool_description = _get_attr(tool, "description", "") - agent_tools.append({"name": tool_name, "description": tool_description}) - - return agent_tools + +FRAMEWORK_NAME = "Google ADK" + +# GenerateContentConfig field to its generic model_settings key. +_GENERATE_CONTENT_CONFIG_KEYS = { + "temperature": "temperature", + "top_p": "top_p", + "top_k": "top_k", + "max_output_tokens": "max_tokens", + "stop_sequences": "stop_sequences", + "seed": "seed", + "presence_penalty": "presence_penalty", + "frequency_penalty": "frequency_penalty", +} + +# JSON Schema names, so a signature-derived tool reads the same as one from a schema. +_JSON_SCHEMA_TYPES = { + "str": "string", + "int": "integer", + "float": "number", + "bool": "boolean", + "list": "array", + "dict": "object", +} + +# Framework-injected arguments, not ones the model fills in. +_IGNORED_TOOL_PARAMETERS = frozenset({"self", "cls", "tool_context"}) + + +def _manifest_labels(agent: Any) -> AgentManifest: + # ADK shows a sub-agent's description to its parent's model when choosing a transfer target, + # which is the role handoff_description plays in the other integrations. + return { + "name": as_str(getattr(agent, "name", None)), + "handoff_description": as_str(getattr(agent, "description", None)), + } + + +def _manifest_instructions(agent: Any) -> AgentManifest: + fields = instruction_fields(getattr(agent, "instruction", None)) + system_prompts: list[str] = [] + global_instruction = getattr(agent, "global_instruction", None) + if isinstance(global_instruction, str) and global_instruction: + system_prompts.append(global_instruction) + elif callable(global_instruction): + fields["extra_instructions"] = fields.get("extra_instructions", []) + [ + {"type": "dynamic_global_instruction", "name": callable_name(global_instruction)} + ] + static_text = _content_text(getattr(agent, "static_instruction", None)) + if static_text: + system_prompts.append(static_text) + fields["system_prompts"] = system_prompts + return fields + + +def _content_text(content: Any) -> str: + """Text of a genai ContentUnion: a str, a Part, a Content, or a list of those.""" + if isinstance(content, str): + return content + if isinstance(content, list): + texts = [_content_text(item) for item in content] + return "\n".join(text for text in texts if text) + parts = getattr(content, "parts", None) + if isinstance(parts, list): + return _content_text(parts) + return as_str(getattr(content, "text", None)) + + +def _manifest_model(agent: Any) -> AgentManifest: + model = getattr(agent, "model", None) + model_name = model if isinstance(model, str) else as_str(getattr(model, "model", None)) + config = getattr(agent, "generate_content_config", None) + settings = {key: getattr(config, field, None) for field, key in _GENERATE_CONTENT_CONFIG_KEYS.items()} + return {"model": model_name, "model_settings": filter_model_settings(settings)} + + +def _manifest_tools(agent: Any) -> AgentManifest: + tools: list[dict[str, Any]] = [] + capabilities: list[AgentCapability] = [] + for tool in getattr(agent, "tools", None) or []: + if isfunction(tool): + entry = normalize_tool(tool.__name__, tool.__doc__, _function_parameters(tool)) + elif hasattr(tool, "get_tools") and not hasattr(tool, "name"): + # A toolset resolves its tools at run time, so only its presence is declared. + kind = "mcp" if "mcp" in type(tool).__name__.lower() else "toolset" + capabilities.append({"name": type(tool).__name__, "type": kind}) + continue + else: + func = getattr(tool, "func", None) + entry = normalize_tool( + getattr(tool, "name", None), + getattr(tool, "description", None), + _function_parameters(func) if callable(func) else None, + ) + if entry: + tools.append(entry) + return {"tools": tools, "capabilities": capabilities} + + +def _function_parameters(fn: Any) -> dict[str, Any]: + """{param: {type?, required?}} read off the signature. The function is never called.""" + try: + signature = inspect.signature(fn) + except (TypeError, ValueError): + return {} + parameters: dict[str, Any] = {} + for name, param in signature.parameters.items(): + if name in _IGNORED_TOOL_PARAMETERS or param.kind in (param.VAR_POSITIONAL, param.VAR_KEYWORD): + continue + spec: dict[str, Any] = {} + if param.annotation is not param.empty: + annotation = param.annotation + # A string annotation comes from a module using postponed evaluation. + annotation_name = annotation if isinstance(annotation, str) else type_name(annotation) + spec["type"] = _JSON_SCHEMA_TYPES.get(annotation_name, annotation_name) + if param.default is param.empty: + spec["required"] = True + parameters[name] = spec + return parameters + + +def _manifest_data_contracts(agent: Any) -> AgentManifest: + contracts: dict[str, Any] = {} + for key, attr in (("input", "input_schema"), ("output", "output_schema")): + schema = getattr(agent, attr, None) + if isinstance(schema, type): + contracts[key] = {"name": type_name(schema)} + return {"data_contracts": contracts} + + +def _manifest_handoffs(agent: Any) -> AgentManifest: + handoffs: list[dict[str, Any]] = [] + for sub_agent in getattr(agent, "sub_agents", None) or []: + name = as_str(getattr(sub_agent, "name", None)) + if name: + handoffs.append( + {"agent_name": name, "handoff_description": as_str(getattr(sub_agent, "description", None))} + ) + return {"handoffs": handoffs} + + +def _manifest_guardrails(agent: Any) -> AgentManifest: + """Callbacks that can veto a model or tool call before it runs, which is how ADK does guardrails.""" + guardrails: list[str] = [] + for attr in ("before_model_callback", "before_tool_callback"): + callbacks = getattr(agent, attr, None) + for callback in callbacks if isinstance(callbacks, list) else [callbacks]: + if callable(callback): + guardrails.append(callable_name(callback)) + return {"guardrails": guardrails} + + +def _manifest_agent_settings(agent: Any) -> AgentManifest: + settings: dict[str, Any] = {} + for attr in ("disallow_transfer_to_parent", "disallow_transfer_to_peers"): + if getattr(agent, attr, None) is True: + settings[attr] = True + include_contents = getattr(agent, "include_contents", None) + if isinstance(include_contents, str) and include_contents != "default": + settings["include_contents"] = include_contents + settings["output_key"] = as_str(getattr(agent, "output_key", None)) + max_iterations = getattr(agent, "max_iterations", None) + if is_number(max_iterations): + settings["max_iterations"] = max_iterations + for attr in ("planner", "code_executor"): + value = getattr(agent, attr, None) + if value is not None: + settings[attr] = type(value).__name__ + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/langgraph.py b/ddtrace/llmobs/_integrations/langgraph.py index 226ec96d1a9..83c66500aad 100644 --- a/ddtrace/llmobs/_integrations/langgraph.py +++ b/ddtrace/llmobs/_integrations/langgraph.py @@ -10,6 +10,14 @@ from ddtrace.internal.utils.formats import format_trace_id from ddtrace.llmobs import LLMObs from ddtrace.llmobs._constants import ROOT_PARENT_ID +from ddtrace.llmobs._integrations.agent_manifest import ALLOWED_MODEL_SETTINGS_KEYS +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import is_number +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.agent_manifest import as_str from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._integrations.constants import LANGGRAPH_ASTREAM_OUTPUT from ddtrace.llmobs._integrations.utils import format_langchain_io @@ -18,6 +26,7 @@ from ddtrace.llmobs._utils import _get_nearest_llmobs_ancestor from ddtrace.llmobs._utils import get_llmobs_parent_id from ddtrace.llmobs._utils import get_llmobs_span_links +from ddtrace.llmobs.types import AgentManifest from ddtrace.llmobs.types import _SpanLink from ddtrace.trace import Span @@ -28,25 +37,13 @@ PREGEL_PUSH = "__pregel_push" # represents a task queued up by a `Send` command PREGEL_TASKS = "__pregel_tasks" # name of ephemeral channel that pregel `Send` commands write to -ALLOWED_MODEL_SETTINGS_KEYS = [ - "max_tokens", - "temperature", - "top_p", - "top_k", - "frequency_penalty", - "presence_penalty", - "stop", - "n", - "logprobs", - "echo", - "logit_bias", -] +FRAMEWORK_NAME = "LangGraph" class LangGraphIntegration(BaseLLMIntegration): _integration_name = "langgraph" _graph_nodes_for_graph_by_task_id: WeakKeyDictionary[Span, dict[str, Any]] = WeakKeyDictionary() - _agent_manifests: WeakKeyDictionary[Any, dict[str, Any]] = WeakKeyDictionary() + _agent_manifests: WeakKeyDictionary[Any, AgentManifest] = WeakKeyDictionary() _graph_spans_to_graph_instances: WeakKeyDictionary[Span, Any] = WeakKeyDictionary() def trace( @@ -126,26 +123,22 @@ def _get_agent_manifest(self, agent, args, config: dict[str, Any]) -> Optional[d if agent is None: return None - agent_manifest = self._agent_manifests.get(agent) - if agent_manifest is None: - tools = _get_tools_from_graph(agent) - agent_manifest = {"name": agent.name or "LangGraph", "tools": tools} - self._agent_manifests[agent] = agent_manifest - - if "framework" not in agent_manifest: - agent_manifest["framework"] = "LangGraph" - if "max_iterations" not in agent_manifest: - agent_manifest["max_iterations"] = _get_attr(config, "recursion_limit", 25) - - if ( - "dependencies" not in agent_manifest - and isinstance(args, tuple) - and len(args) > 0 - and isinstance(args[0], dict) - ): - agent_manifest["dependencies"] = list(args[0].keys()) + declared = self._agent_manifests.get(agent) + if declared is None: + declared = build_agent_manifest( + FRAMEWORK_NAME, + agent, + (("labels", _manifest_labels), ("tools", _manifest_graph_tools)), + self._integration_name, + ) + self._agent_manifests[agent] = declared - return agent_manifest + # Copied so a run's config never leaks into the manifest cached for the next run. + manifest: dict[str, Any] = dict(declared) + recursion_limit = _get_attr(config, "recursion_limit", None) + if is_number(recursion_limit): + manifest["agent_settings"] = {**manifest.get("agent_settings", {}), "recursion_limit": recursion_limit} + return manifest def _get_node_metadata_from_span(self, span: Span, instance_id: str) -> dict[str, Any]: """ @@ -170,35 +163,24 @@ def llmobs_handle_agent_manifest(self, agent, args: tuple, kwargs: dict): if not self.llmobs_enabled: return - model = get_argument_value( - args, kwargs, 0, "model", True - ) # required parameter on the langgraph side, but optional should that ever change - model_name, model_provider, model_settings = _get_model_info(model) - - agent_tools: list[Any] = ( - get_argument_value(args, kwargs, 1, "tools", True) or [] - ) # required parameter on the langgraph side, but optional should that ever change - tools = _get_tools_from_react_agent(agent_tools) - - system_prompt: Optional[str] = _get_system_prompt_from_react_agent(kwargs.get("prompt")) - name: Optional[str] = kwargs.get("name") - - agent_manifest: dict[str, Any] = {} - - if model_name: - agent_manifest["model"] = model_name - if model_provider: - agent_manifest["model_provider"] = model_provider - if model_settings: - agent_manifest["model_settings"] = model_settings - if tools: - agent_manifest["tools"] = tools - if system_prompt: - agent_manifest["instructions"] = system_prompt - if name: - agent_manifest["name"] = name - - self._agent_manifests[agent] = agent_manifest + # model and tools are required parameters on the langgraph side, but optional should that ever change. + declared = { + "name": kwargs.get("name"), + "model": get_argument_value(args, kwargs, 0, "model", True), + "tools": get_argument_value(args, kwargs, 1, "tools", True) or [], + "prompt": kwargs.get("prompt"), + } + self._agent_manifests[agent] = build_agent_manifest( + FRAMEWORK_NAME, + declared, + ( + ("labels", lambda d: {"name": as_str(d["name"]) or "LangGraph"}), + ("model", lambda d: _manifest_model(d["model"])), + ("tools", lambda d: {"tools": _get_tools_from_react_agent(d["tools"]) or []}), + ("instructions", lambda d: _manifest_react_prompt(d["prompt"])), + ), + self._integration_name, + ) def llmobs_handle_pregel_loop_tick( self, finished_tasks: dict, next_tasks: dict, more_tasks: bool, is_subgraph_node: bool = False @@ -333,17 +315,24 @@ def _link_standalone_terminal_tasks( _annotate_llmobs_span_data(graph_span, span_links=graph_span_links) -def _get_model_info(model) -> tuple[Optional[str], Optional[str], dict[str, Any]]: - """Get the model name, provider, and settings from a langchain llm""" - if isinstance(model, str): - # something like "openai:gpt-4" - model_provider_str, model_name_str = model.split(":", maxsplit=1) - return model_name_str, model_provider_str, {} +def _manifest_labels(agent: Any) -> AgentManifest: + return {"name": as_str(_get_attr(agent, "name", None)) or "LangGraph"} + + +def _manifest_graph_tools(agent: Any) -> AgentManifest: + return {"tools": _get_tools_from_graph(agent)} - model_name = _get_attr(model, "model_name", None) - model_provider = _get_model_provider(model) - model_settings = _get_model_settings(model) - return model_name, model_provider, model_settings + +def _manifest_model(model: Any) -> AgentManifest: + """The model name, provider and settings from a langchain chat model or a "provider:model" string.""" + if isinstance(model, str): + provider, sep, name = model.partition(":") + return {"model": name, "model_provider": provider} if sep else {"model": model} + return { + "model": as_str(_get_attr(model, "model_name", None)), + "model_provider": as_str(_get_model_provider(model)), + "model_settings": _get_model_settings(model), + } def _get_model_provider(model) -> Optional[str]: @@ -358,32 +347,20 @@ def _get_model_provider(model) -> Optional[str]: def _get_model_settings(model) -> dict[str, Any]: """Get the model settings from a langchain llm""" - model_dict_fn = _get_attr(model, "dict", None) - if model_dict_fn is None or not callable(model_dict_fn): - return {} - - model_dict: dict = model.dict() - return {key: value for key, value in model_dict.items() if key in ALLOWED_MODEL_SETTINGS_KEYS and value} - - -def _get_system_prompt_from_react_agent(system_prompt) -> Optional[str]: - """ - Get the system prompt from a react agent. - - The system prompt can be: - - a string - - a dict with a "content" key - - a Callable that returns a string or dict - - In the case of a Callable (which is dynamic as a function of state and config), we end up returning None. - """ - if system_prompt is None: - return None + settings = {key: _get_attr(model, key, None) for key in ALLOWED_MODEL_SETTINGS_KEYS} + settings["stop_sequences"] = _get_attr(model, "stop", None) + return filter_model_settings(settings) - if isinstance(system_prompt, str): - return system_prompt - return _get_attr(system_prompt, "content", None) +def _manifest_react_prompt(prompt: Any) -> AgentManifest: + """A react agent's prompt: a string, a SystemMessage, or a callable resolved per run.""" + if prompt is None or isinstance(prompt, str): + return instruction_fields(prompt) + content = _get_attr(prompt, "content", None) + if isinstance(content, str): + return {"instructions": content} + # A Runnable or callable prompt decides the text from the run's state. + return {"extra_instructions": [{"type": "dynamic_prompt", "name": callable_name(prompt)}]} def _get_tools_from_react_agent(tools: Any) -> Optional[list[dict[str, Any]]]: @@ -409,12 +386,27 @@ def _get_tool_repr_from_langchain_base_tool(tool) -> Optional[dict[str, Any]]: """Get the tool representation from a langchain base tool""" if tool is None or isinstance(tool, dict): return None - - return { - "name": _get_attr(tool, "name", ""), - "description": _get_attr(tool, "description", ""), - "parameters": _get_attr(tool, "args", {}), - } + if _get_attr(tool, "name", None) is None and callable(tool): + # A plain function, which langgraph wraps into a tool itself. + return normalize_tool(callable_name(tool), getattr(tool, "__doc__", None)) + return normalize_tool(_get_attr(tool, "name", None), _get_attr(tool, "description", None), _tool_json_schema(tool)) + + +def _tool_json_schema(tool) -> Optional[dict[str, Any]]: + """The tool's JSON Schema, which unlike tool.args also says which parameters are required.""" + args_schema = _get_attr(tool, "args_schema", None) + if isinstance(args_schema, dict): + return args_schema + model_json_schema = getattr(args_schema, "model_json_schema", None) + if callable(model_json_schema): + try: + schema = model_json_schema() + except Exception: + schema = None + if isinstance(schema, dict): + return schema + args = _get_attr(tool, "args", None) + return {"type": "object", "properties": args} if isinstance(args, dict) else None def _is_tool_node(maybe_tool_node): diff --git a/ddtrace/llmobs/_integrations/openai_agents.py b/ddtrace/llmobs/_integrations/openai_agents.py index 69a51a6989a..75c40b47c33 100644 --- a/ddtrace/llmobs/_integrations/openai_agents.py +++ b/ddtrace/llmobs/_integrations/openai_agents.py @@ -2,6 +2,7 @@ from typing import Any from typing import Optional from typing import Union +from typing import get_origin import weakref from ddtrace.internal import core @@ -14,6 +15,15 @@ from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL_OUTPUT_USED from ddtrace.llmobs._constants import OAI_HANDOFF_TOOL_ARG from ddtrace.llmobs._constants import ROOT_PARENT_ID +from ddtrace.llmobs._integrations.agent_manifest import ALLOWED_MODEL_SETTINGS_KEYS +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import callable_name +from ddtrace.llmobs._integrations.agent_manifest import config_value +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.agent_manifest import as_str +from ddtrace.llmobs._integrations.agent_manifest import type_name from ddtrace.llmobs._integrations.base import BaseLLMIntegration from ddtrace.llmobs._integrations.utils import LLMObsTraceInfo from ddtrace.llmobs._integrations.utils import OaiSpanAdapter @@ -23,8 +33,10 @@ from ddtrace.llmobs._utils import get_llmobs_parent_id from ddtrace.llmobs._utils import get_llmobs_span_name from ddtrace.llmobs._utils import get_tool_version_from_llm_span -from ddtrace.llmobs._utils import load_data_value from ddtrace.llmobs._utils import safe_json +from ddtrace.llmobs.types import AgentCapability +from ddtrace.llmobs.types import AgentInstructionResolver +from ddtrace.llmobs.types import AgentManifest from ddtrace.trace import Span @@ -325,140 +337,158 @@ def tag_agent_manifest(self, span: Span, args: list[Any], kwargs: dict[str, Any] def _tag_agent_manifest_from_agent(self, span: Span, agent: Any) -> None: if not agent or not self.llmobs_enabled: return + manifest = build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", _manifest_labels), + ("instructions", _manifest_instructions), + ("model", _manifest_model), + ("tools", _manifest_tools), + ("capabilities", _manifest_capabilities), + ("data_contracts", _manifest_data_contracts), + ("handoffs", _manifest_handoffs), + ("guardrails", _manifest_guardrails), + ("agent_settings", _manifest_agent_settings), + ), + self._integration_name, + ) + if manifest: + _annotate_llmobs_span_data(span, agent_manifest=dict(manifest)) + + +FRAMEWORK_NAME = "OpenAI" + +# Declared config worth diffing on each hosted tool. Everything else on these tools is a callable, +# a client object, or a credential (HostedMCPTool.tool_config carries headers and authorization). +_HOSTED_TOOL_FIELDS = { + "file_search": ("vector_store_ids", "max_num_results", "include_search_results", "ranking_options", "filters"), + "web_search": ("user_location", "search_context_size", "filters"), + "web_search_preview": ("user_location", "search_context_size", "filters"), +} +_HOSTED_MCP_CONFIG_KEYS = ("server_label", "allowed_tools", "require_approval") + + +def _manifest_labels(agent: Any) -> AgentManifest: + return { + "name": as_str(getattr(agent, "name", None)), + "handoff_description": as_str(getattr(agent, "handoff_description", None)), + } + + +def _manifest_instructions(agent: Any) -> AgentManifest: + fields = instruction_fields(getattr(agent, "instructions", None)) + prompt = getattr(agent, "prompt", None) + # A stored prompt is resolved by the Responses API at run time, so it is recorded by id. + resolver: Optional[AgentInstructionResolver] = None + if isinstance(prompt, dict) and isinstance(prompt.get("id"), str): + resolver = {"type": "prompt", "name": prompt["id"]} + elif callable(prompt): + resolver = {"type": "dynamic_prompt", "name": callable_name(prompt)} + if resolver: + fields["extra_instructions"] = fields.get("extra_instructions", []) + [resolver] + return fields + + +def _manifest_model(agent: Any) -> AgentManifest: + model = getattr(agent, "model", None) + model_name = model if isinstance(model, str) else as_str(getattr(model, "model", None)) + settings = getattr(agent, "model_settings", None) + # Read field by field rather than dumped, so extra_headers and extra_body are never touched. + if settings is not None and not isinstance(settings, dict): + settings = {key: getattr(settings, key, None) for key in ALLOWED_MODEL_SETTINGS_KEYS} + return {"model": model_name, "model_settings": filter_model_settings(settings)} + + +def _manifest_tools(agent: Any) -> AgentManifest: + tools: list[dict[str, Any]] = [] + for tool in getattr(agent, "tools", None) or []: + name = getattr(tool, "name", None) + if hasattr(tool, "params_json_schema"): + entry = normalize_tool(name, getattr(tool, "description", None), tool.params_json_schema) + else: + entry = _hosted_tool(tool, name) + if entry: + tools.append(entry) + return {"tools": tools} - manifest = {} - manifest["framework"] = "OpenAI" - if hasattr(agent, "name"): - manifest["name"] = agent.name - if hasattr(agent, "instructions"): - manifest["instructions"] = agent.instructions - if hasattr(agent, "handoff_description"): - manifest["handoff_description"] = agent.handoff_description - if hasattr(agent, "model"): - model = agent.model - manifest["model"] = model if isinstance(model, str) else getattr(model, "model", "") - - model_settings = self._extract_model_settings_from_agent(agent) - if model_settings: - manifest["model_settings"] = model_settings - - tools = self._extract_tools_from_agent(agent) - if tools: - manifest["tools"] = tools - - handoffs = self._extract_handoffs_from_agent(agent) - if handoffs: - manifest["handoffs"] = handoffs - - guardrails = self._extract_guardrails_from_agent(agent) - if guardrails: - manifest["guardrails"] = guardrails - - _annotate_llmobs_span_data(span, agent_manifest=manifest) - - def _extract_model_settings_from_agent(self, agent): - if not hasattr(agent, "model_settings"): - return None - - # convert model_settings to dict if it's not already - model_settings = agent.model_settings - if not isinstance(model_settings, dict): - model_settings = getattr(model_settings, "__dict__", None) - - return load_data_value(model_settings) - - def _extract_tools_from_agent(self, agent): - if not hasattr(agent, "tools") or not agent.tools: - return None - - tools = [] - for tool in agent.tools: - tool_dict = {} - tool_name = getattr(tool, "name", None) - if tool_name: - tool_dict["name"] = tool_name - if tool_name == "web_search_preview": - if hasattr(tool, "user_location"): - tool_dict["user_location"] = tool.user_location - if hasattr(tool, "search_context_size"): - tool_dict["search_context_size"] = tool.search_context_size - elif tool_name == "file_search": - if hasattr(tool, "vector_store_ids"): - tool_dict["vector_store_ids"] = tool.vector_store_ids - if hasattr(tool, "max_num_results"): - tool_dict["max_num_results"] = tool.max_num_results - if hasattr(tool, "include_search_results"): - tool_dict["include_search_results"] = tool.include_search_results - if hasattr(tool, "ranking_options"): - tool_dict["ranking_options"] = tool.ranking_options - if hasattr(tool, "filters"): - tool_dict["filters"] = tool.filters - elif tool_name == "computer_use_preview": - if hasattr(tool, "computer"): - tool_dict["computer"] = tool.computer - if hasattr(tool, "on_safety_check"): - tool_dict["on_safety_check"] = tool.on_safety_check - elif tool_name == "code_interpreter": - if hasattr(tool, "tool_config"): - tool_dict["tool_config"] = tool.tool_config - elif tool_name == "hosted_mcp": - if hasattr(tool, "tool_config"): - tool_dict["tool_config"] = tool.tool_config - if hasattr(tool, "on_approval_request"): - tool_dict["on_approval_request"] = tool.on_approval_request - elif tool_name == "image_generation": - if hasattr(tool, "tool_config"): - tool_dict["tool_config"] = tool.tool_config - elif tool_name == "local_shell": - if hasattr(tool, "executor"): - tool_dict["executor"] = tool.executor - else: - if hasattr(tool, "description"): - tool_dict["description"] = tool.description - if hasattr(tool, "strict_json_schema"): - tool_dict["strict_json_schema"] = tool.strict_json_schema - if hasattr(tool, "params_json_schema"): - parameter_schema = tool.params_json_schema - required_params = {param: True for param in parameter_schema.get("required", [])} - parameters = {} - for param, schema in parameter_schema.get("properties", {}).items(): - param_dict = {} - if "type" in schema: - param_dict["type"] = schema["type"] - if "title" in schema: - param_dict["title"] = schema["title"] - if param in required_params: - param_dict["required"] = True - parameters[param] = param_dict - tool_dict["parameters"] = parameters - tools.append(tool_dict) - - return tools - - def _extract_handoffs_from_agent(self, agent): - if not hasattr(agent, "handoffs") or not agent.handoffs: - return None - handoffs = [] - for handoff in agent.handoffs: - handoff_dict = {} - if hasattr(handoff, "handoff_description") or hasattr(handoff, "tool_description"): - handoff_dict["handoff_description"] = getattr(handoff, "handoff_description", None) or getattr( - handoff, "tool_description", None - ) - if hasattr(handoff, "name") or hasattr(handoff, "agent_name"): - handoff_dict["agent_name"] = getattr(handoff, "name", None) or getattr(handoff, "agent_name", None) - if hasattr(handoff, "tool_name"): - handoff_dict["tool_name"] = handoff.tool_name - if handoff_dict: - handoffs.append(handoff_dict) - - return handoffs - - def _extract_guardrails_from_agent(self, agent): - guardrails = [] - if hasattr(agent, "input_guardrails"): - guardrails.extend([getattr(guardrail, "name", "") for guardrail in agent.input_guardrails]) - if hasattr(agent, "output_guardrails"): - guardrails.extend([getattr(guardrail, "name", "") for guardrail in agent.output_guardrails]) - return guardrails +def _hosted_tool(tool: Any, name: Any) -> Optional[dict[str, Any]]: + if not isinstance(name, str) or not name: + return None + entry: dict[str, Any] = {"name": name} + for field in _HOSTED_TOOL_FIELDS.get(name, ()): + entry[field] = config_value(getattr(tool, field, None)) + if name == "hosted_mcp": + tool_config = getattr(tool, "tool_config", None) + if isinstance(tool_config, dict): + for key in _HOSTED_MCP_CONFIG_KEYS: + entry[key] = config_value(tool_config.get(key)) + return entry + + +def _manifest_capabilities(agent: Any) -> AgentManifest: + capabilities: list[AgentCapability] = [] + for server in getattr(agent, "mcp_servers", None) or []: + name = as_str(getattr(server, "name", None)) + if name: + capabilities.append({"name": name, "type": "mcp"}) + return {"capabilities": capabilities} + + +def _manifest_data_contracts(agent: Any) -> AgentManifest: + output_type = getattr(agent, "output_type", None) + if output_type is None or output_type is str: + return {} + if isinstance(output_type, type) or get_origin(output_type) is not None: + name = type_name(output_type) + else: + # An AgentOutputSchemaBase wraps the declared type; its class name is all that is stable. + name = type(output_type).__name__ + return {"data_contracts": {"output": {"name": name}}} + + +def _manifest_handoffs(agent: Any) -> AgentManifest: + handoffs: list[dict[str, Any]] = [] + for handoff in getattr(agent, "handoffs", None) or []: + # A bare Agent describes itself through handoff_description; a Handoff carries the tool + # description the calling model actually sees. + entry = { + "agent_name": as_str(getattr(handoff, "agent_name", None) or getattr(handoff, "name", None)), + "tool_name": as_str(getattr(handoff, "tool_name", None)), + "handoff_description": as_str( + getattr(handoff, "handoff_description", None) or getattr(handoff, "tool_description", None) + ), + } + if entry["agent_name"]: + handoffs.append(entry) + return {"handoffs": handoffs} + + +def _manifest_guardrails(agent: Any) -> AgentManifest: + guardrails: list[str] = [] + for guardrail in list(getattr(agent, "input_guardrails", None) or []) + list( + getattr(agent, "output_guardrails", None) or [] + ): + name = getattr(guardrail, "name", None) + fn = getattr(guardrail, "guardrail_function", None) + if not name and callable(fn): + name = callable_name(fn) + if isinstance(name, str) and name: + guardrails.append(name) + return {"guardrails": guardrails} + + +def _manifest_agent_settings(agent: Any) -> AgentManifest: + settings: dict[str, Any] = {} + behavior = getattr(agent, "tool_use_behavior", None) + if isinstance(behavior, str): + settings["tool_use_behavior"] = behavior + elif isinstance(behavior, dict): + settings["tool_use_behavior"] = config_value(behavior) + elif callable(behavior): + settings["tool_use_behavior"] = callable_name(behavior) + reset_tool_choice = getattr(agent, "reset_tool_choice", None) + if isinstance(reset_tool_choice, bool): + settings["reset_tool_choice"] = reset_tool_choice + return {"agent_settings": settings} diff --git a/ddtrace/llmobs/_integrations/pydantic_ai.py b/ddtrace/llmobs/_integrations/pydantic_ai.py index 360516d799e..34610db7f81 100644 --- a/ddtrace/llmobs/_integrations/pydantic_ai.py +++ b/ddtrace/llmobs/_integrations/pydantic_ai.py @@ -8,11 +8,10 @@ from ddtrace.internal.logger import get_logger from ddtrace.internal.utils import get_argument_value from ddtrace.llmobs._constants import DISPATCH_ON_TOOL_CALL -from ddtrace.llmobs._integrations.agent_manifest import ALLOWED_MODEL_SETTINGS_KEYS +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest from ddtrace.llmobs._integrations.agent_manifest import callable_name -from ddtrace.llmobs._integrations.agent_manifest import is_flat_scalar_value +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings from ddtrace.llmobs._integrations.agent_manifest import is_number -from ddtrace.llmobs._integrations.agent_manifest import prune_empty from ddtrace.llmobs._integrations.agent_manifest import type_name from ddtrace.llmobs._integrations.agent_manifest import wire_value from ddtrace.llmobs._integrations.base import BaseLLMIntegration @@ -270,28 +269,25 @@ def _build_agent_manifest(self, agent: Any) -> AgentManifest: declared configuration is read, so the manifest is identical run to run, and a field pydantic-ai does not expose is omitted rather than invented. """ - manifest: AgentManifest = {} - for name, section in ( - ("labels", self._manifest_labels), - ("instructions", self._manifest_instructions), - ("model", self._manifest_model), - ("capabilities", self._manifest_capabilities), - ("data_contracts", self._manifest_data_contracts), - ("memory_policies", self._manifest_memory_policies), - ("guardrails", self._manifest_guardrails), - ("agent_settings", self._manifest_agent_settings), - ): - try: - manifest.update(section(agent)) - except Exception: - log.debug("failed to build pydantic_ai agent manifest section %s", name, exc_info=True) - # Sections assign unconditionally so mypy can check every key name against the type; one - # prune here is what drops the fields that mean "not configured". - return prune_empty(manifest) + return build_agent_manifest( + FRAMEWORK_NAME, + agent, + ( + ("labels", self._manifest_labels), + ("instructions", self._manifest_instructions), + ("model", self._manifest_model), + ("capabilities", self._manifest_capabilities), + ("data_contracts", self._manifest_data_contracts), + ("memory_policies", self._manifest_memory_policies), + ("guardrails", self._manifest_guardrails), + ("agent_settings", self._manifest_agent_settings), + ), + self._integration_name, + ) def _manifest_labels(self, agent: Any) -> AgentManifest: """Labels that name the agent. Grouped for failure isolation only; the manifest is flat.""" - fields: AgentManifest = {"framework": FRAMEWORK_NAME} + fields: AgentManifest = {} # placeholder per review, matching the span name fallback. Two unnamed agents # therefore share it, so name is not an identity. agent_name = getattr(agent, "name", None) @@ -336,15 +332,7 @@ def _manifest_model(self, agent: Any) -> AgentManifest: # reprs what it cannot encode, which can carry a connection string. if isinstance(model_name, str): fields["model"] = model_name - settings = getattr(agent, "model_settings", None) - if isinstance(settings, dict): - allowed: dict[str, Any] = {} - for key, value in settings.items(): - if key not in ALLOWED_MODEL_SETTINGS_KEYS or not is_flat_scalar_value(value): - continue - # prune_empty drops what wire_value could not encode, so assign it either way. - allowed[key] = wire_value(value) - fields["model_settings"] = allowed + fields["model_settings"] = filter_model_settings(getattr(agent, "model_settings", None)) return fields def _manifest_capabilities(self, agent: Any) -> AgentManifest: diff --git a/ddtrace/llmobs/types.py b/ddtrace/llmobs/types.py index d7e61075ac6..6a8298ce6b4 100644 --- a/ddtrace/llmobs/types.py +++ b/ddtrace/llmobs/types.py @@ -97,13 +97,15 @@ class AgentManifest(TypedDict, total=False): system_prompts: list[str] extra_instructions: list[AgentInstructionResolver] model: str + model_provider: str model_settings: dict[str, Any] agent_settings: dict[str, Any] tools: list[dict[str, Any]] capabilities: list[AgentCapability] data_contracts: dict[str, Any] guardrails: list[str] - handoffs: list[Any] + # A list of targets, or {"allow_delegation": bool} for frameworks that only report a flag. + handoffs: Union[list[dict[str, Any]], dict[str, Any]] handoff_description: str memory_policies: list[str] metadata: dict[str, Any] diff --git a/releasenotes/notes/llmobs-standardize-agent-manifest-df86615c800061b2.yaml b/releasenotes/notes/llmobs-standardize-agent-manifest-df86615c800061b2.yaml new file mode 100644 index 00000000000..906db0c95c5 --- /dev/null +++ b/releasenotes/notes/llmobs-standardize-agent-manifest-df86615c800061b2.yaml @@ -0,0 +1,17 @@ +--- +features: + - | + LLM Observability: CrewAI, Google ADK, LangGraph, OpenAI Agents, and Claude Agent SDK agent spans + now report agent configuration with a consistent set of fields, including instructions, tool + parameters, handoffs, and settings where the framework declares them. +fixes: + - | + LLM Observability: Fixes an issue where traces fail to be sent in agentless mode with + ``TypeError: Object of type function is not JSON serializable`` when a Google ADK agent uses a + callable instruction provider. + - | + LLM Observability: Fixes an issue where the agent configuration reported for OpenAI Agents could + include ``extra_headers``, ``extra_body``, and hosted MCP tool headers. + - | + LLM Observability: Fixes an issue where the model was missing from the Google ADK agent + configuration when the agent's model is set as a string. diff --git a/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py b/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py index 565f08ff3bf..0538ed592ef 100644 --- a/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py +++ b/tests/contrib/claude_agent_sdk/test_claude_agent_sdk_llmobs.py @@ -141,7 +141,7 @@ async def test_llmobs_query_with_options( metadata={ "max_turns": 3, "stop_reason": "end_turn", - "_dd": {"agent_manifest": expected_agent_manifest(max_iterations=3)}, + "_dd": {"agent_manifest": expected_agent_manifest(max_turns=3)}, }, metrics=EXPECTED_QUERY_USAGE, tags=COMMON_TAGS, @@ -236,7 +236,7 @@ async def test_llmobs_query_error_no_output( span_kind="agent", input_value=safe_json(input_msgs), output_value=safe_json([{"content": ""}]), - metadata={"_dd": {"agent_manifest": {"framework": "Claude Agent SDK"}}}, + metadata={}, metrics={}, tags=COMMON_TAGS, error={"type": "builtins.ValueError", "message": "Connection failed", "stack": ANY}, diff --git a/tests/contrib/claude_agent_sdk/utils.py b/tests/contrib/claude_agent_sdk/utils.py index b6089dbe794..46b75806237 100644 --- a/tests/contrib/claude_agent_sdk/utils.py +++ b/tests/contrib/claude_agent_sdk/utils.py @@ -40,7 +40,7 @@ } -def expected_agent_manifest(max_iterations=None): +def expected_agent_manifest(max_turns=None): """Helper to build expected agent manifest.""" manifest = { "framework": "Claude Agent SDK", @@ -52,10 +52,9 @@ def expected_agent_manifest(max_iterations=None): {"name": "Write"}, {"name": "Grep"}, ], - "dependencies": {"mcp_servers": []}, } - if max_iterations is not None: - manifest["max_iterations"] = max_iterations + if max_turns is not None: + manifest["agent_settings"] = {"max_turns": max_turns} return manifest diff --git a/tests/contrib/crewai/test_crewai_llmobs.py b/tests/contrib/crewai/test_crewai_llmobs.py index 376dc74141d..400c08d0fd9 100644 --- a/tests/contrib/crewai/test_crewai_llmobs.py +++ b/tests/contrib/crewai/test_crewai_llmobs.py @@ -21,60 +21,49 @@ "Senior Research Scientist": { "framework": "CrewAI", "name": "Senior Research Scientist", - "goal": "Uncover cutting-edge developments in AI", - "backstory": "You're a seasoned researcher with a knack for uncovering the latest developments in AI. " + "instructions": "Uncover cutting-edge developments in AI\n\n" + "You're a seasoned researcher with a knack for uncovering the latest developments in AI. " "Known for your ability to find the most relevant information and present it in a clear " "and concise manner.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, - "tools": [], + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, }, "AI Reporting Analyst": { "framework": "CrewAI", "name": "AI Reporting Analyst", - "goal": "Create detailed reports based on AI data analysis and research findings", - "backstory": "You're a meticulous analyst with a keen eye for detail. You're known for your ability to turn " + "instructions": "Create detailed reports based on AI data analysis and research findings\n\n" + "You're a meticulous analyst with a keen eye for detail. You're known for your ability to turn " "complex data into clear and concise reports, making it easy for others to understand and act on the " "information you provide.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, - "tools": [], + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, }, "Python Data Analyst": { "framework": "CrewAI", "name": "Python Data Analyst", - "goal": "Analyze data and provide insights using Python", - "backstory": "You are an experienced data analyst with strong Python skills.", + "instructions": "Analyze data and provide insights using Python\n\n" + "You are an experienced data analyst with strong Python skills.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, "tools": [ { "name": "Average Calculator", - "description": "Tool Name: Average Calculator\nTool Arguments: {'entries': {'description': None, " - "'type': 'list'}}\nTool Description: This tool returns the average of a list of numbers.", + "description": "This tool returns the average of a list of numbers.", + "parameters": {"entries": {"type": "array", "required": True}}, } ], }, "Tour Guide": { "framework": "CrewAI", "name": "Tour Guide", - "goal": "Recommend fun activities for a group of humans.", - "backstory": "You are a tour guide with a passion for finding the best activities for groups of people.", + "instructions": "Recommend fun activities for a group of humans.\n\n" + "You are a tour guide with a passion for finding the best activities for groups of people.", "model": "gpt-4o-mini", - "model_settings": {"max_tokens": None, "temperature": None}, "handoffs": {"allow_delegation": False}, - "code_execution_permissions": {"code_execution_mode": "safe"}, - "max_iterations": 25, - "tools": [], + "agent_settings": {"max_iter": 25, "max_retry_limit": 2}, }, } diff --git a/tests/contrib/google_adk/test_google_adk_llmobs.py b/tests/contrib/google_adk/test_google_adk_llmobs.py index 7847a8d7ee4..7835c2391ab 100644 --- a/tests/contrib/google_adk/test_google_adk_llmobs.py +++ b/tests/contrib/google_adk/test_google_adk_llmobs.py @@ -18,7 +18,7 @@ AGENT_MANIFEST_METADATA = { "_dd": { "agent_manifest": { - "description": "Test agent for ADK integration testing", + "handoff_description": "Test agent for ADK integration testing", "framework": "Google ADK", "instructions": "You are a helpful test agent. You can: " "(1) call tools using the provided " @@ -29,17 +29,23 @@ "capability. Always be helpful and use " "your available capabilities.", "model": "gemini-2.5-pro", - "model_configuration": '{"arbitrary_types_allowed": true, "extra": "forbid"}', "name": "test_agent", - "session_management": { - "session_id": "test-session", - "user_id": "test-user", - "app_name": "TestADKApp", - }, "tools": [ - {"description": "A tiny search tool stub.", "name": "search_docs"}, - {"description": "Simple arithmetic tool.", "name": "multiply"}, + { + "description": "A tiny search tool stub.", + "name": "search_docs", + "parameters": {"query": {"type": "string", "required": True}}, + }, + { + "description": "Simple arithmetic tool.", + "name": "multiply", + "parameters": { + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, + }, + }, ], + "agent_settings": {"code_executor": "UnsafeLocalCodeExecutor"}, } } } diff --git a/tests/contrib/langgraph/test_langgraph_llmobs.py b/tests/contrib/langgraph/test_langgraph_llmobs.py index a16f73ac185..3153f82872d 100644 --- a/tests/contrib/langgraph/test_langgraph_llmobs.py +++ b/tests/contrib/langgraph/test_langgraph_llmobs.py @@ -378,26 +378,17 @@ def test_agent_manifest_simple_graph( expected_agent_a_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list", "which"], "name": "agent_a", - "tools": [], } expected_conditional_agent_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list", "which"], "name": conditional_agent_name, - "tools": [], } expected_agent_d_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list", "which"], "name": "agent_d", - "tools": [], } agent_a_metadata = get_llmobs_metadata(agent_a_span) or {} @@ -419,22 +410,14 @@ def test_agent_manifest_from_create_react_agent(self, langgraph_llmobs, test_spa expected_agent_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["messages"], "name": "not_your_average_bostonian", "tools": [ { "name": "add", "description": "Adds two numbers together", "parameters": { - "a": { - "title": "A", - "type": "integer", - }, - "b": { - "title": "B", - "type": "integer", - }, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -460,22 +443,14 @@ def test_agent_manifest_populates_tools_from_tool_node( expected_agent_manifest = { "framework": "LangGraph", - "max_iterations": 25, - "dependencies": ["a_list"], "name": "custom_agent_with_tool_node", "tools": [ { "name": "add", "description": "Adds two numbers together", "parameters": { - "a": { - "title": "A", - "type": "integer", - }, - "b": { - "title": "B", - "type": "integer", - }, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -496,7 +471,8 @@ def test_agent_manifest_different_recursion_limit( agent_span = _find_span_by_name(spans, "agent") agent_metadata = get_llmobs_metadata(agent_span) or {} - assert agent_metadata.get("_dd", {}).get("agent_manifest", {}).get("max_iterations") == 100 + manifest = agent_metadata.get("_dd", {}).get("agent_manifest", {}) + assert manifest.get("agent_settings", {}).get("recursion_limit") == 100 @pytest.mark.skipif(LANGGRAPH_VERSION < (0, 3, 22), reason="Agent names are only supported in LangGraph 0.3.22+") def test_agent_with_tool_calls_integrations_enabled( diff --git a/tests/contrib/openai_agents/test_openai_agents_llmobs.py b/tests/contrib/openai_agents/test_openai_agents_llmobs.py index c728196def3..43e66b3c0a0 100644 --- a/tests/contrib/openai_agents/test_openai_agents_llmobs.py +++ b/tests/contrib/openai_agents/test_openai_agents_llmobs.py @@ -28,25 +28,22 @@ "framework": "OpenAI", "name": "Simple Agent", "instructions": "You are a helpful assistant who answers questions concisely and accurately.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, # different versions of the library have different model settings + "agent_settings": mock.ANY, }, "Simple Agent with Guardrails": { "framework": "OpenAI", "name": "Simple Agent", "instructions": "You are a helpful assistant specialized in addition calculations.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, # different versions of the library have different model settings + "agent_settings": mock.ANY, "tools": [ { "name": "add", "description": "Add two numbers together", - "strict_json_schema": True, "parameters": { - "a": {"type": "integer", "title": "A", "required": True}, - "b": {"type": "integer", "title": "B", "required": True}, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -56,17 +53,15 @@ "framework": "OpenAI", "name": "Addition Agent", "instructions": mock.ANY, - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, "tools": [ { "name": "add", "description": "Add two numbers together", - "strict_json_schema": True, "parameters": { - "a": {"type": "integer", "title": "A", "required": True}, - "b": {"type": "integer", "title": "B", "required": True}, + "a": {"type": "integer", "required": True}, + "b": {"type": "integer", "required": True}, }, } ], @@ -76,36 +71,32 @@ "name": "Researcher", "instructions": "You are a helpful assistant that can research a topic using your research tool. " "Always research the topic before summarizing.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, "tools": [ { "name": "research", "description": "Research the internet on a topic.", - "strict_json_schema": True, - "parameters": {"query": {"type": "string", "title": "Query", "required": True}}, + "parameters": {"query": {"type": "string", "required": True}}, } ], "handoffs": [ - {"handoff_description": None, "agent_name": "Summarizer"}, + {"agent_name": "Summarizer"}, ], }, "Summarizer": { "framework": "OpenAI", "name": "Summarizer", "instructions": "You are a helpful assistant that can summarize a research results.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, }, "Weather Agent": { "framework": "OpenAI", "name": "Weather Agent", "instructions": "You are a helpful assistant specialized in searching the web for weather information.", - "handoff_description": None, "model": "gpt-4o", - "model_settings": mock.ANY, + "agent_settings": mock.ANY, "tools": [ { "name": "web_search_preview", diff --git a/tests/llmobs/test_agent_manifest_integrations.py b/tests/llmobs/test_agent_manifest_integrations.py new file mode 100644 index 00000000000..891212c8c74 --- /dev/null +++ b/tests/llmobs/test_agent_manifest_integrations.py @@ -0,0 +1,357 @@ +"""Agent manifest contract across integrations, exercised with stand-in framework objects. + +Each builder reads attributes off a framework object, so a SimpleNamespace stands in for one and +these tests need no framework installed. +""" + +import json +from types import SimpleNamespace +from unittest import mock + +import pytest + +from ddtrace.llmobs._constants import LLMOBS_STRUCT +from ddtrace.llmobs._integrations.agent_manifest import build_agent_manifest +from ddtrace.llmobs._integrations.agent_manifest import filter_model_settings +from ddtrace.llmobs._integrations.agent_manifest import instruction_fields +from ddtrace.llmobs._integrations.agent_manifest import normalize_tool +from ddtrace.llmobs._integrations.claude_agent_sdk import ClaudeAgentSdkIntegration +from ddtrace.llmobs._integrations.crewai import CrewAIIntegration +from ddtrace.llmobs._integrations.google_adk import GoogleAdkIntegration +from ddtrace.llmobs._integrations.langgraph import LangGraphIntegration +from ddtrace.llmobs._integrations.openai_agents import OpenAIAgentsIntegration +from ddtrace.llmobs._utils import _get_llmobs_data_metastruct +from ddtrace.llmobs.types import AgentManifest +from ddtrace.trace import Span + + +CANONICAL_KEYS = frozenset(AgentManifest.__annotations__) + + +def _never_called(*args, **kwargs): + raise AssertionError("a declared callable must not be invoked while building the manifest") + + +def _search(query: str, limit: int = 5, tool_context=None) -> list: + """Search the docs.""" + + +class _Answer: + pass + + +class _Graph: + """Stands in for a compiled graph, which the integration keys weakly.""" + + def __init__(self, name): + self.name = name + self.builder = None + + +def _manifest_from_span(span): + meta = _get_llmobs_data_metastruct(span).get(LLMOBS_STRUCT.META, {}) + return meta.get(LLMOBS_STRUCT.METADATA, {}).get(LLMOBS_STRUCT.METADATA_DD, {}).get(LLMOBS_STRUCT.AGENT_MANIFEST) + + +def _adk_agent(**overrides): + agent = dict( + name="billing", + description="Handles billing questions", + model="gemini-2.0-flash", + instruction=_never_called, + global_instruction="Be polite.", + static_instruction=None, + tools=[_search], + generate_content_config=SimpleNamespace(temperature=0.2, max_output_tokens=256), + output_schema=_Answer, + sub_agents=[SimpleNamespace(name="refunds", description="Issues refunds")], + before_model_callback=None, + before_tool_callback=None, + model_config={"arbitrary_types_allowed": True, "extra": "forbid"}, + ) + agent.update(overrides) + return SimpleNamespace(**agent) + + +def _openai_agent(**overrides): + agent = dict( + name="triage", + instructions="Route the request.", + prompt=None, + handoff_description="Routes requests", + model="gpt-4o", + model_settings=SimpleNamespace(temperature=0.1, extra_headers={"Authorization": "Bearer secret"}), + tools=[ + SimpleNamespace( + name="lookup", + description="Look up an order", + params_json_schema={ + "type": "object", + "properties": {"order_id": {"type": "string", "title": "Order Id"}}, + "required": ["order_id"], + }, + ), + SimpleNamespace( + name="hosted_mcp", + tool_config={ + "server_label": "github", + "server_url": "https://mcp.example.com?token=secret", + "headers": {"Authorization": "Bearer secret"}, + "allowed_tools": ["search"], + }, + on_approval_request=_never_called, + ), + SimpleNamespace(name="computer_use_preview", computer=object(), on_safety_check=_never_called), + ], + mcp_servers=[], + output_type=None, + handoffs=[SimpleNamespace(name="refunds", handoff_description="Issues refunds", tools=[], handoffs=[])], + input_guardrails=[SimpleNamespace(name=None, guardrail_function=_never_called)], + output_guardrails=[], + tool_use_behavior="run_llm_again", + reset_tool_choice=True, + ) + agent.update(overrides) + return SimpleNamespace(**agent) + + +def _crewai_agent(**overrides): + agent = dict( + role="AI Researcher", + goal="Research AI", + backstory="An expert in AI.", + llm=SimpleNamespace(model="gpt-4o-mini", temperature=0.0, max_tokens=None), + tools=[], + allow_delegation=False, + max_iter=25, + allow_code_execution=False, + code_execution_mode="safe", + ) + agent.update(overrides) + return SimpleNamespace(**agent) + + +def _claude_options(**overrides): + options = dict( + system_prompt="You are a code reviewer.", + mcp_servers={"fs": {"command": "fs-server", "env": {"TOKEN": "secret"}}}, + agents={"tester": SimpleNamespace(description="Writes tests")}, + can_use_tool=_never_called, + hooks=None, + max_turns=3, + ) + options.update(overrides) + return SimpleNamespace(**options) + + +def _build(integration_name, *args): + """Build a manifest the way each integration does on its agent span.""" + span = Span("agent") + if integration_name == "google_adk": + GoogleAdkIntegration(integration_config=mock.MagicMock())._tag_agent_manifest(span, {}, *args) + elif integration_name == "openai_agents": + integration = OpenAIAgentsIntegration(integration_config=mock.MagicMock()) + with mock.patch.object( + OpenAIAgentsIntegration, "llmobs_enabled", new_callable=mock.PropertyMock, return_value=True + ): + integration._tag_agent_manifest_from_agent(span, *args) + elif integration_name == "crewai": + CrewAIIntegration(integration_config=mock.MagicMock())._tag_agent_manifest(span, *args) + elif integration_name == "claude_agent_sdk": + return ClaudeAgentSdkIntegration(integration_config=mock.MagicMock())._build_agent_manifest(*args) + return _manifest_from_span(span) + + +CONTRACT_CASES = [ + ("google_adk", lambda: (_adk_agent(),)), + ("openai_agents", lambda: (_openai_agent(),)), + ("crewai", lambda: (_crewai_agent(),)), + ("claude_agent_sdk", lambda: ("claude-sonnet", _claude_options(), {"tools": ["Read"], "mcp_servers": []})), +] + + +@pytest.mark.parametrize("integration_name,make_args", CONTRACT_CASES, ids=[c[0] for c in CONTRACT_CASES]) +def test_manifest_contract(integration_name, make_args): + manifest = _build(integration_name, *make_args()) + + assert manifest, "the stand-in agent declares enough to report a manifest" + assert set(manifest) <= CANONICAL_KEYS, set(manifest) - CANONICAL_KEYS + assert manifest["framework"] + # Round-trips through JSON unchanged, so nothing relies on the encoder's repr fallback. + assert json.loads(json.dumps(manifest)) == manifest + # Rebuilt from a fresh but identical declaration, so nothing per process (an address) or per run leaks in. + assert _build(integration_name, *make_args()) == manifest + + +class TestSharedHelpers: + def test_callable_instructions_ship_by_name(self): + assert instruction_fields(_never_called) == { + "extra_instructions": [{"type": "dynamic_instructions", "name": "_never_called"}] + } + assert instruction_fields("Be brief.") == {"instructions": "Be brief."} + assert instruction_fields(object()) == {} + + def test_model_settings_allowlist(self): + assert filter_model_settings( + {"temperature": 0, "extra_headers": {"Authorization": "x"}, "extra_body": {"a": 1}} + ) == {"temperature": 0} + assert filter_model_settings(None) == {} + + def test_normalize_tool_flattens_json_schema(self): + assert normalize_tool( + "add", "Adds", {"type": "object", "properties": {"a": {"type": "integer", "title": "A"}}, "required": ["a"]} + ) == {"name": "add", "description": "Adds", "parameters": {"a": {"type": "integer", "required": True}}} + assert normalize_tool("", "unnamed") is None + assert normalize_tool("t", object())["description"] == "" + + def test_failing_section_costs_only_its_fields(self): + def broken(_): + raise RuntimeError("framework changed") + + manifest = build_agent_manifest("X", None, (("labels", lambda _: {"name": "a"}), ("tools", broken)), "test") + assert manifest == {"name": "a", "framework": "X"} + + def test_empty_manifest_is_not_reported(self): + assert build_agent_manifest("X", None, (("labels", lambda _: {"name": ""}),), "test") == {} + + +class TestGoogleAdk: + def test_manifest(self): + assert _build("google_adk", _adk_agent()) == { + "framework": "Google ADK", + "name": "billing", + "handoff_description": "Handles billing questions", + "extra_instructions": [{"type": "dynamic_instructions", "name": "_never_called"}], + "system_prompts": ["Be polite."], + "model": "gemini-2.0-flash", + "model_settings": {"temperature": 0.2, "max_tokens": 256}, + "tools": [ + { + "name": "_search", + "description": "Search the docs.", + "parameters": {"query": {"type": "string", "required": True}, "limit": {"type": "integer"}}, + } + ], + "data_contracts": {"output": {"name": "_Answer"}}, + "handoffs": [{"agent_name": "refunds", "handoff_description": "Issues refunds"}], + } + + def test_model_object(self): + manifest = _build("google_adk", _adk_agent(model=SimpleNamespace(model="gemini-2.5-pro"))) + assert manifest["model"] == "gemini-2.5-pro" + + def test_static_instruction_content(self): + content = SimpleNamespace(parts=[SimpleNamespace(text="Cached preamble.")]) + manifest = _build("google_adk", _adk_agent(global_instruction="", static_instruction=content)) + assert manifest["system_prompts"] == ["Cached preamble."] + + +class TestOpenAIAgents: + def test_manifest(self): + assert _build("openai_agents", _openai_agent()) == { + "framework": "OpenAI", + "name": "triage", + "handoff_description": "Routes requests", + "instructions": "Route the request.", + "model": "gpt-4o", + "model_settings": {"temperature": 0.1}, + "tools": [ + { + "name": "lookup", + "description": "Look up an order", + "parameters": {"order_id": {"type": "string", "required": True}}, + }, + {"name": "hosted_mcp", "server_label": "github", "allowed_tools": ["search"]}, + {"name": "computer_use_preview"}, + ], + "handoffs": [{"agent_name": "refunds", "handoff_description": "Issues refunds"}], + "guardrails": ["_never_called"], + "agent_settings": {"tool_use_behavior": "run_llm_again", "reset_tool_choice": True}, + } + + def test_callable_instructions_and_stored_prompt(self): + manifest = _build("openai_agents", _openai_agent(instructions=_never_called, prompt={"id": "pmpt_123"})) + assert "instructions" not in manifest + assert manifest["extra_instructions"] == [ + {"type": "dynamic_instructions", "name": "_never_called"}, + {"type": "prompt", "name": "pmpt_123"}, + ] + + +class TestCrewAI: + def test_manifest(self): + assert _build("crewai", _crewai_agent()) == { + "framework": "CrewAI", + "name": "AI Researcher", + "instructions": "Research AI\n\nAn expert in AI.", + "model": "gpt-4o-mini", + "model_settings": {"temperature": 0.0}, + "handoffs": {"allow_delegation": False}, + "agent_settings": {"max_iter": 25}, + } + + def test_code_execution_settings(self): + manifest = _build("crewai", _crewai_agent(allow_code_execution=True, code_execution_mode="unsafe")) + assert manifest["agent_settings"] == { + "max_iter": 25, + "allow_code_execution": True, + "code_execution_mode": "unsafe", + } + + +class TestLangGraph: + @pytest.mark.parametrize( + "model,expected", + [ + ("gpt-4o", {"model": "gpt-4o"}), + ("openai:gpt-4o", {"model": "gpt-4o", "model_provider": "openai"}), + ], + ) + def test_react_agent_model_string(self, model, expected): + integration = LangGraphIntegration(integration_config=mock.MagicMock()) + agent = _Graph("react") + with mock.patch.object( + LangGraphIntegration, "llmobs_enabled", new_callable=mock.PropertyMock, return_value=True + ): + integration.llmobs_handle_agent_manifest(agent, (model, []), {"name": "react", "prompt": _never_called}) + manifest = integration._get_agent_manifest(agent, (), {}) + assert manifest == { + "framework": "LangGraph", + "name": "react", + "extra_instructions": [{"type": "dynamic_prompt", "name": "_never_called"}], + **expected, + } + + def test_run_config_does_not_leak_into_cache(self): + integration = LangGraphIntegration(integration_config=mock.MagicMock()) + graph = _Graph("graph") + first = integration._get_agent_manifest(graph, (), {"recursion_limit": 100}) + second = integration._get_agent_manifest(graph, (), {}) + assert first["agent_settings"] == {"recursion_limit": 100} + assert second == {"framework": "LangGraph", "name": "graph"} + + +class TestClaudeAgentSdk: + def test_manifest(self): + manifest = _build( + "claude_agent_sdk", + "claude-sonnet", + _claude_options(), + {"tools": ["Read", "Bash"], "mcp_servers": [{"name": "fs", "status": "connected"}]}, + ) + assert manifest == { + "framework": "Claude Agent SDK", + "model": "claude-sonnet", + "instructions": "You are a code reviewer.", + "tools": [{"name": "Read"}, {"name": "Bash"}], + "capabilities": [{"name": "fs", "type": "mcp"}], + "handoffs": [{"agent_name": "tester", "handoff_description": "Writes tests"}], + "guardrails": ["_never_called"], + "agent_settings": {"max_turns": 3}, + } + + def test_preset_system_prompt(self): + options = _claude_options(system_prompt={"type": "preset", "preset": "claude_code", "append": "Be terse."}) + manifest = _build("claude_agent_sdk", "claude-sonnet", options, {}) + assert manifest["instructions"] == "Be terse." + assert manifest["extra_instructions"] == [{"type": "preset", "name": "claude_code"}]