Skip to content
Judgment Labs
Esc
↑↓navigate↵open⌘Jpreview
On this page

Claude Agent SDK Tracing

Automatically trace Claude Agent SDK executions, tool usage, and agent interactions.

Claude Agent SDK integration captures traces from your Claude Agent applications, including agent execution flow, LLM interactions, and tool calls. Tool spans are captured for all tools — both built-in Claude Code tools (Bash, file_edit, computer use, etc.) and custom MCP tools — automatically from the agent message stream.

Quickstart

Install Dependencies

uv add claude-agent-sdk judgeval python-dotenv
pip install claude-agent-sdk judgeval python-dotenv

Initialize Integration

from judgeval import Tracer
from judgeval.integrations.claude_agent_sdk import setup_claude_agent_sdk

Tracer.init(project_name="claude_agents_project")
setup_claude_agent_sdk()

Add to Existing Code

Add these lines to your existing Claude Agent SDK application:

import dotenv
import asyncio

dotenv.load_dotenv()

from judgeval import Tracer  
from judgeval.integrations.claude_agent_sdk import setup_claude_agent_sdk  

Tracer.init(project_name="claude_agents_project")  
setup_claude_agent_sdk()  

from claude_agent_sdk import ClaudeSDKClient, ClaudeAgentOptions

options = ClaudeAgentOptions(
    model="claude-sonnet-4-20250514",
)

async def main():
    async with ClaudeSDKClient(options=options) as client:
        await client.query("Hello, how are you?")

        response_text = []
        async for message in client.receive_response():
            if hasattr(message, "content"):
                for block in message.content:
                    if hasattr(block, "text"):
                        response_text.append(block.text)

        print("".join(response_text))

if __name__ == "__main__":
    asyncio.run(main())

All agent executions and tool calls are automatically traced — including built-in tools like Bash and file_edit, not just custom MCP tools.

Example: Agent with Tool Execution

from typing import Any, Dict

dotenv.load_dotenv()

from judgeval import Tracer
from judgeval.integrations.claude_agent_sdk import setup_claude_agent_sdk

Tracer.init(project_name="claude_agents_project")
setup_claude_agent_sdk()

from claude_agent_sdk import (
    ClaudeSDKClient,
    ClaudeAgentOptions,
    tool,
    create_sdk_mcp_server,
)

# Define a calculator tool
@tool(
    "calculator",
    "Calculates simple math expressions. Input should be a string like '2+2' or '10*5'.",
    {"expression": str},
)
async def calculator(args: Dict[str, Any]) -> Dict[str, Any]:
    """Simple calculator tool."""
    try:
        expression = args.get("expression", "")
        # Security: only allow basic math
        allowed_chars = set("0123456789+-*/(). ")
        if not all(c in allowed_chars for c in expression):
            return {
                "content": [{
                    "type": "text",
                    "text": "Error: Invalid characters in expression",
                }],
                "is_error": True,
            }

        result = eval(expression, {"__builtins__": {}}, {}) # This is just an example, we don't recommend using this code in production as it can lead to security vulnerabilities
        return {"content": [{"type": "text", "text": f"{expression} = {result}"}]}
    except Exception as e:
        return {
            "content": [{"type": "text", "text": f"Error: {str(e)}"}],
            "is_error": True,
        }

# Create MCP server with the tool
calc_server = create_sdk_mcp_server(
    name="math_tools",
    version="1.0.0",
    tools=[calculator]
)

@Tracer.observe(span_type="function")  
async def main():
    options = ClaudeAgentOptions(
        model="claude-sonnet-4-20250514",
        system_prompt="You are a helpful assistant. Use the calculator tool for any math.",
        mcp_servers={"math": calc_server},
        allowed_tools=["mcp__math__calculator"],
        permission_mode="acceptEdits",
    )

    # Make API call with tool
    async with ClaudeSDKClient(options=options) as client:
        await client.query("What is 15 multiplied by 23? Use the calculator tool.")

        response_text = []
        async for message in client.receive_response():
            if hasattr(message, "content"):
                for block in message.content:
                    if hasattr(block, "text"):
                        response_text.append(block.text)

        print("".join(response_text))

if __name__ == "__main__":
    asyncio.run(main())

Standalone Query API

Claude Agent SDK also provides a simpler standalone query() function for quick interactions:


dotenv.load_dotenv()

from judgeval import Tracer
from judgeval.integrations.claude_agent_sdk import setup_claude_agent_sdk

Tracer.init(project_name="claude_agents_project")
setup_claude_agent_sdk()

from claude_agent_sdk import query, ClaudeAgentOptions

async def main():
    options = ClaudeAgentOptions(
        model="claude-sonnet-4-20250514",
        system_prompt="You are a helpful assistant",
        permission_mode="acceptEdits",
    )

    # Use the standalone query function
    response_text = []
    async for message in query(
        prompt="What is the square root of 144?",
        options=options,
    ):
        if hasattr(message, "content"):
            for block in message.content:
                if hasattr(block, "text"):
                    response_text.append(block.text)

    print("".join(response_text))

if __name__ == "__main__":
    asyncio.run(main())

Next Steps

Was this page helpful?