{
  "generatedBy": "bench.mjs",
  "startedAt": "2026-08-24T05:56:12.610Z",
  "finishedAt": "2026-08-24T05:57:54.864Z",
  "machine": {
    "cpu": "Apple M1",
    "os": "Darwin 26.3",
    "arch": "arm64",
    "node": "v22.18.0",
    "python": "Python 3.13.5"
  },
  "metricCount": 210,
  "metrics": [
    {
      "metric": "setup_install_wall_ms",
      "value": 1932,
      "unit": "ms",
      "note": "uv install of pinned SDKs"
    },
    {
      "metric": "experiment_model",
      "value": "same tool, prompt, model, structured schema, and fixture"
    },
    {
      "metric": "model_name",
      "value": "gpt-5.4-mini"
    },
    {
      "metric": "task_timeout",
      "value": 90,
      "unit": "seconds"
    },
    {
      "metric": "parity_max_turns",
      "value": 4,
      "unit": "turns"
    },
    {
      "metric": "parity_max_output_tokens",
      "value": 400,
      "unit": "tokens"
    },
    {
      "metric": "tool_set_size",
      "value": 1,
      "unit": "tool"
    },
    {
      "metric": "repetitions_per_cell",
      "value": 3,
      "unit": "runs"
    },
    {
      "metric": "repo_commit",
      "value": "4bf52cb8ca378769a5312a4b961c1253f6ef9422"
    },
    {
      "metric": "fixture_sha256",
      "value": "b7e5bf17bc5b21cd5d11ffcf171670c55b55945d3333670a7874327594e12a46"
    },
    {
      "metric": "pydantic_ai_version",
      "value": "2.33.0"
    },
    {
      "metric": "openai_agents_version",
      "value": "0.22.0"
    },
    {
      "metric": "python_version",
      "value": "3.13.5"
    },
    {
      "metric": "estimated_cost_input_rate",
      "value": 0.75,
      "unit": "USD/million tokens",
      "note": "OpenAI pricing page checked 2026-08-22"
    },
    {
      "metric": "estimated_cost_output_rate",
      "value": 4.5,
      "unit": "USD/million tokens",
      "note": "OpenAI pricing page checked 2026-08-22"
    },
    {
      "metric": "pydantic_ai_default_run_1_wall_ms",
      "value": 6477.08,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_input_tokens",
      "value": 356,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_output_tokens",
      "value": 50,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_total_tokens",
      "value": 406,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_1_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_wall_ms",
      "value": 2652.93,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_input_tokens",
      "value": 356,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_output_tokens",
      "value": 50,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_total_tokens",
      "value": 406,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_2_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_wall_ms",
      "value": 2682.36,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_input_tokens",
      "value": 356,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_output_tokens",
      "value": 50,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_total_tokens",
      "value": 406,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_run_3_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_default_runs",
      "value": 3,
      "unit": "runs"
    },
    {
      "metric": "pydantic_ai_default_median_wall_ms",
      "value": 2682.36,
      "unit": "ms",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_structured",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_correct",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_tool_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_requests",
      "value": 2,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_retries_estimated",
      "value": 0,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_input_tokens",
      "value": 356,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_output_tokens",
      "value": 50,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_total_tokens",
      "value": 406,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_default_median_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_wall_ms",
      "value": 3034.64,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_input_tokens",
      "value": 456,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_output_tokens",
      "value": 45,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_total_tokens",
      "value": 501,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_1_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_wall_ms",
      "value": 2512.77,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_input_tokens",
      "value": 456,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_output_tokens",
      "value": 45,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_total_tokens",
      "value": 501,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_2_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_wall_ms",
      "value": 3363.88,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_input_tokens",
      "value": 456,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_output_tokens",
      "value": 45,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_total_tokens",
      "value": 501,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_run_3_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_default_runs",
      "value": 3,
      "unit": "runs"
    },
    {
      "metric": "openai_agents_sdk_default_median_wall_ms",
      "value": 3034.64,
      "unit": "ms",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_structured",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_correct",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_tool_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_requests",
      "value": 2,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_retries_estimated",
      "value": 0,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_input_tokens",
      "value": 456,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_output_tokens",
      "value": 45,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_total_tokens",
      "value": 501,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_default_median_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_run_1_wall_ms",
      "value": 8662.74,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_input_tokens",
      "value": 356,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_output_tokens",
      "value": 50,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_total_tokens",
      "value": 406,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_1_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_wall_ms",
      "value": 16522.88,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_input_tokens",
      "value": 356,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_output_tokens",
      "value": 50,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_total_tokens",
      "value": 406,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_2_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_wall_ms",
      "value": 8985.54,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_input_tokens",
      "value": 356,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_output_tokens",
      "value": 50,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_total_tokens",
      "value": 406,
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_run_3_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "pydantic_ai_parity_runs",
      "value": 3,
      "unit": "runs"
    },
    {
      "metric": "pydantic_ai_parity_median_wall_ms",
      "value": 8985.54,
      "unit": "ms",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_structured",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_correct",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_tool_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_requests",
      "value": 2,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_retries_estimated",
      "value": 0,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_input_tokens",
      "value": 356,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_output_tokens",
      "value": 50,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_total_tokens",
      "value": 406,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "pydantic_ai_parity_median_estimated_cost_usd",
      "value": 0.000492,
      "unit": "USD",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_wall_ms",
      "value": 15779.32,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_input_tokens",
      "value": 456,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_output_tokens",
      "value": 45,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_total_tokens",
      "value": 501,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_1_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_wall_ms",
      "value": 10232.88,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_input_tokens",
      "value": 456,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_output_tokens",
      "value": 45,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_total_tokens",
      "value": 501,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_2_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_wall_ms",
      "value": 9431.39,
      "unit": "ms",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_structured",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_correct",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_tool_calls",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_tool_success",
      "value": 1,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_requests",
      "value": 2,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_retries_estimated",
      "value": 0,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_input_tokens",
      "value": 456,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_output_tokens",
      "value": 45,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_total_tokens",
      "value": 501,
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_run_3_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "note": "one real SDK run"
    },
    {
      "metric": "openai_agents_sdk_parity_runs",
      "value": 3,
      "unit": "runs"
    },
    {
      "metric": "openai_agents_sdk_parity_median_wall_ms",
      "value": 10232.88,
      "unit": "ms",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_structured",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_correct",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_tool_success",
      "value": 1,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_requests",
      "value": 2,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_retries_estimated",
      "value": 0,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_input_tokens",
      "value": 456,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_output_tokens",
      "value": 45,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_total_tokens",
      "value": 501,
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "openai_agents_sdk_parity_median_estimated_cost_usd",
      "value": 0.0005445,
      "unit": "USD",
      "n": 3,
      "note": "median across three repetitions where available"
    },
    {
      "metric": "all_runs",
      "value": 12,
      "unit": "runs"
    },
    {
      "metric": "all_successful_runs",
      "value": 12,
      "unit": "runs"
    },
    {
      "metric": "live_provider_runs",
      "value": 12,
      "unit": "runs",
      "note": "real OpenAI API calls; no mocks"
    }
  ],
  "log": [
    "Using CPython 3.13.5 interpreter at: /opt/homebrew/opt/python@3.13/bin/python3.13",
    "Creating virtual environment at: $TMPDIR/tmp.A3iIzNdwNI/venv",
    "Activate with: source $TMPDIR/tmp.A3iIzNdwNI/venv/bin/activate",
    ""
  ]
}
