{
  "generatedBy": "bench.mjs",
  "startedAt": "2026-08-21T20:24:37.236Z",
  "finishedAt": "2026-08-21T20:39:20.947Z",
  "machine": {
    "cpu": "Apple M1",
    "os": "Darwin 26.3",
    "arch": "arm64",
    "node": "v22.18.0",
    "python": "Python 3.13.5"
  },
  "metricCount": 121,
  "metrics": [
    {
      "metric": "cli_version_tested",
      "value": "2.1.238",
      "note": "claude --version at run time"
    },
    {
      "metric": "sdk_version_tested",
      "value": "0.3.239",
      "note": "fresh npm install at run time"
    },
    {
      "metric": "sdk_export_count",
      "value": 31,
      "note": "Object.keys on the module namespace"
    },
    {
      "metric": "sdk_0_2_0_publish_date",
      "value": "2026-01-07T02:43:14.869Z",
      "note": "npm registry time field"
    },
    {
      "metric": "sdk_version_published_on_article_publish_date",
      "value": "0.3.193",
      "note": "latest version npm shows published on 2026-06-25, the article's publish date"
    },
    {
      "metric": "fixture_commit",
      "value": "a020a9912ee94fc43b527f73cd508c92928eda65",
      "note": "deterministic fixture repo, same content every run"
    },
    {
      "metric": "fixture_todo_count",
      "value": 3,
      "note": "ground truth for the multistep task"
    },
    {
      "metric": "cli_default_fix_wall_ms_median",
      "value": 24211,
      "unit": "ms",
      "note": "median of 3 runs; range 23019-28751",
      "n": 3
    },
    {
      "metric": "cli_default_fix_cost_usd_median",
      "value": 0.3243,
      "unit": "usd",
      "note": "median of 3 runs; range 0.3207-0.3306",
      "n": 3
    },
    {
      "metric": "cli_default_fix_turns_median",
      "value": 5,
      "unit": "turns",
      "note": "median of 3 runs; range 5-5",
      "n": 3
    },
    {
      "metric": "cli_default_fix_cache_creation_tokens_median",
      "value": 21972,
      "unit": "tokens",
      "note": "median of 3 runs; range 21833-22158",
      "n": 3
    },
    {
      "metric": "cli_default_fix_cache_read_tokens_median",
      "value": 157874,
      "unit": "tokens",
      "note": "median of 3 runs; range 157868-157888",
      "n": 3
    },
    {
      "metric": "cli_default_fix_output_tokens_median",
      "value": 1022,
      "unit": "tokens",
      "note": "median of 3 runs; range 935-1202",
      "n": 3
    },
    {
      "metric": "cli_default_fix_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "cli_default_fix_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "cli_default_fix_model",
      "value": "claude-opus-5[1m]",
      "note": "model actually used"
    },
    {
      "metric": "cli_default_inspect_wall_ms_median",
      "value": 12979,
      "unit": "ms",
      "note": "median of 3 runs; range 11972-13426",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_cost_usd_median",
      "value": 0.2527,
      "unit": "usd",
      "note": "median of 3 runs; range 0.2527-0.2544",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_turns_median",
      "value": 3,
      "unit": "turns",
      "note": "median of 3 runs; range 3-3",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_cache_creation_tokens_median",
      "value": 20431,
      "unit": "tokens",
      "note": "median of 3 runs; range 20431-20431",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_cache_read_tokens_median",
      "value": 82958,
      "unit": "tokens",
      "note": "median of 3 runs; range 82958-82958",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_output_tokens_median",
      "value": 276,
      "unit": "tokens",
      "note": "median of 3 runs; range 276-344",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "cli_default_inspect_model",
      "value": "claude-opus-5[1m]",
      "note": "model actually used"
    },
    {
      "metric": "cli_default_multistep_wall_ms_median",
      "value": 17896,
      "unit": "ms",
      "note": "median of 3 runs; range 16014-20191",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_cost_usd_median",
      "value": 0.2676,
      "unit": "usd",
      "note": "median of 3 runs; range 0.2667-0.288",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_turns_median",
      "value": 3,
      "unit": "turns",
      "note": "median of 3 runs; range 3-4",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_cache_creation_tokens_median",
      "value": 20886,
      "unit": "tokens",
      "note": "median of 3 runs; range 20839-21044",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_cache_read_tokens_median",
      "value": 83125,
      "unit": "tokens",
      "note": "median of 3 runs; range 83085-120152",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_output_tokens_median",
      "value": 684,
      "unit": "tokens",
      "note": "median of 3 runs; range 669-697",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "cli_default_multistep_model",
      "value": "claude-opus-5[1m]",
      "note": "model actually used"
    },
    {
      "metric": "cli_parity_fix_wall_ms_median",
      "value": 37404,
      "unit": "ms",
      "note": "median of 3 runs; range 37371-69893",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_cost_usd_median",
      "value": 0.3359,
      "unit": "usd",
      "note": "median of 3 runs; range 0.2912-0.4023",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_turns_median",
      "value": 11,
      "unit": "turns",
      "note": "median of 3 runs; range 9-14",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_cache_creation_tokens_median",
      "value": 26023,
      "unit": "tokens",
      "note": "median of 3 runs; range 25991-27138",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_cache_read_tokens_median",
      "value": 526895,
      "unit": "tokens",
      "note": "median of 3 runs; range 376086-687376",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_output_tokens_median",
      "value": 1477,
      "unit": "tokens",
      "note": "median of 3 runs; range 1455-2210",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_pass_count",
      "value": 2,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "cli_parity_fix_model",
      "value": "claude-sonnet-5",
      "note": "model actually used"
    },
    {
      "metric": "cli_parity_inspect_wall_ms_median",
      "value": 6272,
      "unit": "ms",
      "note": "median of 3 runs; range 5955-6745",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_cost_usd_median",
      "value": 0.1607,
      "unit": "usd",
      "note": "median of 3 runs; range 0.1605-0.1608",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_turns_median",
      "value": 2,
      "unit": "turns",
      "note": "median of 3 runs; range 2-2",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_cache_creation_tokens_median",
      "value": 23050,
      "unit": "tokens",
      "note": "median of 3 runs; range 23039-23057",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_cache_read_tokens_median",
      "value": 69487,
      "unit": "tokens",
      "note": "median of 3 runs; range 69487-69487",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_output_tokens_median",
      "value": 103,
      "unit": "tokens",
      "note": "median of 3 runs; range 91-109",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "cli_parity_inspect_model",
      "value": "claude-sonnet-5",
      "note": "model actually used"
    },
    {
      "metric": "cli_parity_multistep_wall_ms_median",
      "value": 12874,
      "unit": "ms",
      "note": "median of 3 runs; range 11788-17465",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_cost_usd_median",
      "value": 0.187,
      "unit": "usd",
      "note": "median of 3 runs; range 0.1834-0.2062",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_turns_median",
      "value": 3,
      "unit": "turns",
      "note": "median of 3 runs; range 3-4",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_cache_creation_tokens_median",
      "value": 23737,
      "unit": "tokens",
      "note": "median of 3 runs; range 23569-23984",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_cache_read_tokens_median",
      "value": 119305,
      "unit": "tokens",
      "note": "median of 3 runs; range 119272-169203",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_output_tokens_median",
      "value": 586,
      "unit": "tokens",
      "note": "median of 3 runs; range 415-765",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "cli_parity_multistep_model",
      "value": "claude-sonnet-5",
      "note": "model actually used"
    },
    {
      "metric": "sdk_default_fix_wall_ms_median",
      "value": 25635,
      "unit": "ms",
      "note": "median of 3 runs; range 25548-28739",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_cost_usd_median",
      "value": 0.2446,
      "unit": "usd",
      "note": "median of 3 runs; range 0.2083-0.3104",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_turns_median",
      "value": 8,
      "unit": "turns",
      "note": "median of 3 runs; range 8-10",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_tool_calls_median",
      "value": 5,
      "unit": "calls",
      "note": "median of 3 runs; range 5-6",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_cache_creation_tokens_median",
      "value": 9438,
      "unit": "tokens",
      "note": "median of 3 runs; range 8862-19374",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_cache_read_tokens_median",
      "value": 190541,
      "unit": "tokens",
      "note": "median of 3 runs; range 179294-225972",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_output_tokens_median",
      "value": 1078,
      "unit": "tokens",
      "note": "median of 3 runs; range 973-1488",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_default_fix_model",
      "value": "claude-opus-5",
      "note": "model actually used"
    },
    {
      "metric": "sdk_default_inspect_wall_ms_median",
      "value": 14251,
      "unit": "ms",
      "note": "median of 3 runs; range 10138-19119",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_cost_usd_median",
      "value": 0.1223,
      "unit": "usd",
      "note": "median of 3 runs; range 0.0972-0.3521",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_turns_median",
      "value": 2,
      "unit": "turns",
      "note": "median of 3 runs; range 2-3",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_tool_calls_median",
      "value": 1,
      "unit": "calls",
      "note": "median of 3 runs; range 1-2",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_cache_creation_tokens_median",
      "value": 7142,
      "unit": "tokens",
      "note": "median of 3 runs; range 6736-33408",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_cache_read_tokens_median",
      "value": 53232,
      "unit": "tokens",
      "note": "median of 3 runs; range 26616-86614",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_output_tokens_median",
      "value": 186,
      "unit": "tokens",
      "note": "median of 3 runs; range 130-303",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_default_inspect_model",
      "value": "claude-opus-5",
      "note": "model actually used"
    },
    {
      "metric": "sdk_default_multistep_wall_ms_median",
      "value": 22648,
      "unit": "ms",
      "note": "median of 3 runs; range 21872-30013",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_cost_usd_median",
      "value": 0.1896,
      "unit": "usd",
      "note": "median of 3 runs; range 0.1504-0.26",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_turns_median",
      "value": 6,
      "unit": "turns",
      "note": "median of 3 runs; range 5-7",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_tool_calls_median",
      "value": 3,
      "unit": "calls",
      "note": "median of 3 runs; range 3-4",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_cache_creation_tokens_median",
      "value": 8119,
      "unit": "tokens",
      "note": "median of 3 runs; range 7455-18343",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_cache_read_tokens_median",
      "value": 120646,
      "unit": "tokens",
      "note": "median of 3 runs; range 110193-154446",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_output_tokens_median",
      "value": 859,
      "unit": "tokens",
      "note": "median of 3 runs; range 619-1245",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_default_multistep_model",
      "value": "claude-opus-5",
      "note": "model actually used"
    },
    {
      "metric": "sdk_parity_fix_wall_ms_median",
      "value": 28793,
      "unit": "ms",
      "note": "median of 3 runs; range 24036-28843",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_cost_usd_median",
      "value": 0.1136,
      "unit": "usd",
      "note": "median of 3 runs; range 0.0919-0.1412",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_turns_median",
      "value": 12,
      "unit": "turns",
      "note": "median of 3 runs; range 12-14",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_tool_calls_median",
      "value": 8,
      "unit": "calls",
      "note": "median of 3 runs; range 8-9",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_cache_creation_tokens_median",
      "value": 9304,
      "unit": "tokens",
      "note": "median of 3 runs; range 9089-19621",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_cache_read_tokens_median",
      "value": 266003,
      "unit": "tokens",
      "note": "median of 3 runs; range 236084-318834",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_output_tokens_median",
      "value": 947,
      "unit": "tokens",
      "note": "median of 3 runs; range 829-1257",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_parity_fix_model",
      "value": "claude-sonnet-5",
      "note": "model actually used"
    },
    {
      "metric": "sdk_parity_inspect_wall_ms_median",
      "value": 9623,
      "unit": "ms",
      "note": "median of 3 runs; range 8598-12401",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_cost_usd_median",
      "value": 0.042,
      "unit": "usd",
      "note": "median of 3 runs; range 0.042-0.1713",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_turns_median",
      "value": 3,
      "unit": "turns",
      "note": "median of 3 runs; range 3-3",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_tool_calls_median",
      "value": 1,
      "unit": "calls",
      "note": "median of 3 runs; range 1-1",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_cache_creation_tokens_median",
      "value": 6793,
      "unit": "tokens",
      "note": "median of 3 runs; range 6786-40812",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_cache_read_tokens_median",
      "value": 68058,
      "unit": "tokens",
      "note": "median of 3 runs; range 34029-68058",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_output_tokens_median",
      "value": 123,
      "unit": "tokens",
      "note": "median of 3 runs; range 123-126",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_parity_inspect_model",
      "value": "claude-sonnet-5",
      "note": "model actually used"
    },
    {
      "metric": "sdk_parity_multistep_wall_ms_median",
      "value": 23446,
      "unit": "ms",
      "note": "median of 3 runs; range 22081-26368",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_cost_usd_median",
      "value": 0.0781,
      "unit": "usd",
      "note": "median of 3 runs; range 0.0755-0.1168",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_turns_median",
      "value": 8,
      "unit": "turns",
      "note": "median of 3 runs; range 8-9",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_tool_calls_median",
      "value": 4,
      "unit": "calls",
      "note": "median of 3 runs; range 4-4",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_cache_creation_tokens_median",
      "value": 7812,
      "unit": "tokens",
      "note": "median of 3 runs; range 7682-18428",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_cache_read_tokens_median",
      "value": 191589,
      "unit": "tokens",
      "note": "median of 3 runs; range 180983-191646",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_output_tokens_median",
      "value": 688,
      "unit": "tokens",
      "note": "median of 3 runs; range 640-849",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_pass_count",
      "value": 3,
      "unit": "runs",
      "note": "objective grading, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_error_count",
      "value": 0,
      "unit": "runs",
      "note": "surface reported is_error, out of 3",
      "n": 3
    },
    {
      "metric": "sdk_parity_multistep_model",
      "value": "claude-sonnet-5",
      "note": "model actually used"
    }
  ],
  "publicationNote": "Original saved measurements. Runtime logs omitted and local filesystem paths redacted. Not a new benchmark run.",
  "sourceSha256": "df91ee3af2a628603585b29fcd817827f350c5104a884c26b6c3803768171935"
}
