{
  "study": "Machine Pidgin Benchmark 002: mathematical notation versus vernacular",
  "timestamp_utc": "20260802T221708Z",
  "design": {
    "split": "held_out",
    "repetitions": 2,
    "models": [
      "openai/gpt-4o-mini",
      "openai/gpt-5.6-luna",
      "openai/gpt-5.6-terra",
      "openai/gpt-5.6-sol"
    ],
    "conditions": [
      "vernacular",
      "formal"
    ],
    "system_instruction_identical": "Solve the user's task from the information provided. Return one valid JSON object only, with exactly the requested keys and no markdown or commentary.",
    "reasoning_effort": "low for GPT-5.6 family; provider default for GPT-4o mini",
    "temperature": "provider default; fixed seed requested",
    "scoring": "Exact equality with preregistered JSON is on-task; recursive expected-leaf match is constraint score.",
    "tasks_sha256": "8f7dd4fbe931ec68c3d9a1cf0a2b6985907114bd0d0ee48455e41be08866c033",
    "max_study_spend_usd": 2.0
  },
  "summary": {
    "overall": {
      "vernacular": {
        "n": 160,
        "errors": 0,
        "on_task_count": 134,
        "on_task_rate": 0.8375,
        "mean_constraint_score": 0.906875,
        "valid_json_rate": 1.0,
        "reported_cost_usd": 0.0795901,
        "mean_latency_seconds": 10.0069
      },
      "formal": {
        "n": 160,
        "errors": 0,
        "on_task_count": 138,
        "on_task_rate": 0.8625,
        "mean_constraint_score": 0.91625,
        "valid_json_rate": 1.0,
        "reported_cost_usd": 0.0823999,
        "mean_latency_seconds": 10.37595625
      },
      "absolute_lift": 0.025000000000000022
    },
    "by_model": {
      "openai/gpt-4o-mini": {
        "vernacular": {
          "n": 40,
          "errors": 0,
          "on_task_count": 24,
          "on_task_rate": 0.6,
          "mean_constraint_score": 0.765,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.0011559,
          "mean_latency_seconds": 1.34315
        },
        "formal": {
          "n": 40,
          "errors": 0,
          "on_task_count": 21,
          "on_task_rate": 0.525,
          "mean_constraint_score": 0.7150000000000001,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.0010502999999999999,
          "mean_latency_seconds": 1.3211249999999999
        },
        "absolute_lift": -0.07499999999999996
      },
      "openai/gpt-5.6-luna": {
        "vernacular": {
          "n": 40,
          "errors": 0,
          "on_task_count": 36,
          "on_task_rate": 0.9,
          "mean_constraint_score": 0.925,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.0017602,
          "mean_latency_seconds": 1.8166
        },
        "formal": {
          "n": 40,
          "errors": 0,
          "on_task_count": 38,
          "on_task_rate": 0.95,
          "mean_constraint_score": 0.9666666666666666,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.0019336,
          "mean_latency_seconds": 1.7899999999999998
        },
        "absolute_lift": 0.04999999999999993
      },
      "openai/gpt-5.6-terra": {
        "vernacular": {
          "n": 40,
          "errors": 0,
          "on_task_count": 36,
          "on_task_rate": 0.9,
          "mean_constraint_score": 0.9625,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.014914,
          "mean_latency_seconds": 1.5644
        },
        "formal": {
          "n": 40,
          "errors": 0,
          "on_task_count": 40,
          "on_task_rate": 1.0,
          "mean_constraint_score": 1.0,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.016576,
          "mean_latency_seconds": 1.7469249999999998
        },
        "absolute_lift": 0.09999999999999998
      },
      "openai/gpt-5.6-sol": {
        "vernacular": {
          "n": 40,
          "errors": 0,
          "on_task_count": 38,
          "on_task_rate": 0.95,
          "mean_constraint_score": 0.975,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.06176,
          "mean_latency_seconds": 35.30345
        },
        "formal": {
          "n": 40,
          "errors": 0,
          "on_task_count": 39,
          "on_task_rate": 0.975,
          "mean_constraint_score": 0.9833333333333334,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.06284,
          "mean_latency_seconds": 36.645775
        },
        "absolute_lift": 0.025000000000000022
      }
    },
    "by_expected_effect": {
      "notation_friendly": {
        "vernacular": {
          "n": 144,
          "errors": 0,
          "on_task_count": 118,
          "on_task_rate": 0.8194444444444444,
          "mean_constraint_score": 0.8965277777777777,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.0745567,
          "mean_latency_seconds": 9.956486111111111
        },
        "formal": {
          "n": 144,
          "errors": 0,
          "on_task_count": 122,
          "on_task_rate": 0.8472222222222222,
          "mean_constraint_score": 0.9069444444444444,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.0776139,
          "mean_latency_seconds": 10.363375
        },
        "absolute_lift": 0.02777777777777779
      },
      "negative_controls": {
        "vernacular": {
          "n": 16,
          "errors": 0,
          "on_task_count": 16,
          "on_task_rate": 1.0,
          "mean_constraint_score": 1.0,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.0050334,
          "mean_latency_seconds": 10.460625
        },
        "formal": {
          "n": 16,
          "errors": 0,
          "on_task_count": 16,
          "on_task_rate": 1.0,
          "mean_constraint_score": 1.0,
          "valid_json_rate": 1.0,
          "reported_cost_usd": 0.004786,
          "mean_latency_seconds": 10.4891875
        },
        "absolute_lift": 0.0
      }
    },
    "paired": {
      "formal_repairs": 12,
      "formal_regressions": 8,
      "concordant_success": 126,
      "concordant_failure": 14,
      "mcnemar_exact_two_sided_descriptive_p": 0.5034446716308594
    }
  },
  "rows": [
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 140,
        "completion_tokens": 10,
        "total_tokens": 150,
        "cost": 2.7e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.7e-05,
          "upstream_inference_prompt_cost": 2.1e-05,
          "upstream_inference_completions_cost": 6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.423,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 140,
        "completion_tokens": 11,
        "total_tokens": 151,
        "cost": 2.76e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.76e-05,
          "upstream_inference_prompt_cost": 2.1e-05,
          "upstream_inference_completions_cost": 6.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.504,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\n  \"node\": \"Cedar\",\n  \"cost\": 8\n}",
      "usage": {
        "prompt_tokens": 122,
        "completion_tokens": 18,
        "total_tokens": 140,
        "cost": 2.91e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.91e-05,
          "upstream_inference_prompt_cost": 1.83e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.392,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\n  \"node\": \"Cedar\",\n  \"cost\": 8\n}",
      "usage": {
        "prompt_tokens": 122,
        "completion_tokens": 17,
        "total_tokens": 139,
        "cost": 2.85e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.85e-05,
          "upstream_inference_prompt_cost": 1.83e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.062,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.3333333333333333,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 6,
        "theory": 5,
        "communication": 4
      },
      "raw_response": "{\"replication\":6,\"theory\":5,\"communication\":4}",
      "usage": {
        "prompt_tokens": 108,
        "completion_tokens": 16,
        "total_tokens": 124,
        "cost": 2.58e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.58e-05,
          "upstream_inference_prompt_cost": 1.62e-05,
          "upstream_inference_completions_cost": 9.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.378,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.3333333333333333,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 6,
        "theory": 5,
        "communication": 4
      },
      "raw_response": "{\"replication\":6,\"theory\":5,\"communication\":4}",
      "usage": {
        "prompt_tokens": 108,
        "completion_tokens": 15,
        "total_tokens": 123,
        "cost": 2.52e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.52e-05,
          "upstream_inference_prompt_cost": 1.62e-05,
          "upstream_inference_completions_cost": 9e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.293,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.3333333333333333,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 6,
        "theory": 5,
        "communication": 4
      },
      "raw_response": "{\n  \"replication\": 6,\n  \"theory\": 5,\n  \"communication\": 4\n}",
      "usage": {
        "prompt_tokens": 96,
        "completion_tokens": 25,
        "total_tokens": 121,
        "cost": 2.94e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.94e-05,
          "upstream_inference_prompt_cost": 1.44e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.417,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.3333333333333333,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 6,
        "theory": 5,
        "communication": 4
      },
      "raw_response": "{\n  \"replication\": 6,\n  \"theory\": 5,\n  \"communication\": 4\n}",
      "usage": {
        "prompt_tokens": 96,
        "completion_tokens": 25,
        "total_tokens": 121,
        "cost": 2.94e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.94e-05,
          "upstream_inference_prompt_cost": 1.44e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.579,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 17,
        "total_tokens": 109,
        "cost": 2.4e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.4e-05,
          "upstream_inference_prompt_cost": 1.38e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.083,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 16,
        "total_tokens": 108,
        "cost": 2.34e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.34e-05,
          "upstream_inference_prompt_cost": 1.38e-05,
          "upstream_inference_completions_cost": 9.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.973,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\n  \"order\": [\"A\", \"B\", \"C\", \"D\", \"E\", \"F\"]\n}",
      "usage": {
        "prompt_tokens": 98,
        "completion_tokens": 25,
        "total_tokens": 123,
        "cost": 2.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.97e-05,
          "upstream_inference_prompt_cost": 1.47e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.21,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\n  \"order\": [\"A\", \"B\", \"C\", \"D\", \"E\", \"F\"]\n}",
      "usage": {
        "prompt_tokens": 98,
        "completion_tokens": 25,
        "total_tokens": 123,
        "cost": 2.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.97e-05,
          "upstream_inference_prompt_cost": 1.47e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.058,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.6666666666666666,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "anonymization failed",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"anonymization failed\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 25,
        "total_tokens": 149,
        "cost": 3.36e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.36e-05,
          "upstream_inference_prompt_cost": 1.86e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.508,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.6666666666666666,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "anonymization failed",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"anonymization failed\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 25,
        "total_tokens": 149,
        "cost": 3.36e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.36e-05,
          "upstream_inference_prompt_cost": 1.86e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.583,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\n  \"release\": false,\n  \"blocking\": [\n    \"consent unverified\",\n    \"risk not below 4\"\n  ]\n}",
      "usage": {
        "prompt_tokens": 105,
        "completion_tokens": 30,
        "total_tokens": 135,
        "cost": 3.375e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.375e-05,
          "upstream_inference_prompt_cost": 1.575e-05,
          "upstream_inference_completions_cost": 1.8e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.196,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\n  \"release\": false,\n  \"blocking\": [\n    \"consent unverified\",\n    \"risk not below 4\"\n  ]\n}",
      "usage": {
        "prompt_tokens": 105,
        "completion_tokens": 31,
        "total_tokens": 136,
        "cost": 3.435e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.435e-05,
          "upstream_inference_prompt_cost": 1.575e-05,
          "upstream_inference_completions_cost": 1.86e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.803,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 153,
        "completion_tokens": 24,
        "total_tokens": 177,
        "cost": 3.735e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.735e-05,
          "upstream_inference_prompt_cost": 2.295e-05,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.009,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+2",
        "population_source": "audited report",
        "timezone_source": "memo"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+2\",\"population_source\":\"audited report\",\"timezone_source\":\"memo\"}",
      "usage": {
        "prompt_tokens": 153,
        "completion_tokens": 24,
        "total_tokens": 177,
        "cost": 3.735e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.735e-05,
          "upstream_inference_prompt_cost": 2.295e-05,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.029,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\n  \"population\": 1200,\n  \"timezone\": \"UTC+1\",\n  \"population_source\": \"audited report\",\n  \"timezone_source\": \"registry\"\n}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 37,
        "total_tokens": 152,
        "cost": 3.945e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.945e-05,
          "upstream_inference_prompt_cost": 1.725e-05,
          "upstream_inference_completions_cost": 2.22e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.374,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\n  \"population\": 1200,\n  \"timezone\": \"UTC+1\",\n  \"population_source\": \"audited report\",\n  \"timezone_source\": \"registry\"\n}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 37,
        "total_tokens": 152,
        "cost": 3.945e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.945e-05,
          "upstream_inference_prompt_cost": 1.725e-05,
          "upstream_inference_completions_cost": 2.22e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.398,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 137,
        "completion_tokens": 17,
        "total_tokens": 154,
        "cost": 3.075e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.075e-05,
          "upstream_inference_prompt_cost": 2.055e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.084,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 137,
        "completion_tokens": 18,
        "total_tokens": 155,
        "cost": 3.135e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.135e-05,
          "upstream_inference_prompt_cost": 2.055e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.727,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\n  \"allow\": [\"B\", \"D\"],\n  \"deny\": [\"A\", \"C\", \"E\"]\n}",
      "usage": {
        "prompt_tokens": 120,
        "completion_tokens": 26,
        "total_tokens": 146,
        "cost": 3.36e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.36e-05,
          "upstream_inference_prompt_cost": 1.8e-05,
          "upstream_inference_completions_cost": 1.56e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.326,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\n  \"allow\": [\"B\", \"D\"],\n  \"deny\": [\"A\", \"C\", \"E\"]\n}",
      "usage": {
        "prompt_tokens": 120,
        "completion_tokens": 25,
        "total_tokens": 145,
        "cost": 3.3e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.3e-05,
          "upstream_inference_prompt_cost": 1.8e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.198,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 48000.0
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":48000.0}",
      "usage": {
        "prompt_tokens": 110,
        "completion_tokens": 15,
        "total_tokens": 125,
        "cost": 2.55e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.55e-05,
          "upstream_inference_prompt_cost": 1.65e-05,
          "upstream_inference_completions_cost": 9e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.372,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 48000.0
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":48000.0}",
      "usage": {
        "prompt_tokens": 110,
        "completion_tokens": 15,
        "total_tokens": 125,
        "cost": 2.55e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.55e-05,
          "upstream_inference_prompt_cost": 1.65e-05,
          "upstream_inference_completions_cost": 9e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.438,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\n  \"duration_seconds\": 200,\n  \"volume_ml\": 8000\n}",
      "usage": {
        "prompt_tokens": 80,
        "completion_tokens": 20,
        "total_tokens": 100,
        "cost": 2.4e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.4e-05,
          "upstream_inference_prompt_cost": 1.2e-05,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.524,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\n  \"duration_seconds\": 200,\n  \"volume_ml\": 8000\n}",
      "usage": {
        "prompt_tokens": 80,
        "completion_tokens": 20,
        "total_tokens": 100,
        "cost": 2.4e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.4e-05,
          "upstream_inference_prompt_cost": 1.2e-05,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.416,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 120,
        "completion_tokens": 17,
        "total_tokens": 137,
        "cost": 2.82e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.82e-05,
          "upstream_inference_prompt_cost": 1.8e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.343,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 120,
        "completion_tokens": 18,
        "total_tokens": 138,
        "cost": 2.88e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.88e-05,
          "upstream_inference_prompt_cost": 1.8e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.308,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\n  \"granted\": [\"Kai\"],\n  \"denied\": [\"Luz\", \"Mina\"]\n}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 23,
        "total_tokens": 126,
        "cost": 2.925e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.925e-05,
          "upstream_inference_prompt_cost": 1.545e-05,
          "upstream_inference_completions_cost": 1.38e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.374,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\n  \"granted\": [\"Kai\"],\n  \"denied\": [\"Luz\", \"Mina\"]\n}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 24,
        "total_tokens": 127,
        "cost": 2.985e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.985e-05,
          "upstream_inference_prompt_cost": 1.545e-05,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.449,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 116,
        "completion_tokens": 13,
        "total_tokens": 129,
        "cost": 2.52e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.52e-05,
          "upstream_inference_prompt_cost": 1.74e-05,
          "upstream_inference_completions_cost": 7.8e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.244,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 116,
        "completion_tokens": 13,
        "total_tokens": 129,
        "cost": 2.52e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.52e-05,
          "upstream_inference_prompt_cost": 1.74e-05,
          "upstream_inference_completions_cost": 7.8e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.191,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 98,
        "completion_tokens": 13,
        "total_tokens": 111,
        "cost": 2.25e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.25e-05,
          "upstream_inference_prompt_cost": 1.47e-05,
          "upstream_inference_completions_cost": 7.8e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.084,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\n  \"start\": \"10:00\",\n  \"end\": \"10:45\"\n}",
      "usage": {
        "prompt_tokens": 98,
        "completion_tokens": 21,
        "total_tokens": 119,
        "cost": 2.73e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.73e-05,
          "upstream_inference_prompt_cost": 1.47e-05,
          "upstream_inference_completions_cost": 1.26e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.586,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 100,
        "completion_tokens": 18,
        "total_tokens": 118,
        "cost": 2.58e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.58e-05,
          "upstream_inference_prompt_cost": 1.5e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.696,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 100,
        "completion_tokens": 17,
        "total_tokens": 117,
        "cost": 2.52e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.52e-05,
          "upstream_inference_prompt_cost": 1.5e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.824,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 94,
        "completion_tokens": 17,
        "total_tokens": 111,
        "cost": 2.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.43e-05,
          "upstream_inference_prompt_cost": 1.41e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.138,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 94,
        "completion_tokens": 17,
        "total_tokens": 111,
        "cost": 2.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.43e-05,
          "upstream_inference_prompt_cost": 1.41e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.984,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "Q",
        "score": 27
      },
      "raw_response": "{\"project\":\"Q\",\"score\":27}",
      "usage": {
        "prompt_tokens": 130,
        "completion_tokens": 9,
        "total_tokens": 139,
        "cost": 2.49e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.49e-05,
          "upstream_inference_prompt_cost": 1.95e-05,
          "upstream_inference_completions_cost": 5.4e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.508,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 25
      },
      "raw_response": "{\"project\":\"R\",\"score\":25}",
      "usage": {
        "prompt_tokens": 130,
        "completion_tokens": 10,
        "total_tokens": 140,
        "cost": 2.55e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.55e-05,
          "upstream_inference_prompt_cost": 1.95e-05,
          "upstream_inference_completions_cost": 6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.484,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 31
      },
      "raw_response": "{\n  \"project\": \"R\",\n  \"score\": 31\n}",
      "usage": {
        "prompt_tokens": 125,
        "completion_tokens": 16,
        "total_tokens": 141,
        "cost": 2.835e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.835e-05,
          "upstream_inference_prompt_cost": 1.875e-05,
          "upstream_inference_completions_cost": 9.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.039,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 31
      },
      "raw_response": "{\n  \"project\": \"R\",\n  \"score\": 31\n}",
      "usage": {
        "prompt_tokens": 125,
        "completion_tokens": 16,
        "total_tokens": 141,
        "cost": 2.835e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.835e-05,
          "upstream_inference_prompt_cost": 1.875e-05,
          "upstream_inference_completions_cost": 9.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.997,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 98,
        "completion_tokens": 17,
        "total_tokens": 115,
        "cost": 2.49e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.49e-05,
          "upstream_inference_prompt_cost": 1.47e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.059,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 98,
        "completion_tokens": 18,
        "total_tokens": 116,
        "cost": 2.55e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.55e-05,
          "upstream_inference_prompt_cost": 1.47e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.572,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\n  \"overlap_start\": 20,\n  \"overlap_end\": 35,\n  \"duration\": 15\n}",
      "usage": {
        "prompt_tokens": 94,
        "completion_tokens": 27,
        "total_tokens": 121,
        "cost": 3.03e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.03e-05,
          "upstream_inference_prompt_cost": 1.41e-05,
          "upstream_inference_completions_cost": 1.62e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.138,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\n  \"overlap_start\": 20,\n  \"overlap_end\": 35,\n  \"duration\": 15\n}",
      "usage": {
        "prompt_tokens": 94,
        "completion_tokens": 27,
        "total_tokens": 121,
        "cost": 3.03e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.03e-05,
          "upstream_inference_prompt_cost": 1.41e-05,
          "upstream_inference_completions_cost": 1.62e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.169,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.8,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Dee"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Dee\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 133,
        "completion_tokens": 22,
        "total_tokens": 155,
        "cost": 3.315e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.315e-05,
          "upstream_inference_prompt_cost": 1.995e-05,
          "upstream_inference_completions_cost": 1.32e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.269,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.8,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Dee"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Dee\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 133,
        "completion_tokens": 21,
        "total_tokens": 154,
        "cost": 3.255e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.255e-05,
          "upstream_inference_prompt_cost": 1.995e-05,
          "upstream_inference_completions_cost": 1.26e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.734,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.8,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Dee"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\n  \"team\": [\"Ada\", \"Dee\"],\n  \"covered_skills\": [\"analysis\", \"coding\", \"outreach\"]\n}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 29,
        "total_tokens": 143,
        "cost": 3.45e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.45e-05,
          "upstream_inference_prompt_cost": 1.71e-05,
          "upstream_inference_completions_cost": 1.74e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.401,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.8,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Dee"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\n  \"team\": [\"Ada\", \"Dee\"],\n  \"covered_skills\": [\"analysis\", \"coding\", \"outreach\"]\n}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 29,
        "total_tokens": 143,
        "cost": 3.45e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.45e-05,
          "upstream_inference_prompt_cost": 1.71e-05,
          "upstream_inference_completions_cost": 1.74e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.379,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 14,
        "total_tokens": 118,
        "cost": 2.4e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.4e-05,
          "upstream_inference_prompt_cost": 1.56e-05,
          "upstream_inference_completions_cost": 8.4e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.482,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 13,
        "total_tokens": 117,
        "cost": 2.34e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.34e-05,
          "upstream_inference_prompt_cost": 1.56e-05,
          "upstream_inference_completions_cost": 7.8e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.334,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open_vote",
        "may_publish": false
      },
      "raw_response": "{\n  \"next_action\": \"open_vote\",\n  \"may_publish\": false\n}",
      "usage": {
        "prompt_tokens": 96,
        "completion_tokens": 18,
        "total_tokens": 114,
        "cost": 2.52e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.52e-05,
          "upstream_inference_prompt_cost": 1.44e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.987,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open_vote",
        "may_publish": false
      },
      "raw_response": "{\n  \"next_action\": \"open_vote\",\n  \"may_publish\": false\n}",
      "usage": {
        "prompt_tokens": 96,
        "completion_tokens": 18,
        "total_tokens": 114,
        "cost": 2.52e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.52e-05,
          "upstream_inference_prompt_cost": 1.44e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.983,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 4,
        "y": 7,
        "difference": 3
      },
      "raw_response": "{\"x\":4,\"y\":7,\"difference\":3}",
      "usage": {
        "prompt_tokens": 90,
        "completion_tokens": 14,
        "total_tokens": 104,
        "cost": 2.19e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.19e-05,
          "upstream_inference_prompt_cost": 1.35e-05,
          "upstream_inference_completions_cost": 8.4e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.398,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 4,
        "y": 7,
        "difference": 3
      },
      "raw_response": "{\"x\":4,\"y\":7,\"difference\":3}",
      "usage": {
        "prompt_tokens": 90,
        "completion_tokens": 14,
        "total_tokens": 104,
        "cost": 2.19e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.19e-05,
          "upstream_inference_prompt_cost": 1.35e-05,
          "upstream_inference_completions_cost": 8.4e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.343,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 4,
        "y": 7,
        "difference": 3
      },
      "raw_response": "{\n  \"x\": 4,\n  \"y\": 7,\n  \"difference\": 3\n}",
      "usage": {
        "prompt_tokens": 96,
        "completion_tokens": 24,
        "total_tokens": 120,
        "cost": 2.88e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.88e-05,
          "upstream_inference_prompt_cost": 1.44e-05,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.661,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 4,
        "y": 7,
        "difference": 3
      },
      "raw_response": "{\n  \"x\": 4,\n  \"y\": 7,\n  \"difference\": 3\n}",
      "usage": {
        "prompt_tokens": 96,
        "completion_tokens": 23,
        "total_tokens": 119,
        "cost": 2.82e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.82e-05,
          "upstream_inference_prompt_cost": 1.44e-05,
          "upstream_inference_completions_cost": 1.38e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.748,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "C",
        "net_value": 1
      },
      "raw_response": "{\"test\":\"C\",\"net_value\":1}",
      "usage": {
        "prompt_tokens": 112,
        "completion_tokens": 11,
        "total_tokens": 123,
        "cost": 2.34e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.34e-05,
          "upstream_inference_prompt_cost": 1.68e-05,
          "upstream_inference_completions_cost": 6.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.517,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "C",
        "net_value": 1
      },
      "raw_response": "{\"test\":\"C\",\"net_value\":1}",
      "usage": {
        "prompt_tokens": 112,
        "completion_tokens": 10,
        "total_tokens": 122,
        "cost": 2.28e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.28e-05,
          "upstream_inference_prompt_cost": 1.68e-05,
          "upstream_inference_completions_cost": 6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.031,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "Test A",
        "net_value": 5
      },
      "raw_response": "{\n  \"test\": \"Test A\",\n  \"net_value\": 5\n}",
      "usage": {
        "prompt_tokens": 112,
        "completion_tokens": 19,
        "total_tokens": 131,
        "cost": 2.82e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.82e-05,
          "upstream_inference_prompt_cost": 1.68e-05,
          "upstream_inference_completions_cost": 1.14e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.532,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "Test A",
        "net_value": 5
      },
      "raw_response": "{\n  \"test\": \"Test A\",\n  \"net_value\": 5\n}",
      "usage": {
        "prompt_tokens": 112,
        "completion_tokens": 18,
        "total_tokens": 130,
        "cost": 2.76e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.76e-05,
          "upstream_inference_prompt_cost": 1.68e-05,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.083,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 67,
        "completion_tokens": 11,
        "total_tokens": 78,
        "cost": 1.665e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.665e-05,
          "upstream_inference_prompt_cost": 1.005e-05,
          "upstream_inference_completions_cost": 6.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.939,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 67,
        "completion_tokens": 11,
        "total_tokens": 78,
        "cost": 1.665e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.665e-05,
          "upstream_inference_prompt_cost": 1.005e-05,
          "upstream_inference_completions_cost": 6.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.973,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\n  \"codename\": \"Lark\",\n  \"revision\": 6\n}",
      "usage": {
        "prompt_tokens": 62,
        "completion_tokens": 19,
        "total_tokens": 81,
        "cost": 2.07e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.07e-05,
          "upstream_inference_prompt_cost": 9.3e-06,
          "upstream_inference_completions_cost": 1.14e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.281,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\n  \"codename\": \"Lark\",\n  \"revision\": 6\n}",
      "usage": {
        "prompt_tokens": 62,
        "completion_tokens": 18,
        "total_tokens": 80,
        "cost": 2.01e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.01e-05,
          "upstream_inference_prompt_cost": 9.3e-06,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.181,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 13,
        "total_tokens": 106,
        "cost": 2.175e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.175e-05,
          "upstream_inference_prompt_cost": 1.395e-05,
          "upstream_inference_completions_cost": 7.8e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.111,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 13,
        "total_tokens": 106,
        "cost": 2.175e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.175e-05,
          "upstream_inference_prompt_cost": 1.395e-05,
          "upstream_inference_completions_cost": 7.8e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.055,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\n  \"urgent\": [\"A\", \"C\"],\n  \"routine\": [\"B\"]\n}",
      "usage": {
        "prompt_tokens": 78,
        "completion_tokens": 20,
        "total_tokens": 98,
        "cost": 2.37e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.37e-05,
          "upstream_inference_prompt_cost": 1.17e-05,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.48,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\n  \"urgent\": [\"A\", \"C\"],\n  \"routine\": [\"B\"]\n}",
      "usage": {
        "prompt_tokens": 78,
        "completion_tokens": 20,
        "total_tokens": 98,
        "cost": 2.37e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.37e-05,
          "upstream_inference_prompt_cost": 1.17e-05,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.44,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36.0,
        "after_addition": 43.0,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36.0,\"after_addition\":43.0,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 25,
        "total_tokens": 118,
        "cost": 2.895e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.895e-05,
          "upstream_inference_prompt_cost": 1.395e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.167,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36.0,
        "after_addition": 43.0,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36.0,\"after_addition\":43.0,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 25,
        "total_tokens": 118,
        "cost": 2.895e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.895e-05,
          "upstream_inference_prompt_cost": 1.395e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.919,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.6666666666666666,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8
      },
      "raw_response": "{\n  \"after_removal\": 36,\n  \"after_addition\": 43,\n  \"per_bin\": 8\n}",
      "usage": {
        "prompt_tokens": 82,
        "completion_tokens": 28,
        "total_tokens": 110,
        "cost": 2.91e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.91e-05,
          "upstream_inference_prompt_cost": 1.23e-05,
          "upstream_inference_completions_cost": 1.68e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.108,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.6666666666666666,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8
      },
      "raw_response": "{\n  \"after_removal\": 36,\n  \"after_addition\": 43,\n  \"per_bin\": 8\n}",
      "usage": {
        "prompt_tokens": 82,
        "completion_tokens": 28,
        "total_tokens": 110,
        "cost": 2.91e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.91e-05,
          "upstream_inference_prompt_cost": 1.23e-05,
          "upstream_inference_completions_cost": 1.68e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.154,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          1,
          2,
          3,
          0
        ]
      },
      "raw_response": "{\"order\":[1,2,3,0]}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 11,
        "total_tokens": 118,
        "cost": 2.265e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.265e-05,
          "upstream_inference_prompt_cost": 1.605e-05,
          "upstream_inference_completions_cost": 6.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.008,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          1,
          2,
          3,
          0
        ]
      },
      "raw_response": "{\"order\":[1,2,3,0]}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 11,
        "total_tokens": 118,
        "cost": 2.265e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.265e-05,
          "upstream_inference_prompt_cost": 1.605e-05,
          "upstream_inference_completions_cost": 6.6e-06
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 0.932,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "reports": [
          "C",
          "A",
          "B",
          "D"
        ]
      },
      "raw_response": "{\n  \"reports\": [\n    \"C\",\n    \"A\",\n    \"B\",\n    \"D\"\n  ]\n}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 26,
        "total_tokens": 141,
        "cost": 3.285e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.285e-05,
          "upstream_inference_prompt_cost": 1.725e-05,
          "upstream_inference_completions_cost": 1.56e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "Azure",
      "latency_seconds": 1.775,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-4o-mini",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "reports": [
          "C",
          "A",
          "B",
          "D"
        ]
      },
      "raw_response": "{\n  \"reports\": [\n    \"C\",\n    \"A\",\n    \"B\",\n    \"D\"\n  ]\n}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 25,
        "total_tokens": 140,
        "cost": 3.225e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.225e-05,
          "upstream_inference_prompt_cost": 1.725e-05,
          "upstream_inference_completions_cost": 1.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.622,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 139,
        "completion_tokens": 54,
        "total_tokens": 193,
        "cost": 4.63e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.63e-05,
          "upstream_inference_prompt_cost": 1.39e-05,
          "upstream_inference_completions_cost": 3.24e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 35,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.38,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 139,
        "completion_tokens": 17,
        "total_tokens": 156,
        "cost": 2.41e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.41e-05,
          "upstream_inference_prompt_cost": 1.39e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.232,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 121,
        "completion_tokens": 17,
        "total_tokens": 138,
        "cost": 2.23e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.23e-05,
          "upstream_inference_prompt_cost": 1.21e-05,
          "upstream_inference_completions_cost": 1.02e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.077,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 121,
        "completion_tokens": 50,
        "total_tokens": 171,
        "cost": 4.21e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.21e-05,
          "upstream_inference_prompt_cost": 1.21e-05,
          "upstream_inference_completions_cost": 3e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 31,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.391,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 88,
        "total_tokens": 195,
        "cost": 6.35e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 6.35e-05,
          "upstream_inference_prompt_cost": 1.07e-05,
          "upstream_inference_completions_cost": 5.28e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 64,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.032,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 83,
        "total_tokens": 190,
        "cost": 6.05e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 6.05e-05,
          "upstream_inference_prompt_cost": 1.07e-05,
          "upstream_inference_completions_cost": 4.98e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 59,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.958,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 74,
        "total_tokens": 169,
        "cost": 5.39e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.39e-05,
          "upstream_inference_prompt_cost": 9.5e-06,
          "upstream_inference_completions_cost": 4.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 50,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.085,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 81,
        "total_tokens": 176,
        "cost": 5.81e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.81e-05,
          "upstream_inference_prompt_cost": 9.5e-06,
          "upstream_inference_completions_cost": 4.86e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 57,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.862,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 91,
        "completion_tokens": 121,
        "total_tokens": 212,
        "cost": 8.17e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 8.17e-05,
          "upstream_inference_prompt_cost": 9.1e-06,
          "upstream_inference_completions_cost": 7.26e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 96,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.308,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 91,
        "completion_tokens": 104,
        "total_tokens": 195,
        "cost": 7.15e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.15e-05,
          "upstream_inference_prompt_cost": 9.1e-06,
          "upstream_inference_completions_cost": 6.24e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 79,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.742,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 75,
        "total_tokens": 172,
        "cost": 5.47e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.47e-05,
          "upstream_inference_prompt_cost": 9.7e-06,
          "upstream_inference_completions_cost": 4.5e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 50,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 3.839,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 100,
        "total_tokens": 197,
        "cost": 6.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 6.97e-05,
          "upstream_inference_prompt_cost": 9.7e-06,
          "upstream_inference_completions_cost": 6e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 75,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.131,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.3333333333333333,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 123,
        "completion_tokens": 49,
        "total_tokens": 172,
        "cost": 4.17e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.17e-05,
          "upstream_inference_prompt_cost": 1.23e-05,
          "upstream_inference_completions_cost": 2.94e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 26,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.678,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.3333333333333333,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 123,
        "completion_tokens": 57,
        "total_tokens": 180,
        "cost": 4.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.65e-05,
          "upstream_inference_prompt_cost": 1.23e-05,
          "upstream_inference_completions_cost": 3.42e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 34,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.436,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 49,
        "total_tokens": 153,
        "cost": 3.98e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.98e-05,
          "upstream_inference_prompt_cost": 1.04e-05,
          "upstream_inference_completions_cost": 2.94e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 21,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.437,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 57,
        "total_tokens": 161,
        "cost": 4.46e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.46e-05,
          "upstream_inference_prompt_cost": 1.04e-05,
          "upstream_inference_completions_cost": 3.42e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 29,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.48,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 152,
        "completion_tokens": 31,
        "total_tokens": 183,
        "cost": 3.38e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.38e-05,
          "upstream_inference_prompt_cost": 1.52e-05,
          "upstream_inference_completions_cost": 1.86e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.432,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 152,
        "completion_tokens": 31,
        "total_tokens": 183,
        "cost": 3.38e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.38e-05,
          "upstream_inference_prompt_cost": 1.52e-05,
          "upstream_inference_completions_cost": 1.86e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.43,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 31,
        "total_tokens": 145,
        "cost": 3e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3e-05,
          "upstream_inference_prompt_cost": 1.14e-05,
          "upstream_inference_completions_cost": 1.86e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.032,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 77,
        "total_tokens": 191,
        "cost": 5.76e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.76e-05,
          "upstream_inference_prompt_cost": 1.14e-05,
          "upstream_inference_completions_cost": 4.62e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 44,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 3.007,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 136,
        "completion_tokens": 66,
        "total_tokens": 202,
        "cost": 5.32e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.32e-05,
          "upstream_inference_prompt_cost": 1.36e-05,
          "upstream_inference_completions_cost": 3.96e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 40,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.674,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 136,
        "completion_tokens": 62,
        "total_tokens": 198,
        "cost": 5.08e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.08e-05,
          "upstream_inference_prompt_cost": 1.36e-05,
          "upstream_inference_completions_cost": 3.72e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 36,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.607,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 54,
        "total_tokens": 173,
        "cost": 4.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.43e-05,
          "upstream_inference_prompt_cost": 1.19e-05,
          "upstream_inference_completions_cost": 3.24e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 28,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.633,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 51,
        "total_tokens": 170,
        "cost": 4.25e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.25e-05,
          "upstream_inference_prompt_cost": 1.19e-05,
          "upstream_inference_completions_cost": 3.06e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 25,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.563,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 109,
        "completion_tokens": 65,
        "total_tokens": 174,
        "cost": 4.99e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.99e-05,
          "upstream_inference_prompt_cost": 1.09e-05,
          "upstream_inference_completions_cost": 3.9e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 44,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.774,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 109,
        "completion_tokens": 66,
        "total_tokens": 175,
        "cost": 5.05e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.05e-05,
          "upstream_inference_prompt_cost": 1.09e-05,
          "upstream_inference_completions_cost": 3.96e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 45,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.391,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 79,
        "completion_tokens": 58,
        "total_tokens": 137,
        "cost": 4.27e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.27e-05,
          "upstream_inference_prompt_cost": 7.9e-06,
          "upstream_inference_completions_cost": 3.48e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 37,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.667,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 79,
        "completion_tokens": 69,
        "total_tokens": 148,
        "cost": 4.93e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.93e-05,
          "upstream_inference_prompt_cost": 7.9e-06,
          "upstream_inference_completions_cost": 4.14e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 48,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.121,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 55,
        "total_tokens": 174,
        "cost": 4.49e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.49e-05,
          "upstream_inference_prompt_cost": 1.19e-05,
          "upstream_inference_completions_cost": 3.3e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 29,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.417,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 48,
        "total_tokens": 167,
        "cost": 4.07e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.07e-05,
          "upstream_inference_prompt_cost": 1.19e-05,
          "upstream_inference_completions_cost": 2.88e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 22,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.391,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 102,
        "completion_tokens": 61,
        "total_tokens": 163,
        "cost": 4.68e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.68e-05,
          "upstream_inference_prompt_cost": 1.02e-05,
          "upstream_inference_completions_cost": 3.66e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 35,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.165,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 102,
        "completion_tokens": 61,
        "total_tokens": 163,
        "cost": 4.68e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.68e-05,
          "upstream_inference_prompt_cost": 1.02e-05,
          "upstream_inference_completions_cost": 3.66e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 35,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.862,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 57,
        "total_tokens": 172,
        "cost": 4.57e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.57e-05,
          "upstream_inference_prompt_cost": 1.15e-05,
          "upstream_inference_completions_cost": 3.42e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 35,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.242,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 66,
        "total_tokens": 181,
        "cost": 5.11e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.11e-05,
          "upstream_inference_prompt_cost": 1.15e-05,
          "upstream_inference_completions_cost": 3.96e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 44,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.8,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 20,
        "total_tokens": 117,
        "cost": 2.17e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.17e-05,
          "upstream_inference_prompt_cost": 9.7e-06,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.482,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 58,
        "total_tokens": 155,
        "cost": 4.45e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.45e-05,
          "upstream_inference_prompt_cost": 9.7e-06,
          "upstream_inference_completions_cost": 3.48e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 36,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.389,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 99,
        "completion_tokens": 24,
        "total_tokens": 123,
        "cost": 2.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.43e-05,
          "upstream_inference_prompt_cost": 9.9e-06,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.493,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 99,
        "completion_tokens": 24,
        "total_tokens": 123,
        "cost": 2.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.43e-05,
          "upstream_inference_prompt_cost": 9.9e-06,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.384,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 2.37e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.37e-05,
          "upstream_inference_prompt_cost": 9.3e-06,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.308,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 2.37e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.37e-05,
          "upstream_inference_prompt_cost": 9.3e-06,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.255,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 129,
        "completion_tokens": 69,
        "total_tokens": 198,
        "cost": 5.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.43e-05,
          "upstream_inference_prompt_cost": 1.29e-05,
          "upstream_inference_completions_cost": 4.14e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 51,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.782,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 129,
        "completion_tokens": 69,
        "total_tokens": 198,
        "cost": 5.43e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.43e-05,
          "upstream_inference_prompt_cost": 1.29e-05,
          "upstream_inference_completions_cost": 4.14e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 51,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.772,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 62,
        "total_tokens": 186,
        "cost": 4.96e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.96e-05,
          "upstream_inference_prompt_cost": 1.24e-05,
          "upstream_inference_completions_cost": 3.72e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 44,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.382,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 68,
        "total_tokens": 192,
        "cost": 5.32e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.32e-05,
          "upstream_inference_prompt_cost": 1.24e-05,
          "upstream_inference_completions_cost": 4.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 50,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.579,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 44,
        "total_tokens": 141,
        "cost": 3.61e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.61e-05,
          "upstream_inference_prompt_cost": 9.7e-06,
          "upstream_inference_completions_cost": 2.64e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 18,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.512,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 48,
        "total_tokens": 145,
        "cost": 3.85e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.85e-05,
          "upstream_inference_prompt_cost": 9.7e-06,
          "upstream_inference_completions_cost": 2.88e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 22,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.01,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 2.37e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.37e-05,
          "upstream_inference_prompt_cost": 9.3e-06,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.238,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 2.37e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.37e-05,
          "upstream_inference_prompt_cost": 9.3e-06,
          "upstream_inference_completions_cost": 1.44e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.423,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 132,
        "completion_tokens": 123,
        "total_tokens": 255,
        "cost": 8.7e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 8.7e-05,
          "upstream_inference_prompt_cost": 1.32e-05,
          "upstream_inference_completions_cost": 7.38e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 94,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.858,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 132,
        "completion_tokens": 149,
        "total_tokens": 281,
        "cost": 0.0001026,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0001026,
          "upstream_inference_prompt_cost": 1.32e-05,
          "upstream_inference_completions_cost": 8.94e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 120,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 3.545,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 113,
        "completion_tokens": 111,
        "total_tokens": 224,
        "cost": 7.79e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.79e-05,
          "upstream_inference_prompt_cost": 1.13e-05,
          "upstream_inference_completions_cost": 6.66e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 82,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.944,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 113,
        "completion_tokens": 142,
        "total_tokens": 255,
        "cost": 9.65e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 9.65e-05,
          "upstream_inference_prompt_cost": 1.13e-05,
          "upstream_inference_completions_cost": 8.52e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 113,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.064,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 50,
        "total_tokens": 153,
        "cost": 4.03e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.03e-05,
          "upstream_inference_prompt_cost": 1.03e-05,
          "upstream_inference_completions_cost": 3e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 28,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.583,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 20,
        "total_tokens": 123,
        "cost": 2.23e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.23e-05,
          "upstream_inference_prompt_cost": 1.03e-05,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.318,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open_vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open_vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 19,
        "total_tokens": 114,
        "cost": 2.09e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.09e-05,
          "upstream_inference_prompt_cost": 9.5e-06,
          "upstream_inference_completions_cost": 1.14e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.696,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 44,
        "total_tokens": 139,
        "cost": 3.59e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.59e-05,
          "upstream_inference_prompt_cost": 9.5e-06,
          "upstream_inference_completions_cost": 2.64e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 23,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.496,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 89,
        "completion_tokens": 87,
        "total_tokens": 176,
        "cost": 6.11e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 6.11e-05,
          "upstream_inference_prompt_cost": 8.9e-06,
          "upstream_inference_completions_cost": 5.22e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 65,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.037,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 89,
        "completion_tokens": 86,
        "total_tokens": 175,
        "cost": 6.05e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 6.05e-05,
          "upstream_inference_prompt_cost": 8.9e-06,
          "upstream_inference_completions_cost": 5.16e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 64,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.724,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 82,
        "total_tokens": 177,
        "cost": 5.87e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.87e-05,
          "upstream_inference_prompt_cost": 9.5e-06,
          "upstream_inference_completions_cost": 4.92e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 60,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.696,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 89,
        "total_tokens": 184,
        "cost": 6.29e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 6.29e-05,
          "upstream_inference_prompt_cost": 9.5e-06,
          "upstream_inference_completions_cost": 5.34e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 67,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.724,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 39,
        "total_tokens": 150,
        "cost": 3.45e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.45e-05,
          "upstream_inference_prompt_cost": 1.11e-05,
          "upstream_inference_completions_cost": 2.34e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 20,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.291,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 39,
        "total_tokens": 150,
        "cost": 3.45e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.45e-05,
          "upstream_inference_prompt_cost": 1.11e-05,
          "upstream_inference_completions_cost": 2.34e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 20,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.259,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 44,
        "total_tokens": 155,
        "cost": 3.75e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.75e-05,
          "upstream_inference_prompt_cost": 1.11e-05,
          "upstream_inference_completions_cost": 2.64e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 25,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.566,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 41,
        "total_tokens": 152,
        "cost": 3.57e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 3.57e-05,
          "upstream_inference_prompt_cost": 1.11e-05,
          "upstream_inference_completions_cost": 2.46e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 22,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.618,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 66,
        "completion_tokens": 18,
        "total_tokens": 84,
        "cost": 1.74e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.74e-05,
          "upstream_inference_prompt_cost": 6.6e-06,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.141,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 66,
        "completion_tokens": 18,
        "total_tokens": 84,
        "cost": 1.74e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.74e-05,
          "upstream_inference_prompt_cost": 6.6e-06,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.716,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 61,
        "completion_tokens": 18,
        "total_tokens": 79,
        "cost": 1.69e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.69e-05,
          "upstream_inference_prompt_cost": 6.1e-06,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.632,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 61,
        "completion_tokens": 18,
        "total_tokens": 79,
        "cost": 1.69e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.69e-05,
          "upstream_inference_prompt_cost": 6.1e-06,
          "upstream_inference_completions_cost": 1.08e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.493,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 20,
        "total_tokens": 112,
        "cost": 2.12e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.12e-05,
          "upstream_inference_prompt_cost": 9.2e-06,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.053,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 20,
        "total_tokens": 112,
        "cost": 2.12e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 2.12e-05,
          "upstream_inference_prompt_cost": 9.2e-06,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.336,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 77,
        "completion_tokens": 20,
        "total_tokens": 97,
        "cost": 1.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.97e-05,
          "upstream_inference_prompt_cost": 7.7e-06,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.251,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 77,
        "completion_tokens": 20,
        "total_tokens": 97,
        "cost": 1.97e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 1.97e-05,
          "upstream_inference_prompt_cost": 7.7e-06,
          "upstream_inference_completions_cost": 1.2e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.311,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 61,
        "total_tokens": 153,
        "cost": 4.58e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.58e-05,
          "upstream_inference_prompt_cost": 9.2e-06,
          "upstream_inference_completions_cost": 3.66e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 32,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.254,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 61,
        "total_tokens": 153,
        "cost": 4.58e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.58e-05,
          "upstream_inference_prompt_cost": 9.2e-06,
          "upstream_inference_completions_cost": 3.66e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 32,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.771,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 81,
        "completion_tokens": 62,
        "total_tokens": 143,
        "cost": 4.53e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 4.53e-05,
          "upstream_inference_prompt_cost": 8.1e-06,
          "upstream_inference_completions_cost": 3.72e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 33,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.575,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 81,
        "completion_tokens": 82,
        "total_tokens": 163,
        "cost": 5.73e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 5.73e-05,
          "upstream_inference_prompt_cost": 8.1e-06,
          "upstream_inference_completions_cost": 4.92e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 53,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.709,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 106,
        "completion_tokens": 149,
        "total_tokens": 255,
        "cost": 0.0001,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0001,
          "upstream_inference_prompt_cost": 1.06e-05,
          "upstream_inference_completions_cost": 8.94e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 128,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.852,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 106,
        "completion_tokens": 149,
        "total_tokens": 255,
        "cost": 0.0001,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0001,
          "upstream_inference_prompt_cost": 1.06e-05,
          "upstream_inference_completions_cost": 8.94e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 128,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.985,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "reports": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"reports\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 112,
        "total_tokens": 226,
        "cost": 7.86e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.86e-05,
          "upstream_inference_prompt_cost": 1.14e-05,
          "upstream_inference_completions_cost": 6.72e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 91,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.924,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-luna",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "reports": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"reports\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 99,
        "total_tokens": 213,
        "cost": 7.08e-05,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 7.08e-05,
          "upstream_inference_prompt_cost": 1.14e-05,
          "upstream_inference_completions_cost": 5.94e-05
        },
        "completion_tokens_details": {
          "reasoning_tokens": 78,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 3.557,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 139,
        "completion_tokens": 17,
        "total_tokens": 156,
        "cost": 0.000241,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000241,
          "upstream_inference_prompt_cost": 0.000139,
          "upstream_inference_completions_cost": 0.000102
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.344,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 139,
        "completion_tokens": 17,
        "total_tokens": 156,
        "cost": 0.000241,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000241,
          "upstream_inference_prompt_cost": 0.000139,
          "upstream_inference_completions_cost": 0.000102
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.145,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 121,
        "completion_tokens": 17,
        "total_tokens": 138,
        "cost": 0.000223,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000223,
          "upstream_inference_prompt_cost": 0.000121,
          "upstream_inference_completions_cost": 0.000102
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.231,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 121,
        "completion_tokens": 17,
        "total_tokens": 138,
        "cost": 0.000223,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000223,
          "upstream_inference_prompt_cost": 0.000121,
          "upstream_inference_completions_cost": 0.000102
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.79,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 66,
        "total_tokens": 173,
        "cost": 0.000503,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000503,
          "upstream_inference_prompt_cost": 0.000107,
          "upstream_inference_completions_cost": 0.000396
        },
        "completion_tokens_details": {
          "reasoning_tokens": 42,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.148,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 80,
        "total_tokens": 187,
        "cost": 0.000587,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000587,
          "upstream_inference_prompt_cost": 0.000107,
          "upstream_inference_completions_cost": 0.00048
        },
        "completion_tokens_details": {
          "reasoning_tokens": 56,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.648,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 81,
        "total_tokens": 176,
        "cost": 0.000581,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000581,
          "upstream_inference_prompt_cost": 9.5e-05,
          "upstream_inference_completions_cost": 0.000486
        },
        "completion_tokens_details": {
          "reasoning_tokens": 57,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.969,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 79,
        "total_tokens": 174,
        "cost": 0.000569,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000569,
          "upstream_inference_prompt_cost": 9.5e-05,
          "upstream_inference_completions_cost": 0.000474
        },
        "completion_tokens_details": {
          "reasoning_tokens": 55,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.606,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 91,
        "completion_tokens": 134,
        "total_tokens": 225,
        "cost": 0.000895,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000895,
          "upstream_inference_prompt_cost": 9.1e-05,
          "upstream_inference_completions_cost": 0.000804
        },
        "completion_tokens_details": {
          "reasoning_tokens": 109,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.682,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 91,
        "completion_tokens": 113,
        "total_tokens": 204,
        "cost": 0.000769,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000769,
          "upstream_inference_prompt_cost": 9.1e-05,
          "upstream_inference_completions_cost": 0.000678
        },
        "completion_tokens_details": {
          "reasoning_tokens": 88,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.245,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 23,
        "total_tokens": 120,
        "cost": 0.000235,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000235,
          "upstream_inference_prompt_cost": 9.7e-05,
          "upstream_inference_completions_cost": 0.000138
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.445,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 82,
        "total_tokens": 179,
        "cost": 0.000589,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000589,
          "upstream_inference_prompt_cost": 9.7e-05,
          "upstream_inference_completions_cost": 0.000492
        },
        "completion_tokens_details": {
          "reasoning_tokens": 57,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.103,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 123,
        "completion_tokens": 57,
        "total_tokens": 180,
        "cost": 0.000465,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000465,
          "upstream_inference_prompt_cost": 0.000123,
          "upstream_inference_completions_cost": 0.000342
        },
        "completion_tokens_details": {
          "reasoning_tokens": 29,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.919,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 123,
        "completion_tokens": 58,
        "total_tokens": 181,
        "cost": 0.000471,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000471,
          "upstream_inference_prompt_cost": 0.000123,
          "upstream_inference_completions_cost": 0.000348
        },
        "completion_tokens_details": {
          "reasoning_tokens": 30,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.615,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 50,
        "total_tokens": 154,
        "cost": 0.000404,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000404,
          "upstream_inference_prompt_cost": 0.000104,
          "upstream_inference_completions_cost": 0.0003
        },
        "completion_tokens_details": {
          "reasoning_tokens": 22,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.611,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 48,
        "total_tokens": 152,
        "cost": 0.000392,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000392,
          "upstream_inference_prompt_cost": 0.000104,
          "upstream_inference_completions_cost": 0.000288
        },
        "completion_tokens_details": {
          "reasoning_tokens": 20,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.642,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 152,
        "completion_tokens": 31,
        "total_tokens": 183,
        "cost": 0.000338,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000338,
          "upstream_inference_prompt_cost": 0.000152,
          "upstream_inference_completions_cost": 0.000186
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.5,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 152,
        "completion_tokens": 31,
        "total_tokens": 183,
        "cost": 0.000338,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000338,
          "upstream_inference_prompt_cost": 0.000152,
          "upstream_inference_completions_cost": 0.000186
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.464,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.75,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "Registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"Registry\"}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 31,
        "total_tokens": 145,
        "cost": 0.0003,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0003,
          "upstream_inference_prompt_cost": 0.000114,
          "upstream_inference_completions_cost": 0.000186
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.372,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.75,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "Registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"Registry\"}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 31,
        "total_tokens": 145,
        "cost": 0.0003,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0003,
          "upstream_inference_prompt_cost": 0.000114,
          "upstream_inference_completions_cost": 0.000186
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.271,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 136,
        "completion_tokens": 56,
        "total_tokens": 192,
        "cost": 0.000472,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000472,
          "upstream_inference_prompt_cost": 0.000136,
          "upstream_inference_completions_cost": 0.000336
        },
        "completion_tokens_details": {
          "reasoning_tokens": 30,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.439,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 136,
        "completion_tokens": 24,
        "total_tokens": 160,
        "cost": 0.00028,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00028,
          "upstream_inference_prompt_cost": 0.000136,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.199,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.000263,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000263,
          "upstream_inference_prompt_cost": 0.000119,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.209,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.000263,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000263,
          "upstream_inference_prompt_cost": 0.000119,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.341,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 109,
        "completion_tokens": 58,
        "total_tokens": 167,
        "cost": 0.000457,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000457,
          "upstream_inference_prompt_cost": 0.000109,
          "upstream_inference_completions_cost": 0.000348
        },
        "completion_tokens_details": {
          "reasoning_tokens": 37,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.659,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 109,
        "completion_tokens": 58,
        "total_tokens": 167,
        "cost": 0.000457,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000457,
          "upstream_inference_prompt_cost": 0.000109,
          "upstream_inference_completions_cost": 0.000348
        },
        "completion_tokens_details": {
          "reasoning_tokens": 37,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.769,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 79,
        "completion_tokens": 59,
        "total_tokens": 138,
        "cost": 0.000433,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000433,
          "upstream_inference_prompt_cost": 7.9e-05,
          "upstream_inference_completions_cost": 0.000354
        },
        "completion_tokens_details": {
          "reasoning_tokens": 38,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.584,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 79,
        "completion_tokens": 51,
        "total_tokens": 130,
        "cost": 0.000385,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000385,
          "upstream_inference_prompt_cost": 7.9e-05,
          "upstream_inference_completions_cost": 0.000306
        },
        "completion_tokens_details": {
          "reasoning_tokens": 30,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.489,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.000263,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000263,
          "upstream_inference_prompt_cost": 0.000119,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.545,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.000263,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000263,
          "upstream_inference_prompt_cost": 0.000119,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.179,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 102,
        "completion_tokens": 71,
        "total_tokens": 173,
        "cost": 0.000528,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000528,
          "upstream_inference_prompt_cost": 0.000102,
          "upstream_inference_completions_cost": 0.000426
        },
        "completion_tokens_details": {
          "reasoning_tokens": 45,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.101,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 102,
        "completion_tokens": 64,
        "total_tokens": 166,
        "cost": 0.000486,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000486,
          "upstream_inference_prompt_cost": 0.000102,
          "upstream_inference_completions_cost": 0.000384
        },
        "completion_tokens_details": {
          "reasoning_tokens": 38,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.706,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 58,
        "total_tokens": 173,
        "cost": 0.000463,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000463,
          "upstream_inference_prompt_cost": 0.000115,
          "upstream_inference_completions_cost": 0.000348
        },
        "completion_tokens_details": {
          "reasoning_tokens": 36,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.582,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 55,
        "total_tokens": 170,
        "cost": 0.000445,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000445,
          "upstream_inference_prompt_cost": 0.000115,
          "upstream_inference_completions_cost": 0.00033
        },
        "completion_tokens_details": {
          "reasoning_tokens": 33,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.722,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 56,
        "total_tokens": 153,
        "cost": 0.000433,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000433,
          "upstream_inference_prompt_cost": 9.7e-05,
          "upstream_inference_completions_cost": 0.000336
        },
        "completion_tokens_details": {
          "reasoning_tokens": 34,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.458,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 55,
        "total_tokens": 152,
        "cost": 0.000427,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000427,
          "upstream_inference_prompt_cost": 9.7e-05,
          "upstream_inference_completions_cost": 0.00033
        },
        "completion_tokens_details": {
          "reasoning_tokens": 33,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.38,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 99,
        "completion_tokens": 24,
        "total_tokens": 123,
        "cost": 0.000243,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000243,
          "upstream_inference_prompt_cost": 9.9e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.274,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 99,
        "completion_tokens": 24,
        "total_tokens": 123,
        "cost": 0.000243,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000243,
          "upstream_inference_prompt_cost": 9.9e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.487,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.000237,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000237,
          "upstream_inference_prompt_cost": 9.3e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.499,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.000237,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000237,
          "upstream_inference_prompt_cost": 9.3e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.129,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 129,
        "completion_tokens": 61,
        "total_tokens": 190,
        "cost": 0.000495,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000495,
          "upstream_inference_prompt_cost": 0.000129,
          "upstream_inference_completions_cost": 0.000366
        },
        "completion_tokens_details": {
          "reasoning_tokens": 43,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.512,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 129,
        "completion_tokens": 59,
        "total_tokens": 188,
        "cost": 0.000483,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000483,
          "upstream_inference_prompt_cost": 0.000129,
          "upstream_inference_completions_cost": 0.000354
        },
        "completion_tokens_details": {
          "reasoning_tokens": 41,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 5.299,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 60,
        "total_tokens": 184,
        "cost": 0.000484,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000484,
          "upstream_inference_prompt_cost": 0.000124,
          "upstream_inference_completions_cost": 0.00036
        },
        "completion_tokens_details": {
          "reasoning_tokens": 42,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.527,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 65,
        "total_tokens": 189,
        "cost": 0.000514,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000514,
          "upstream_inference_prompt_cost": 0.000124,
          "upstream_inference_completions_cost": 0.00039
        },
        "completion_tokens_details": {
          "reasoning_tokens": 47,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.602,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 24,
        "total_tokens": 121,
        "cost": 0.000241,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000241,
          "upstream_inference_prompt_cost": 9.7e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.092,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 24,
        "total_tokens": 121,
        "cost": 0.000241,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000241,
          "upstream_inference_prompt_cost": 9.7e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.361,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.000237,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000237,
          "upstream_inference_prompt_cost": 9.3e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.195,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.000237,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000237,
          "upstream_inference_prompt_cost": 9.3e-05,
          "upstream_inference_completions_cost": 0.000144
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.466,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 132,
        "completion_tokens": 95,
        "total_tokens": 227,
        "cost": 0.000702,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000702,
          "upstream_inference_prompt_cost": 0.000132,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 66,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.64,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 132,
        "completion_tokens": 110,
        "total_tokens": 242,
        "cost": 0.000792,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000792,
          "upstream_inference_prompt_cost": 0.000132,
          "upstream_inference_completions_cost": 0.00066
        },
        "completion_tokens_details": {
          "reasoning_tokens": 81,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.918,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 113,
        "completion_tokens": 96,
        "total_tokens": 209,
        "cost": 0.000689,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000689,
          "upstream_inference_prompt_cost": 0.000113,
          "upstream_inference_completions_cost": 0.000576
        },
        "completion_tokens_details": {
          "reasoning_tokens": 67,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.055,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 113,
        "completion_tokens": 95,
        "total_tokens": 208,
        "cost": 0.000683,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000683,
          "upstream_inference_prompt_cost": 0.000113,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 66,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.18,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 20,
        "total_tokens": 123,
        "cost": 0.000223,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000223,
          "upstream_inference_prompt_cost": 0.000103,
          "upstream_inference_completions_cost": 0.00012
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.436,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 20,
        "total_tokens": 123,
        "cost": 0.000223,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000223,
          "upstream_inference_prompt_cost": 0.000103,
          "upstream_inference_completions_cost": 0.00012
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.153,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open_vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open_vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 19,
        "total_tokens": 114,
        "cost": 0.000209,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000209,
          "upstream_inference_prompt_cost": 9.5e-05,
          "upstream_inference_completions_cost": 0.000114
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.21,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open_vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open_vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 19,
        "total_tokens": 114,
        "cost": 0.000209,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000209,
          "upstream_inference_prompt_cost": 9.5e-05,
          "upstream_inference_completions_cost": 0.000114
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.171,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 89,
        "completion_tokens": 76,
        "total_tokens": 165,
        "cost": 0.000545,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000545,
          "upstream_inference_prompt_cost": 8.9e-05,
          "upstream_inference_completions_cost": 0.000456
        },
        "completion_tokens_details": {
          "reasoning_tokens": 54,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.01,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 89,
        "completion_tokens": 78,
        "total_tokens": 167,
        "cost": 0.000557,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000557,
          "upstream_inference_prompt_cost": 8.9e-05,
          "upstream_inference_completions_cost": 0.000468
        },
        "completion_tokens_details": {
          "reasoning_tokens": 56,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 3.258,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 66,
        "total_tokens": 161,
        "cost": 0.000491,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000491,
          "upstream_inference_prompt_cost": 9.5e-05,
          "upstream_inference_completions_cost": 0.000396
        },
        "completion_tokens_details": {
          "reasoning_tokens": 44,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.616,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 62,
        "total_tokens": 157,
        "cost": 0.000467,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000467,
          "upstream_inference_prompt_cost": 9.5e-05,
          "upstream_inference_completions_cost": 0.000372
        },
        "completion_tokens_details": {
          "reasoning_tokens": 40,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.581,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 38,
        "total_tokens": 149,
        "cost": 0.000339,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000339,
          "upstream_inference_prompt_cost": 0.000111,
          "upstream_inference_completions_cost": 0.000228
        },
        "completion_tokens_details": {
          "reasoning_tokens": 19,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.452,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 37,
        "total_tokens": 148,
        "cost": 0.000333,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000333,
          "upstream_inference_prompt_cost": 0.000111,
          "upstream_inference_completions_cost": 0.000222
        },
        "completion_tokens_details": {
          "reasoning_tokens": 18,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.686,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 40,
        "total_tokens": 151,
        "cost": 0.000351,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000351,
          "upstream_inference_prompt_cost": 0.000111,
          "upstream_inference_completions_cost": 0.00024
        },
        "completion_tokens_details": {
          "reasoning_tokens": 21,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.41,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 43,
        "total_tokens": 154,
        "cost": 0.000369,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000369,
          "upstream_inference_prompt_cost": 0.000111,
          "upstream_inference_completions_cost": 0.000258
        },
        "completion_tokens_details": {
          "reasoning_tokens": 24,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.253,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 66,
        "completion_tokens": 18,
        "total_tokens": 84,
        "cost": 0.000174,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000174,
          "upstream_inference_prompt_cost": 6.6e-05,
          "upstream_inference_completions_cost": 0.000108
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.216,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 66,
        "completion_tokens": 18,
        "total_tokens": 84,
        "cost": 0.000174,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000174,
          "upstream_inference_prompt_cost": 6.6e-05,
          "upstream_inference_completions_cost": 0.000108
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.559,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 61,
        "completion_tokens": 18,
        "total_tokens": 79,
        "cost": 0.000169,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000169,
          "upstream_inference_prompt_cost": 6.1e-05,
          "upstream_inference_completions_cost": 0.000108
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.235,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 61,
        "completion_tokens": 18,
        "total_tokens": 79,
        "cost": 0.000169,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000169,
          "upstream_inference_prompt_cost": 6.1e-05,
          "upstream_inference_completions_cost": 0.000108
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.307,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 20,
        "total_tokens": 112,
        "cost": 0.000212,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000212,
          "upstream_inference_prompt_cost": 9.2e-05,
          "upstream_inference_completions_cost": 0.00012
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.202,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 20,
        "total_tokens": 112,
        "cost": 0.000212,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000212,
          "upstream_inference_prompt_cost": 9.2e-05,
          "upstream_inference_completions_cost": 0.00012
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.138,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 77,
        "completion_tokens": 20,
        "total_tokens": 97,
        "cost": 0.000197,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000197,
          "upstream_inference_prompt_cost": 7.7e-05,
          "upstream_inference_completions_cost": 0.00012
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.595,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 77,
        "completion_tokens": 20,
        "total_tokens": 97,
        "cost": 0.000197,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000197,
          "upstream_inference_prompt_cost": 7.7e-05,
          "upstream_inference_completions_cost": 0.00012
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.201,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 47,
        "total_tokens": 139,
        "cost": 0.000374,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000374,
          "upstream_inference_prompt_cost": 9.2e-05,
          "upstream_inference_completions_cost": 0.000282
        },
        "completion_tokens_details": {
          "reasoning_tokens": 18,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.818,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 54,
        "total_tokens": 146,
        "cost": 0.000416,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000416,
          "upstream_inference_prompt_cost": 9.2e-05,
          "upstream_inference_completions_cost": 0.000324
        },
        "completion_tokens_details": {
          "reasoning_tokens": 25,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.313,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 81,
        "completion_tokens": 63,
        "total_tokens": 144,
        "cost": 0.000459,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000459,
          "upstream_inference_prompt_cost": 8.1e-05,
          "upstream_inference_completions_cost": 0.000378
        },
        "completion_tokens_details": {
          "reasoning_tokens": 34,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.783,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 81,
        "completion_tokens": 65,
        "total_tokens": 146,
        "cost": 0.000471,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000471,
          "upstream_inference_prompt_cost": 8.1e-05,
          "upstream_inference_completions_cost": 0.00039
        },
        "completion_tokens_details": {
          "reasoning_tokens": 36,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.396,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 106,
        "completion_tokens": 112,
        "total_tokens": 218,
        "cost": 0.000778,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000778,
          "upstream_inference_prompt_cost": 0.000106,
          "upstream_inference_completions_cost": 0.000672
        },
        "completion_tokens_details": {
          "reasoning_tokens": 91,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.043,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 106,
        "completion_tokens": 87,
        "total_tokens": 193,
        "cost": 0.000628,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000628,
          "upstream_inference_prompt_cost": 0.000106,
          "upstream_inference_completions_cost": 0.000522
        },
        "completion_tokens_details": {
          "reasoning_tokens": 66,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 2.206,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 19,
        "total_tokens": 133,
        "cost": 0.000228,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000228,
          "upstream_inference_prompt_cost": 0.000114,
          "upstream_inference_completions_cost": 0.000114
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.116,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-terra",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 77,
        "total_tokens": 191,
        "cost": 0.000576,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000576,
          "upstream_inference_prompt_cost": 0.000114,
          "upstream_inference_completions_cost": 0.000462
        },
        "completion_tokens_details": {
          "reasoning_tokens": 56,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.741,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 139,
        "completion_tokens": 17,
        "total_tokens": 156,
        "cost": 0.001205,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001205,
          "upstream_inference_prompt_cost": 0.000695,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 14.18,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 139,
        "completion_tokens": 17,
        "total_tokens": 156,
        "cost": 0.001205,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001205,
          "upstream_inference_prompt_cost": 0.000695,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 19.33,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 121,
        "completion_tokens": 17,
        "total_tokens": 138,
        "cost": 0.001115,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001115,
          "upstream_inference_prompt_cost": 0.000605,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 1.397,
      "error": null
    },
    {
      "task_id": "test-math-01-selection",
      "split": "held_out",
      "category": "constrained selection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "node": "Cedar",
        "cost": 8
      },
      "response": {
        "node": "Cedar",
        "cost": 8
      },
      "raw_response": "{\"node\":\"Cedar\",\"cost\":8}",
      "usage": {
        "prompt_tokens": 121,
        "completion_tokens": 17,
        "total_tokens": 138,
        "cost": 0.001115,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001115,
          "upstream_inference_prompt_cost": 0.000605,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 7.391,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 67,
        "total_tokens": 174,
        "cost": 0.002545,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.002545,
          "upstream_inference_prompt_cost": 0.000535,
          "upstream_inference_completions_cost": 0.00201
        },
        "completion_tokens_details": {
          "reasoning_tokens": 43,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.289,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 107,
        "completion_tokens": 64,
        "total_tokens": 171,
        "cost": 0.002455,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.002455,
          "upstream_inference_prompt_cost": 0.000535,
          "upstream_inference_completions_cost": 0.00192
        },
        "completion_tokens_details": {
          "reasoning_tokens": 40,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.859,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 69,
        "total_tokens": 164,
        "cost": 0.002545,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.002545,
          "upstream_inference_prompt_cost": 0.000475,
          "upstream_inference_completions_cost": 0.00207
        },
        "completion_tokens_details": {
          "reasoning_tokens": 45,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 26.618,
      "error": null
    },
    {
      "task_id": "test-math-02-lexicographic",
      "split": "held_out",
      "category": "lexicographic allocation",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "response": {
        "replication": 7,
        "theory": 5,
        "communication": 3
      },
      "raw_response": "{\"replication\":7,\"theory\":5,\"communication\":3}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 69,
        "total_tokens": 164,
        "cost": 0.002545,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.002545,
          "upstream_inference_prompt_cost": 0.000475,
          "upstream_inference_completions_cost": 0.00207
        },
        "completion_tokens_details": {
          "reasoning_tokens": 45,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 32.772,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 91,
        "completion_tokens": 99,
        "total_tokens": 190,
        "cost": 0.003425,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.003425,
          "upstream_inference_prompt_cost": 0.000455,
          "upstream_inference_completions_cost": 0.00297
        },
        "completion_tokens_details": {
          "reasoning_tokens": 74,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.882,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 91,
        "completion_tokens": 110,
        "total_tokens": 201,
        "cost": 0.003755,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.003755,
          "upstream_inference_prompt_cost": 0.000455,
          "upstream_inference_completions_cost": 0.0033
        },
        "completion_tokens_details": {
          "reasoning_tokens": 85,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.412,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 107,
        "total_tokens": 204,
        "cost": 0.003695,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.003695,
          "upstream_inference_prompt_cost": 0.000485,
          "upstream_inference_completions_cost": 0.00321
        },
        "completion_tokens_details": {
          "reasoning_tokens": 82,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.969,
      "error": null
    },
    {
      "task_id": "test-math-03-topological",
      "split": "held_out",
      "category": "precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "response": {
        "order": [
          "A",
          "B",
          "C",
          "D",
          "E",
          "F"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"B\",\"C\",\"D\",\"E\",\"F\"]}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 96,
        "total_tokens": 193,
        "cost": 0.003365,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.003365,
          "upstream_inference_prompt_cost": 0.000485,
          "upstream_inference_completions_cost": 0.00288
        },
        "completion_tokens_details": {
          "reasoning_tokens": 71,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.905,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 123,
        "completion_tokens": 26,
        "total_tokens": 149,
        "cost": 0.001395,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001395,
          "upstream_inference_prompt_cost": 0.000615,
          "upstream_inference_completions_cost": 0.00078
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.213,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 123,
        "completion_tokens": 26,
        "total_tokens": 149,
        "cost": 0.001395,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001395,
          "upstream_inference_prompt_cost": 0.000615,
          "upstream_inference_completions_cost": 0.00078
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 35.906,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 26,
        "total_tokens": 130,
        "cost": 0.0013,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0013,
          "upstream_inference_prompt_cost": 0.00052,
          "upstream_inference_completions_cost": 0.00078
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.277,
      "error": null
    },
    {
      "task_id": "test-math-04-stop-gates",
      "split": "held_out",
      "category": "boolean gates",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "response": {
        "release": false,
        "blocking": [
          "consent unverified",
          "risk not below 4"
        ]
      },
      "raw_response": "{\"release\":false,\"blocking\":[\"consent unverified\",\"risk not below 4\"]}",
      "usage": {
        "prompt_tokens": 104,
        "completion_tokens": 26,
        "total_tokens": 130,
        "cost": 0.0013,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0013,
          "upstream_inference_prompt_cost": 0.00052,
          "upstream_inference_completions_cost": 0.00078
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.485,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 152,
        "completion_tokens": 31,
        "total_tokens": 183,
        "cost": 0.00169,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00169,
          "upstream_inference_prompt_cost": 0.00076,
          "upstream_inference_completions_cost": 0.00093
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.736,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 152,
        "completion_tokens": 31,
        "total_tokens": 183,
        "cost": 0.00169,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00169,
          "upstream_inference_prompt_cost": 0.00076,
          "upstream_inference_completions_cost": 0.00093
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 35.893,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 31,
        "total_tokens": 145,
        "cost": 0.0015,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0015,
          "upstream_inference_prompt_cost": 0.00057,
          "upstream_inference_completions_cost": 0.00093
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 35.941,
      "error": null
    },
    {
      "task_id": "test-math-05-provenance",
      "split": "held_out",
      "category": "source precedence",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "response": {
        "population": 1200,
        "timezone": "UTC+1",
        "population_source": "audited report",
        "timezone_source": "registry"
      },
      "raw_response": "{\"population\":1200,\"timezone\":\"UTC+1\",\"population_source\":\"audited report\",\"timezone_source\":\"registry\"}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 31,
        "total_tokens": 145,
        "cost": 0.0015,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0015,
          "upstream_inference_prompt_cost": 0.00057,
          "upstream_inference_completions_cost": 0.00093
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.652,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 136,
        "completion_tokens": 24,
        "total_tokens": 160,
        "cost": 0.0014,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0014,
          "upstream_inference_prompt_cost": 0.00068,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.058,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 136,
        "completion_tokens": 24,
        "total_tokens": 160,
        "cost": 0.0014,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0014,
          "upstream_inference_prompt_cost": 0.00068,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 35.807,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.001315,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001315,
          "upstream_inference_prompt_cost": 0.000595,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.558,
      "error": null
    },
    {
      "task_id": "test-math-06-nested-exception",
      "split": "held_out",
      "category": "nested exception",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "response": {
        "allow": [
          "B",
          "D"
        ],
        "deny": [
          "A",
          "C",
          "E"
        ]
      },
      "raw_response": "{\"allow\":[\"B\",\"D\"],\"deny\":[\"A\",\"C\",\"E\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.001315,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001315,
          "upstream_inference_prompt_cost": 0.000595,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.302,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 109,
        "completion_tokens": 19,
        "total_tokens": 128,
        "cost": 0.001115,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001115,
          "upstream_inference_prompt_cost": 0.000545,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.537,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 109,
        "completion_tokens": 19,
        "total_tokens": 128,
        "cost": 0.001115,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001115,
          "upstream_inference_prompt_cost": 0.000545,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.145,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 79,
        "completion_tokens": 19,
        "total_tokens": 98,
        "cost": 0.000965,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000965,
          "upstream_inference_prompt_cost": 0.000395,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.46,
      "error": null
    },
    {
      "task_id": "test-math-07-unit-chain",
      "split": "held_out",
      "category": "unit conversion",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "response": {
        "duration_seconds": 200,
        "volume_ml": 8000
      },
      "raw_response": "{\"duration_seconds\":200,\"volume_ml\":8000}",
      "usage": {
        "prompt_tokens": 79,
        "completion_tokens": 19,
        "total_tokens": 98,
        "cost": 0.000965,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000965,
          "upstream_inference_prompt_cost": 0.000395,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.393,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.001315,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001315,
          "upstream_inference_prompt_cost": 0.000595,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.239,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 119,
        "completion_tokens": 24,
        "total_tokens": 143,
        "cost": 0.001315,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001315,
          "upstream_inference_prompt_cost": 0.000595,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.179,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 102,
        "completion_tokens": 24,
        "total_tokens": 126,
        "cost": 0.00123,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00123,
          "upstream_inference_prompt_cost": 0.00051,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.11,
      "error": null
    },
    {
      "task_id": "test-math-08-access-set",
      "split": "held_out",
      "category": "set membership",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "response": {
        "granted": [
          "Kai"
        ],
        "denied": [
          "Luz",
          "Mina"
        ]
      },
      "raw_response": "{\"granted\":[\"Kai\"],\"denied\":[\"Luz\",\"Mina\"]}",
      "usage": {
        "prompt_tokens": 102,
        "completion_tokens": 24,
        "total_tokens": 126,
        "cost": 0.00123,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00123,
          "upstream_inference_prompt_cost": 0.00051,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.485,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 20,
        "total_tokens": 135,
        "cost": 0.001175,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001175,
          "upstream_inference_prompt_cost": 0.000575,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.177,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 115,
        "completion_tokens": 20,
        "total_tokens": 135,
        "cost": 0.001175,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001175,
          "upstream_inference_prompt_cost": 0.000575,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.436,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 20,
        "total_tokens": 117,
        "cost": 0.001085,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001085,
          "upstream_inference_prompt_cost": 0.000485,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.806,
      "error": null
    },
    {
      "task_id": "test-math-09-time-window",
      "split": "held_out",
      "category": "temporal interval",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "start": "10:00",
        "end": "10:45"
      },
      "response": {
        "start": "10:00",
        "end": "10:45"
      },
      "raw_response": "{\"start\":\"10:00\",\"end\":\"10:45\"}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 20,
        "total_tokens": 117,
        "cost": 0.001085,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001085,
          "upstream_inference_prompt_cost": 0.000485,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.24,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 99,
        "completion_tokens": 24,
        "total_tokens": 123,
        "cost": 0.001215,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001215,
          "upstream_inference_prompt_cost": 0.000495,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.809,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 99,
        "completion_tokens": 24,
        "total_tokens": 123,
        "cost": 0.001215,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001215,
          "upstream_inference_prompt_cost": 0.000495,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.386,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.001185,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001185,
          "upstream_inference_prompt_cost": 0.000465,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.584,
      "error": null
    },
    {
      "task_id": "test-math-10-output-schema",
      "split": "held_out",
      "category": "typed output contract",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "response": {
        "id": "Z9",
        "accepted": true,
        "score_percent": 87.5
      },
      "raw_response": "{\"id\":\"Z9\",\"accepted\":true,\"score_percent\":87.5}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.001185,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001185,
          "upstream_inference_prompt_cost": 0.000465,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.535,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 129,
        "completion_tokens": 59,
        "total_tokens": 188,
        "cost": 0.002415,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.002415,
          "upstream_inference_prompt_cost": 0.000645,
          "upstream_inference_completions_cost": 0.00177
        },
        "completion_tokens_details": {
          "reasoning_tokens": 41,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 39.943,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 129,
        "completion_tokens": 62,
        "total_tokens": 191,
        "cost": 0.002505,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.002505,
          "upstream_inference_prompt_cost": 0.000645,
          "upstream_inference_completions_cost": 0.00186
        },
        "completion_tokens_details": {
          "reasoning_tokens": 44,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.033,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 66,
        "total_tokens": 190,
        "cost": 0.0026,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0026,
          "upstream_inference_prompt_cost": 0.00062,
          "upstream_inference_completions_cost": 0.00198
        },
        "completion_tokens_details": {
          "reasoning_tokens": 48,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.42,
      "error": null
    },
    {
      "task_id": "test-math-11-weighted-score",
      "split": "held_out",
      "category": "scoring function",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "project": "R",
        "score": 17
      },
      "response": {
        "project": "R",
        "score": 17
      },
      "raw_response": "{\"project\":\"R\",\"score\":17}",
      "usage": {
        "prompt_tokens": 124,
        "completion_tokens": 66,
        "total_tokens": 190,
        "cost": 0.0026,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.0026,
          "upstream_inference_prompt_cost": 0.00062,
          "upstream_inference_completions_cost": 0.00198
        },
        "completion_tokens_details": {
          "reasoning_tokens": 48,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.985,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 24,
        "total_tokens": 121,
        "cost": 0.001205,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001205,
          "upstream_inference_prompt_cost": 0.000485,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.845,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 97,
        "completion_tokens": 24,
        "total_tokens": 121,
        "cost": 0.001205,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001205,
          "upstream_inference_prompt_cost": 0.000485,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 44.093,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.001185,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001185,
          "upstream_inference_prompt_cost": 0.000465,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 39.99,
      "error": null
    },
    {
      "task_id": "test-math-12-overlap",
      "split": "held_out",
      "category": "interval intersection",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "response": {
        "overlap_start": 20,
        "overlap_end": 35,
        "duration": 15
      },
      "raw_response": "{\"overlap_start\":20,\"overlap_end\":35,\"duration\":15}",
      "usage": {
        "prompt_tokens": 93,
        "completion_tokens": 24,
        "total_tokens": 117,
        "cost": 0.001185,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001185,
          "upstream_inference_prompt_cost": 0.000465,
          "upstream_inference_completions_cost": 0.00072
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 40.663,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 132,
        "completion_tokens": 27,
        "total_tokens": 159,
        "cost": 0.00147,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00147,
          "upstream_inference_prompt_cost": 0.00066,
          "upstream_inference_completions_cost": 0.00081
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.809,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 132,
        "completion_tokens": 27,
        "total_tokens": 159,
        "cost": 0.00147,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00147,
          "upstream_inference_prompt_cost": 0.00066,
          "upstream_inference_completions_cost": 0.00081
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 35.319,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 113,
        "completion_tokens": 101,
        "total_tokens": 214,
        "cost": 0.003595,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.003595,
          "upstream_inference_prompt_cost": 0.000565,
          "upstream_inference_completions_cost": 0.00303
        },
        "completion_tokens_details": {
          "reasoning_tokens": 72,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.298,
      "error": null
    },
    {
      "task_id": "test-math-13-set-cover",
      "split": "held_out",
      "category": "set coverage",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "response": {
        "team": [
          "Ada",
          "Cy"
        ],
        "covered_skills": [
          "analysis",
          "coding",
          "outreach"
        ]
      },
      "raw_response": "{\"team\":[\"Ada\",\"Cy\"],\"covered_skills\":[\"analysis\",\"coding\",\"outreach\"]}",
      "usage": {
        "prompt_tokens": 113,
        "completion_tokens": 109,
        "total_tokens": 222,
        "cost": 0.003835,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.003835,
          "upstream_inference_prompt_cost": 0.000565,
          "upstream_inference_completions_cost": 0.00327
        },
        "completion_tokens_details": {
          "reasoning_tokens": 80,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.554,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 20,
        "total_tokens": 123,
        "cost": 0.001115,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001115,
          "upstream_inference_prompt_cost": 0.000515,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.596,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open the vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 103,
        "completion_tokens": 20,
        "total_tokens": 123,
        "cost": 0.001115,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001115,
          "upstream_inference_prompt_cost": 0.000515,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.34,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open_vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open_vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 19,
        "total_tokens": 114,
        "cost": 0.001045,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001045,
          "upstream_inference_prompt_cost": 0.000475,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.283,
      "error": null
    },
    {
      "task_id": "test-math-14-workflow",
      "split": "held_out",
      "category": "state transition",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.5,
      "expected": {
        "next_action": "open the vote",
        "may_publish": false
      },
      "response": {
        "next_action": "open_vote",
        "may_publish": false
      },
      "raw_response": "{\"next_action\":\"open_vote\",\"may_publish\":false}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 19,
        "total_tokens": 114,
        "cost": 0.001045,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001045,
          "upstream_inference_prompt_cost": 0.000475,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 31.735,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 89,
        "completion_tokens": 20,
        "total_tokens": 109,
        "cost": 0.001045,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001045,
          "upstream_inference_prompt_cost": 0.000445,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.524,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": false,
      "constraint_score": 0.3333333333333333,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 6,
        "y": 5,
        "difference": 1
      },
      "raw_response": "{\"x\":6,\"y\":5,\"difference\":1}",
      "usage": {
        "prompt_tokens": 89,
        "completion_tokens": 20,
        "total_tokens": 109,
        "cost": 0.001045,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001045,
          "upstream_inference_prompt_cost": 0.000445,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.405,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 20,
        "total_tokens": 115,
        "cost": 0.001075,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001075,
          "upstream_inference_prompt_cost": 0.000475,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 35.235,
      "error": null
    },
    {
      "task_id": "test-math-15-satisfiability",
      "split": "held_out",
      "category": "constraint satisfiability",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "response": {
        "x": 5,
        "y": 6,
        "difference": 1
      },
      "raw_response": "{\"x\":5,\"y\":6,\"difference\":1}",
      "usage": {
        "prompt_tokens": 95,
        "completion_tokens": 20,
        "total_tokens": 115,
        "cost": 0.001075,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001075,
          "upstream_inference_prompt_cost": 0.000475,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.485,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 17,
        "total_tokens": 128,
        "cost": 0.001065,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001065,
          "upstream_inference_prompt_cost": 0.000555,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.675,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 17,
        "total_tokens": 128,
        "cost": 0.001065,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001065,
          "upstream_inference_prompt_cost": 0.000555,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.447,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 17,
        "total_tokens": 128,
        "cost": 0.001065,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001065,
          "upstream_inference_prompt_cost": 0.000555,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.735,
      "error": null
    },
    {
      "task_id": "test-math-16-information-value",
      "split": "held_out",
      "category": "decision theory",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "test": "A",
        "net_value": 5
      },
      "response": {
        "test": "A",
        "net_value": 5
      },
      "raw_response": "{\"test\":\"A\",\"net_value\":5}",
      "usage": {
        "prompt_tokens": 111,
        "completion_tokens": 17,
        "total_tokens": 128,
        "cost": 0.001065,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001065,
          "upstream_inference_prompt_cost": 0.000555,
          "upstream_inference_completions_cost": 0.00051
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.458,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 66,
        "completion_tokens": 18,
        "total_tokens": 84,
        "cost": 0.00087,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00087,
          "upstream_inference_prompt_cost": 0.00033,
          "upstream_inference_completions_cost": 0.00054
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.766,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 66,
        "completion_tokens": 18,
        "total_tokens": 84,
        "cost": 0.00087,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00087,
          "upstream_inference_prompt_cost": 0.00033,
          "upstream_inference_completions_cost": 0.00054
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.631,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 61,
        "completion_tokens": 18,
        "total_tokens": 79,
        "cost": 0.000845,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000845,
          "upstream_inference_prompt_cost": 0.000305,
          "upstream_inference_completions_cost": 0.00054
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.402,
      "error": null
    },
    {
      "task_id": "test-math-17-simple-lookup",
      "split": "held_out",
      "category": "negative control: lookup",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "codename": "Lark",
        "revision": 6
      },
      "response": {
        "codename": "Lark",
        "revision": 6
      },
      "raw_response": "{\"codename\":\"Lark\",\"revision\":6}",
      "usage": {
        "prompt_tokens": 61,
        "completion_tokens": 18,
        "total_tokens": 79,
        "cost": 0.000845,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000845,
          "upstream_inference_prompt_cost": 0.000305,
          "upstream_inference_completions_cost": 0.00054
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.396,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 20,
        "total_tokens": 112,
        "cost": 0.00106,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00106,
          "upstream_inference_prompt_cost": 0.00046,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 39.256,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 20,
        "total_tokens": 112,
        "cost": 0.00106,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00106,
          "upstream_inference_prompt_cost": 0.00046,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.735,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 77,
        "completion_tokens": 36,
        "total_tokens": 113,
        "cost": 0.001465,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001465,
          "upstream_inference_prompt_cost": 0.000385,
          "upstream_inference_completions_cost": 0.00108
        },
        "completion_tokens_details": {
          "reasoning_tokens": 14,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.967,
      "error": null
    },
    {
      "task_id": "test-math-18-simple-classification",
      "split": "held_out",
      "category": "negative control: direct classification",
      "notation_expected_to_help": false,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "response": {
        "urgent": [
          "A",
          "C"
        ],
        "routine": [
          "B"
        ]
      },
      "raw_response": "{\"urgent\":[\"A\",\"C\"],\"routine\":[\"B\"]}",
      "usage": {
        "prompt_tokens": 77,
        "completion_tokens": 20,
        "total_tokens": 97,
        "cost": 0.000985,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.000985,
          "upstream_inference_prompt_cost": 0.000385,
          "upstream_inference_completions_cost": 0.0006
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.198,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 27,
        "total_tokens": 119,
        "cost": 0.00127,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00127,
          "upstream_inference_prompt_cost": 0.00046,
          "upstream_inference_completions_cost": 0.00081
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.088,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 92,
        "completion_tokens": 27,
        "total_tokens": 119,
        "cost": 0.00127,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00127,
          "upstream_inference_prompt_cost": 0.00046,
          "upstream_inference_completions_cost": 0.00081
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.164,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 81,
        "completion_tokens": 27,
        "total_tokens": 108,
        "cost": 0.001215,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001215,
          "upstream_inference_prompt_cost": 0.000405,
          "upstream_inference_completions_cost": 0.00081
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 36.888,
      "error": null
    },
    {
      "task_id": "test-math-19-arithmetic-chain",
      "split": "held_out",
      "category": "arithmetic chain",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "response": {
        "after_removal": 36,
        "after_addition": 43,
        "per_bin": 8.6
      },
      "raw_response": "{\"after_removal\":36,\"after_addition\":43,\"per_bin\":8.6}",
      "usage": {
        "prompt_tokens": 81,
        "completion_tokens": 27,
        "total_tokens": 108,
        "cost": 0.001215,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.001215,
          "upstream_inference_prompt_cost": 0.000405,
          "upstream_inference_completions_cost": 0.00081
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.234,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 106,
        "completion_tokens": 91,
        "total_tokens": 197,
        "cost": 0.00326,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00326,
          "upstream_inference_prompt_cost": 0.00053,
          "upstream_inference_completions_cost": 0.00273
        },
        "completion_tokens_details": {
          "reasoning_tokens": 70,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 38.83,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "formal",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 106,
        "completion_tokens": 91,
        "total_tokens": 197,
        "cost": 0.00326,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00326,
          "upstream_inference_prompt_cost": 0.00053,
          "upstream_inference_completions_cost": 0.00273
        },
        "completion_tokens_details": {
          "reasoning_tokens": 70,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 39.859,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 1,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 19,
        "total_tokens": 133,
        "cost": 0.00114,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00114,
          "upstream_inference_prompt_cost": 0.00057,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 35.941,
      "error": null
    },
    {
      "task_id": "test-math-20-conditional-priority",
      "split": "held_out",
      "category": "conditional priority",
      "notation_expected_to_help": true,
      "model": "openai/gpt-5.6-sol",
      "condition": "vernacular",
      "repetition": 2,
      "valid_json": true,
      "on_task": true,
      "constraint_score": 1.0,
      "expected": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "response": {
        "order": [
          "A",
          "C",
          "B",
          "D"
        ]
      },
      "raw_response": "{\"order\":[\"A\",\"C\",\"B\",\"D\"]}",
      "usage": {
        "prompt_tokens": 114,
        "completion_tokens": 19,
        "total_tokens": 133,
        "cost": 0.00114,
        "is_byok": false,
        "prompt_tokens_details": {
          "cached_tokens": 0,
          "cache_write_tokens": 0,
          "audio_tokens": 0,
          "video_tokens": 0
        },
        "cost_details": {
          "upstream_inference_cost": 0.00114,
          "upstream_inference_prompt_cost": 0.00057,
          "upstream_inference_completions_cost": 0.00057
        },
        "completion_tokens_details": {
          "reasoning_tokens": 0,
          "image_tokens": 0,
          "audio_tokens": 0
        }
      },
      "provider": "OpenAI",
      "latency_seconds": 37.391,
      "error": null
    }
  ]
}
