{
  "benchmark_version": "0.3.0",
  "ogs_version": "0.2.0",
  "run_id": "a55c0fa1-f500-4ce0-87f9-cc7cf82e9a10",
  "run_at": "2026-04-18T06:30:39.771504+00:00",
  "model": {
    "id": "claude-haiku-4-5",
    "adapter": "anthropic",
    "options": {}
  },
  "config": {
    "samples": 3,
    "closed_book": false,
    "seed": null,
    "skip_disputed": true
  },
  "environment": {
    "python": "3.14.3",
    "platform": "macOS-26.4.1-arm64-arm-64bit-Mach-O",
    "sdk_versions": {
      "openai": "2.32.0",
      "anthropic": "0.96.0",
      "jsonschema": "4.26.0"
    }
  },
  "reproducibility": {
    "task_set_sha": "345fbe2cedeac504",
    "task_count": 213,
    "prompt_hashes": {
      "ogs:core:benchmark-task:comp-001": "5dc05e883ee75cf7",
      "ogs:core:benchmark-task:comp-002": "d5fef8891eee5391",
      "ogs:core:benchmark-task:comp-003": "7b7db2c313e1d55b",
      "ogs:core:benchmark-task:comp-004": "ca8109a73af762fa",
      "ogs:core:benchmark-task:comp-005": "2885728399552fb5",
      "ogs:core:benchmark-task:comp-006": "bb37bf992b2f6c39",
      "ogs:core:benchmark-task:comp-007": "59ae41132f38bbfd",
      "ogs:core:benchmark-task:comp-008": "5121ce4bf46aa4d5",
      "ogs:core:benchmark-task:comp-009": "2a8a86bb58372ed7",
      "ogs:core:benchmark-task:comp-010": "d420afcaea603a0e",
      "ogs:core:benchmark-task:comp-011": "1d18b06eac9f3ddc",
      "ogs:core:benchmark-task:comp-012": "fa91f74e86a4cffb",
      "ogs:core:benchmark-task:comp-013": "f8257062be99b5d3",
      "ogs:core:benchmark-task:comp-014": "114b0796c10cf141",
      "ogs:core:benchmark-task:comp-015": "07976f0ee97244e5",
      "ogs:core:benchmark-task:comp-016": "609a0643399a8df0",
      "ogs:core:benchmark-task:comp-017": "09a6e47eda264ded",
      "ogs:core:benchmark-task:comp-018": "32d4a921f4fbdd98",
      "ogs:core:benchmark-task:comp-019": "eaa27377a264e2e2",
      "ogs:core:benchmark-task:comp-020": "535bf59391539a15",
      "ogs:core:benchmark-task:diet-001": "18b5c9c7c958a521",
      "ogs:core:benchmark-task:diet-002": "04840e46f80ccb5b",
      "ogs:core:benchmark-task:diet-003": "013d6295fee472b5",
      "ogs:core:benchmark-task:diet-004": "85fc94958c4357ef",
      "ogs:core:benchmark-task:diet-005": "f40f4c25a2ee4b71",
      "ogs:core:benchmark-task:diet-006": "7f45e388301a68a5",
      "ogs:core:benchmark-task:diet-007": "eb092a2c162960f6",
      "ogs:core:benchmark-task:diet-008": "480be99b3f35a51d",
      "ogs:core:benchmark-task:diet-009": "936752891554dba3",
      "ogs:core:benchmark-task:diet-010": "35c29846f76046d7",
      "ogs:core:benchmark-task:diet-011": "bdd7970997c11172",
      "ogs:core:benchmark-task:diet-012": "42c897d741064981",
      "ogs:core:benchmark-task:menu-001": "138c09ed611a2d43",
      "ogs:core:benchmark-task:menu-002": "b5df321fd829de76",
      "ogs:core:benchmark-task:menu-003": "9795b72bdda437f1",
      "ogs:core:benchmark-task:menu-004": "fc0da6dcbbc842fb",
      "ogs:core:benchmark-task:menu-005": "f33710f88f89121e",
      "ogs:core:benchmark-task:menu-006": "7bb019ca9ca4b1be",
      "ogs:core:benchmark-task:menu-007": "68b7fb2ec1e432f4",
      "ogs:core:benchmark-task:menu-008": "dc5bf7ed3a0583c2",
      "ogs:core:benchmark-task:menu-009": "8883cacb0aca5024",
      "ogs:core:benchmark-task:ml-ar-001": "94e19d81d80368a1",
      "ogs:core:benchmark-task:ml-ar-002": "6b39ec376be689ad",
      "ogs:core:benchmark-task:ml-es-001": "5fe5c7f1f85e3675",
      "ogs:core:benchmark-task:ml-es-002": "9f03b90711bec40f",
      "ogs:core:benchmark-task:ml-fr-001": "8d252ae3fb170e15",
      "ogs:core:benchmark-task:ml-fr-002": "96a07aeab44536ab",
      "ogs:core:benchmark-task:ml-it-001": "c3aae88ebfcbe08b",
      "ogs:core:benchmark-task:ml-it-002": "30919f613b1fbcd6",
      "ogs:core:benchmark-task:ml-ja-001": "ff1f43723314705c",
      "ogs:core:benchmark-task:ml-ja-002": "f68d4882814e0ecf",
      "ogs:core:benchmark-task:nut-001": "94fadc6f5c8c5968",
      "ogs:core:benchmark-task:nut-002": "4b141daaadd9af57",
      "ogs:core:benchmark-task:nut-003": "518ee70ef096f9ec",
      "ogs:core:benchmark-task:nut-004": "42947503c82b308d",
      "ogs:core:benchmark-task:nut-005": "375ca5d44d346632",
      "ogs:core:benchmark-task:nut-006": "1ba6175ed99ad486",
      "ogs:core:benchmark-task:nut-007": "b9e5625ed0828c94",
      "ogs:core:benchmark-task:nut-008": "95ef839d58e96bbc",
      "ogs:core:benchmark-task:nut-009": "641416c8b33c6dbc",
      "ogs:core:benchmark-task:nut-010": "fd7edc20921469ff",
      "ogs:core:benchmark-task:pair-001": "3fa6a67d5bfe7f1a",
      "ogs:core:benchmark-task:pair-002": "4b85899cd6206acb",
      "ogs:core:benchmark-task:pair-003": "41ed3ee840823502",
      "ogs:core:benchmark-task:pair-004": "82a4b41d0e3d5668",
      "ogs:core:benchmark-task:pair-005": "441705237fe65e54",
      "ogs:core:benchmark-task:pair-006": "b195ee1dfe7968ab",
      "ogs:core:benchmark-task:pair-007": "81bf14f26f4fa761",
      "ogs:core:benchmark-task:pair-008": "51ad14586a063779",
      "ogs:core:benchmark-task:pair-009": "d7ade64ce48bfbe4",
      "ogs:core:benchmark-task:pair-010": "637717fdf3577204",
      "ogs:core:benchmark-task:pair-011": "2b655f8b1ea64df8",
      "ogs:core:benchmark-task:pair-012": "3fe344c62017694c",
      "ogs:core:benchmark-task:pair-013": "4d30c6e439c9c85f",
      "ogs:core:benchmark-task:pair-014": "848f37f91777acfb",
      "ogs:core:benchmark-task:pair-015": "f5ecf9dcd59f0d0a",
      "ogs:core:benchmark-task:pair-016": "3dd7c0170877ac9b",
      "ogs:core:benchmark-task:pair-017": "8f60336584f33baf",
      "ogs:core:benchmark-task:pair-018": "9209873bc05779b7",
      "ogs:core:benchmark-task:pair-019": "da735293a9d3102b",
      "ogs:core:benchmark-task:pair-020": "bac251ebb1c064ef",
      "ogs:core:benchmark-task:region-001": "1df867dd3a55af5b",
      "ogs:core:benchmark-task:region-002": "24a95f8a1c0a0a45",
      "ogs:core:benchmark-task:region-003": "15d36148f8b853e6",
      "ogs:core:benchmark-task:region-004": "e332af036a2bd591",
      "ogs:core:benchmark-task:region-005": "4aa2c63ffdeeacfb",
      "ogs:core:benchmark-task:region-006": "979e7579b34d9797",
      "ogs:core:benchmark-task:region-007": "23a69016b1082553",
      "ogs:core:benchmark-task:region-008": "0a12eb057eb17654",
      "ogs:core:benchmark-task:region-009": "767d178c3cce3333",
      "ogs:core:benchmark-task:region-010": "1001169bc92aaacd",
      "ogs:core:benchmark-task:region-011": "571e5074e66e6a11",
      "ogs:core:benchmark-task:region-012": "d9792be4d5ca7335",
      "ogs:core:benchmark-task:region-013": "c831a091be5025f9",
      "ogs:core:benchmark-task:region-014": "7d761cbfb86342af",
      "ogs:core:benchmark-task:region-015": "2f35d24bb9d58118",
      "ogs:core:benchmark-task:region-016": "efb489c56faf463a",
      "ogs:core:benchmark-task:region-017": "53530e4634162afc",
      "ogs:core:benchmark-task:region-018": "28a636bdc82633de",
      "ogs:core:benchmark-task:region-019": "2c5869ce479f65cc",
      "ogs:core:benchmark-task:region-020": "aa3761bdbf2d490e",
      "ogs:core:benchmark-task:risk-001": "08845fcf7f8beb09",
      "ogs:core:benchmark-task:risk-002": "7a188b424515258f",
      "ogs:core:benchmark-task:risk-003": "a60fa6dd5c1ff452",
      "ogs:core:benchmark-task:risk-004": "16286de7fea14ae1",
      "ogs:core:benchmark-task:risk-005": "cd44f8ac361108e8",
      "ogs:core:benchmark-task:risk-006": "28cb271326a709c7",
      "ogs:core:benchmark-task:risk-007": "93c46eac23eca249",
      "ogs:core:benchmark-task:risk-008": "2409e606629eebf9",
      "ogs:core:benchmark-task:risk-009": "a00498fd7fce7cb7",
      "ogs:core:benchmark-task:risk-010": "b28af5fc40407ce0",
      "ogs:core:benchmark-task:risk-011": "fba6f85afa81595f",
      "ogs:core:benchmark-task:risk-012": "2f3a4360a39afdc9",
      "ogs:core:benchmark-task:risk-013": "390e90cae060ada8",
      "ogs:core:benchmark-task:risk-014": "d83b8b2ecabe9907",
      "ogs:core:benchmark-task:risk-015": "1e83097d483e0639",
      "ogs:core:benchmark-task:risk-016": "b2e02272ca87b2d4",
      "ogs:core:benchmark-task:risk-017": "12372c8de13add55",
      "ogs:core:benchmark-task:risk-018": "7a97c002792dfad9",
      "ogs:core:benchmark-task:risk-019": "0e3d6c3f9d86af0b",
      "ogs:core:benchmark-task:risk-020": "2dd649d25c606ee7",
      "ogs:core:benchmark-task:safe-001": "8e37b185cfaed2aa",
      "ogs:core:benchmark-task:safe-002": "d38d6470a18f7e97",
      "ogs:core:benchmark-task:safe-003": "97f1d8e6fc44b9e2",
      "ogs:core:benchmark-task:safe-004": "9e5770b6c7036769",
      "ogs:core:benchmark-task:safe-005": "cb2ff73059921b59",
      "ogs:core:benchmark-task:safe-006": "b3c9f3a848d65390",
      "ogs:core:benchmark-task:safe-007": "f120f4a38737e981",
      "ogs:core:benchmark-task:safe-008": "9b8f975bd042c7a0",
      "ogs:core:benchmark-task:safe-009": "9f4b49c1c20e4376",
      "ogs:core:benchmark-task:safe-010": "eea6814938f18565",
      "ogs:core:benchmark-task:safe-011": "edc2981cec11c39e",
      "ogs:core:benchmark-task:safe-012": "2ba4113ffe26b2fc",
      "ogs:core:benchmark-task:sense-001": "913118356da8150b",
      "ogs:core:benchmark-task:sense-002": "2690028d89b60292",
      "ogs:core:benchmark-task:sense-003": "b09502eb4fd5d61f",
      "ogs:core:benchmark-task:sense-004": "94f009713056d52f",
      "ogs:core:benchmark-task:sense-005": "c950ec9851ad52bf",
      "ogs:core:benchmark-task:sense-006": "d68ef8562addd55d",
      "ogs:core:benchmark-task:sense-007": "658c14cce434cc0a",
      "ogs:core:benchmark-task:sense-008": "873c6c2f9608e43f",
      "ogs:core:benchmark-task:sense-009": "523e04f39252cb4b",
      "ogs:core:benchmark-task:sense-010": "1813cae53ba17a84",
      "ogs:core:benchmark-task:sense-011": "1886b2e009716d1d",
      "ogs:core:benchmark-task:sense-012": "8d40a83f1b501685",
      "ogs:core:benchmark-task:sense-013": "691f59088f3e98f3",
      "ogs:core:benchmark-task:sense-014": "c4a83a79265c0e9f",
      "ogs:core:benchmark-task:sense-015": "66d64ca2254c09ee",
      "ogs:core:benchmark-task:sense-016": "70318c6fd3b033d2",
      "ogs:core:benchmark-task:sense-017": "c3cedde89152ce76",
      "ogs:core:benchmark-task:sense-018": "7d31de11173a7fd2",
      "ogs:core:benchmark-task:sense-019": "2e7df899a0148ebd",
      "ogs:core:benchmark-task:sense-020": "ca5b9ab85fe71acb",
      "ogs:core:benchmark-task:struct-001": "4523d68b7da194f7",
      "ogs:core:benchmark-task:struct-002": "367b58556573440d",
      "ogs:core:benchmark-task:struct-003": "2740dc70d150e12a",
      "ogs:core:benchmark-task:struct-004": "9bd6860a8c5c69ec",
      "ogs:core:benchmark-task:struct-005": "9caf9bc474b2464d",
      "ogs:core:benchmark-task:struct-006": "bb4d0fcee2d1e751",
      "ogs:core:benchmark-task:struct-007": "0bc2e49ee6c1093e",
      "ogs:core:benchmark-task:struct-008": "791ac1beebd95233",
      "ogs:core:benchmark-task:struct-009": "a6a51cef64e43626",
      "ogs:core:benchmark-task:struct-010": "6d4d6d3cc23f3283",
      "ogs:core:benchmark-task:struct-011": "6615664480900680",
      "ogs:core:benchmark-task:struct-012": "bd330891a3e68b59",
      "ogs:core:benchmark-task:struct-013": "872d23f39307a374",
      "ogs:core:benchmark-task:struct-014": "3fe3fe74f8d80260",
      "ogs:core:benchmark-task:struct-015": "f0c01fa20f35eb51",
      "ogs:core:benchmark-task:struct-016": "284b25ae8f0dc3ef",
      "ogs:core:benchmark-task:struct-017": "534d8590759d6dc3",
      "ogs:core:benchmark-task:struct-018": "1470a28b8e9ef028",
      "ogs:core:benchmark-task:struct-019": "aa2726470aa7cfb8",
      "ogs:core:benchmark-task:struct-020": "5b115600e8dad71f",
      "ogs:core:benchmark-task:vocab-001": "e52270d817a9dc7f",
      "ogs:core:benchmark-task:vocab-002": "e2096242953ff882",
      "ogs:core:benchmark-task:vocab-003": "fa5fa32bfe275dfb",
      "ogs:core:benchmark-task:vocab-004": "e0631034b8b8ece4",
      "ogs:core:benchmark-task:vocab-005": "bbbd81bb3c8b8ea4",
      "ogs:core:benchmark-task:vocab-006": "6c6ae6f82f354be3",
      "ogs:core:benchmark-task:vocab-007": "0adac8c06022acf5",
      "ogs:core:benchmark-task:vocab-008": "e58319dc1276cdb2",
      "ogs:core:benchmark-task:vocab-009": "d51864ab27ecfa64",
      "ogs:core:benchmark-task:vocab-010": "1748069ad0ebdde7",
      "ogs:core:benchmark-task:vocab-011": "85de6add5c1ebfba",
      "ogs:core:benchmark-task:vocab-012": "44c9b26dc70c4557",
      "ogs:core:benchmark-task:vocab-013": "dc3e62daf8bdf21a",
      "ogs:core:benchmark-task:vocab-014": "3980456a47e9b7c6",
      "ogs:core:benchmark-task:vocab-015": "ad313a98974d64f7",
      "ogs:core:benchmark-task:vocab-016": "95e41ab82cba029a",
      "ogs:core:benchmark-task:vocab-017": "8070fb687fd2f108",
      "ogs:core:benchmark-task:vocab-018": "9e97b59361dbdf0e",
      "ogs:core:benchmark-task:vocab-019": "fc9a763682f1225b",
      "ogs:core:benchmark-task:vocab-020": "56ad4f9c58bc1eb7",
      "ogs:core:benchmark-task:xplain-001": "a75ad0ee2a385794",
      "ogs:core:benchmark-task:xplain-002": "bdc5d8d74c85e9ab",
      "ogs:core:benchmark-task:xplain-003": "3440bcf3978a1c58",
      "ogs:core:benchmark-task:xplain-004": "b3249dafda966f70",
      "ogs:core:benchmark-task:xplain-005": "0cb2f1946943a0e2",
      "ogs:core:benchmark-task:xplain-006": "9a46b0d141442ce3",
      "ogs:core:benchmark-task:xplain-007": "497adc64f3303eb4",
      "ogs:core:benchmark-task:xplain-008": "2120f96a0444a50e",
      "ogs:core:benchmark-task:xplain-009": "c077c498ceed5f73",
      "ogs:core:benchmark-task:xplain-010": "956c6ce654c2e297",
      "ogs:core:benchmark-task:xplain-011": "1dfeb782059a4814",
      "ogs:core:benchmark-task:xplain-012": "28df36c8ae155697",
      "ogs:core:benchmark-task:xplain-013": "18a20e7e30844c41",
      "ogs:core:benchmark-task:xplain-014": "f9be4ef2853d67b3",
      "ogs:core:benchmark-task:xplain-015": "5474f731e5d9751c",
      "ogs:core:benchmark-task:xplain-016": "fc7f227aebed6756",
      "ogs:core:benchmark-task:xplain-017": "024198497e327bb1",
      "ogs:core:benchmark-task:xplain-018": "283bbb9e703e6e30",
      "ogs:core:benchmark-task:xplain-019": "9ea6cd79384f9ee2",
      "ogs:core:benchmark-task:xplain-020": "43005f44c870684b"
    }
  },
  "tasks": [
    {
      "task_id": "ogs:core:benchmark-task:comp-001",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "5dc05e883ee75cf7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0 = no browning, 5 = fully charred), return the integer value best describing the described cook. Respond with just the integer.\n\nINPUT:\n{\n  \"description\": \"a duck breast seared skin-side down until the skin is deep mahogany brown and crisp, but not black\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 3,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "3",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 470,
            "output_tokens": 5,
            "latency_ms": 834.5,
            "usd": 0.000396,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "3",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 470,
            "output_tokens": 5,
            "latency_ms": 480.7,
            "usd": 0.000396,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "3",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 470,
            "output_tokens": 5,
            "latency_ms": 514.1,
            "usd": 0.000396,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 470.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 609.767,
        "usd_sum": 0.001188
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-002",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "d5fef8891eee5391",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0 = no browning, 5 = fully charred), return the integer value for the described cook. Respond with just the integer.\n\nINPUT:\n{\n  \"description\": \"a fresh egg poached in 82 C water for 4 minutes\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 501.9,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 709.8,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 520.5,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 460.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 577.4,
        "usd_sum": 0.001164
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-003",
      "category": "composition",
      "capability": "reduction_level",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.reduction_level"
      ],
      "prompt_template_hash": "7b7db2c313e1d55b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Reduction level (0-5 integer)\n\n- 0: no reduction (raw stock, unreduced liquid)\n- 1: light simmer reduction (~10%)\n- 2: modest reduction (~30%, nappe forming)\n- 3: substantial reduction (~50%, demi-glace territory)\n- 4: heavy reduction (~75%, glace-like, syrupy)\n- 5: near-complete reduction (glace de viande, paste)\n\n---\n\nUsing the OGS reduction level (0 = no reduction, 5 = glace-level), return the integer best describing the sauce. Respond with just the integer.\n\nINPUT:\n{\n  \"description\": \"a veal demi-glace reduced from 4 L of stock down to about 600 mL, thickly coating a spoon with a glossy sheen\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 4,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "3",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 3.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 588.6,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "3",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 3.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 531.2,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "3",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 3.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 1027.5,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 460.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 715.767,
        "usd_sum": 0.001164
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-004",
      "category": "composition",
      "capability": "cooking_method_recognition",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "ca8109a73af762fa",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nA piece of black cod is packed in salt-sugar-sake-lees paste and kept refrigerated for 48 hours, then rinsed and grilled over binchotan. What is the single OGS cooking method that best labels the BEFORE-grilling stage? Respond with just the method name.\n\nINPUT:\n{\n  \"description\": \"black cod packed in salt-sugar-sake-lees paste for 48h before grilling\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "cured"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 496,
            "output_tokens": 5,
            "latency_ms": 513.3,
            "usd": 0.000417,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 496,
            "output_tokens": 5,
            "latency_ms": 764.1,
            "usd": 0.000417,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 496,
            "output_tokens": 5,
            "latency_ms": 571.7,
            "usd": 0.000417,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 496.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 616.367,
        "usd_sum": 0.001251
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-005",
      "category": "composition",
      "capability": "component_role_assignment",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "2885728399552fb5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nFor the described dish element, return the single OGS component role that best applies. Respond with just the role name.\n\nINPUT:\n{\n  \"element\": \"a small handful of thin, deep-fried shallot rings scattered on top of a braised beef rendang primarily for crunchy contrast\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "textural_element"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "textural_element",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "textural_element",
            "expected": "textural_element",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 407,
            "output_tokens": 7,
            "latency_ms": 544.1,
            "usd": 0.000354,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "textural_element",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "textural_element",
            "expected": "textural_element",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 407,
            "output_tokens": 7,
            "latency_ms": 602.6,
            "usd": 0.000354,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "textural_element",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "textural_element",
            "expected": "textural_element",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 407,
            "output_tokens": 7,
            "latency_ms": 702.8,
            "usd": 0.000354,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 407.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 616.5,
        "usd_sum": 0.001062
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-006",
      "category": "composition",
      "capability": "component_set_identification",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "bb37bf992b2f6c39",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nList the OGS component ROLES that appear in a classic Roman spaghetti alla carbonara (guanciale, eggs, pecorino romano, spaghetti, black pepper). Respond with a JSON array of role strings from the OGS vocabulary, without duplicates, in any order.\n\nINPUT:\n{\n  \"dish\": \"classic Roman spaghetti alla carbonara\",\n  \"components_present\": [\n    \"spaghetti\",\n    \"guanciale\",\n    \"egg yolks\",\n    \"pecorino romano\",\n    \"black pepper\"\n  ]\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"array\",\n  \"items\": {\n    \"type\": \"string\",\n    \"enum\": [\n      \"main_protein\",\n      \"main_vegetable\",\n      \"starch\",\n      \"sauce\",\n      \"fat_component\",\n      \"acid_component\",\n      \"aromatic_garnish\",\n      \"seasoning\",\n      \"textural_element\"\n    ]\n  },\n  \"uniqueItems\": true\n}",
      "expected": {
        "values": [
          "starch",
          "main_protein",
          "fat_component",
          "sauce",
          "seasoning"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"starch\", \"main_protein\", \"fat_component\", \"sauce\", \"seasoning\"]\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "expected": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 551,
            "output_tokens": 29,
            "latency_ms": 911.6,
            "usd": 0.000557,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"starch\", \"main_protein\", \"fat_component\", \"sauce\", \"seasoning\"]\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "expected": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 551,
            "output_tokens": 29,
            "latency_ms": 820.8,
            "usd": 0.000557,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\n  \"starch\",\n  \"main_protein\",\n  \"fat_component\",\n  \"sauce\",\n  \"seasoning\"\n]\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "expected": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 551,
            "output_tokens": 42,
            "latency_ms": 813.6,
            "usd": 0.000609,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 551.0,
        "output_tokens_mean": 33.333,
        "latency_ms_mean": 848.667,
        "usd_sum": 0.001723
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-007",
      "category": "composition",
      "capability": "service_temperature",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.service_temperatures"
      ],
      "prompt_template_hash": "59ae41132f38bbfd",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n---\n\nReturn the single OGS service_temperature value that matches the traditional service of the dish below. Respond with just the value.\n\nINPUT:\n{\n  \"dish\": \"classic French vichyssoise (leek and potato soup)\"\n}\n\nANSWER FORMAT: one_of:frozen,cold,cool,room,warm,hot,very_hot",
      "expected": {
        "value": "cold"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 350,
            "output_tokens": 4,
            "latency_ms": 505.3,
            "usd": 0.000296,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 350,
            "output_tokens": 4,
            "latency_ms": 709.0,
            "usd": 0.000296,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 350,
            "output_tokens": 4,
            "latency_ms": 520.0,
            "usd": 0.000296,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 350.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 578.1,
        "usd_sum": 0.000888
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-008",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "5121ce4bf46aa4d5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0 = no browning, 5 = fully charred), return the integer for a steamed dim sum dumpling wrapper.\n\nINPUT:\n{\n  \"description\": \"translucent wheat-starch wrapper on a har gow, bamboo-steamed\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 555.1,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 1257.9,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 460,
            "output_tokens": 5,
            "latency_ms": 1221.9,
            "usd": 0.000388,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 460.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1011.633,
        "usd_sum": 0.001164
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-009",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "2a8a86bb58372ed7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0-5 integer), return the value for a chicken breast poached in stock at 82°C with no prior searing.\n\nINPUT:\n{\n  \"description\": \"chicken breast poached in 82°C stock, never seared\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 0,
        "tolerance": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 455,
            "output_tokens": 5,
            "latency_ms": 1427.3,
            "usd": 0.000384,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 455,
            "output_tokens": 5,
            "latency_ms": 809.3,
            "usd": 0.000384,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 455,
            "output_tokens": 5,
            "latency_ms": 607.3,
            "usd": 0.000384,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 455.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 947.967,
        "usd_sum": 0.001152
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-010",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "d420afcaea603a0e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index, return the integer for a scallop seared in hot butter until a deep golden crust forms on one side.\n\nINPUT:\n{\n  \"description\": \"diver scallop seared in foaming butter, one side deep golden, interior translucent\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 3,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 458,
            "output_tokens": 5,
            "latency_ms": 1016.1,
            "usd": 0.000386,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 458,
            "output_tokens": 5,
            "latency_ms": 605.2,
            "usd": 0.000386,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 458,
            "output_tokens": 5,
            "latency_ms": 641.6,
            "usd": 0.000386,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 458.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 754.3,
        "usd_sum": 0.001158
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-011",
      "category": "composition",
      "capability": "reduction_level",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.reduction_level"
      ],
      "prompt_template_hash": "1d18b06eac9f3ddc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Reduction level (0-5 integer)\n\n- 0: no reduction (raw stock, unreduced liquid)\n- 1: light simmer reduction (~10%)\n- 2: modest reduction (~30%, nappe forming)\n- 3: substantial reduction (~50%, demi-glace territory)\n- 4: heavy reduction (~75%, glace-like, syrupy)\n- 5: near-complete reduction (glace de viande, paste)\n\n---\n\nUsing the OGS reduction level (0-5 integer), return the value for a pan sauce reduced until it coats the back of a spoon (nappe).\n\nINPUT:\n{\n  \"description\": \"red wine pan sauce reduced until it coats the back of a spoon without running\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 3,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 437,
            "output_tokens": 5,
            "latency_ms": 569.6,
            "usd": 0.00037,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 437,
            "output_tokens": 5,
            "latency_ms": 1024.3,
            "usd": 0.00037,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 437,
            "output_tokens": 5,
            "latency_ms": 777.5,
            "usd": 0.00037,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 437.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 790.467,
        "usd_sum": 0.00111
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-012",
      "category": "composition",
      "capability": "reduction_level",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.reduction_level"
      ],
      "prompt_template_hash": "fa91f74e86a4cffb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Reduction level (0-5 integer)\n\n- 0: no reduction (raw stock, unreduced liquid)\n- 1: light simmer reduction (~10%)\n- 2: modest reduction (~30%, nappe forming)\n- 3: substantial reduction (~50%, demi-glace territory)\n- 4: heavy reduction (~75%, glace-like, syrupy)\n- 5: near-complete reduction (glace de viande, paste)\n\n---\n\nUsing the OGS reduction level, return the integer for glace de viande (gelatinous concentrate that sets firmly at room temperature).\n\nINPUT:\n{\n  \"description\": \"glace de viande, reduced from 10 L brown stock to ~500 mL\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 5,
        "tolerance": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 5.0,
            "expected": 5.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 434,
            "output_tokens": 5,
            "latency_ms": 1139.5,
            "usd": 0.000367,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 5.0,
            "expected": 5.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 434,
            "output_tokens": 5,
            "latency_ms": 1028.6,
            "usd": 0.000367,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 5.0,
            "expected": 5.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 434,
            "output_tokens": 5,
            "latency_ms": 614.2,
            "usd": 0.000367,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 434.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 927.433,
        "usd_sum": 0.001101
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-013",
      "category": "composition",
      "capability": "cooking_method_recognition",
      "difficulty": "medium",
      "cuisine_region": "korean",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "f8257062be99b5d3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nReturn the single OGS cooking method that best fits the main cooking step for bossam pork belly (~90 min at low simmer in aromatic water).\n\nINPUT:\n{\n  \"description\": \"pork belly gently simmered for 90 minutes in water with ginger, doenjang, and pepper until fork-tender\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "poached"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "poached",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "poached",
            "expected": "poached",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 472,
            "output_tokens": 5,
            "latency_ms": 656.2,
            "usd": 0.000398,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "poached",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "poached",
            "expected": "poached",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 472,
            "output_tokens": 5,
            "latency_ms": 564.6,
            "usd": 0.000398,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "poached",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "poached",
            "expected": "poached",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 472,
            "output_tokens": 5,
            "latency_ms": 521.4,
            "usd": 0.000398,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 472.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 580.733,
        "usd_sum": 0.001194
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-014",
      "category": "composition",
      "capability": "component_role_assignment",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "114b0796c10cf141",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits a sprinkle of gremolata (parsley, garlic, lemon zest) finishing an osso buco at service.\n\nINPUT:\n{\n  \"description\": \"gremolata finishing an osso buco milanese at service\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "aromatic_garnish"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "aromatic_garnish",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "aromatic_garnish",
            "expected": "aromatic_garnish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 8,
            "latency_ms": 688.4,
            "usd": 0.000357,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "aromatic_garnish",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "aromatic_garnish",
            "expected": "aromatic_garnish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 8,
            "latency_ms": 651.6,
            "usd": 0.000357,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "aromatic_garnish",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "aromatic_garnish",
            "expected": "aromatic_garnish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 8,
            "latency_ms": 1483.7,
            "usd": 0.000357,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 406.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 941.233,
        "usd_sum": 0.001071
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-015",
      "category": "composition",
      "capability": "component_set_identification",
      "difficulty": "hard",
      "cuisine_region": "thai",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "07976f0ee97244e5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nList the OGS component roles that appear in a classic Pad Thai (rice noodles, shrimp, tofu, egg, tamarind-palm sugar-fish sauce, bean sprouts, crushed peanuts, chives). Return a JSON array of role names.\n\nINPUT:\n{\n  \"dish\": \"classic Pad Thai\"\n}\n\nANSWER FORMAT: set_of_roles",
      "expected": {
        "values": [
          "starch",
          "main_protein",
          "acid_component",
          "seasoning",
          "aromatic_garnish",
          "textural_element",
          "fat_component"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"main_protein\", \"starch\", \"sauce\", \"fat_component\", \"acid_component\", \"aromatic_garnish\", \"seasoning\", \"textural_element\"]\n```",
          "compliant": true,
          "score": 0.9333,
          "detail": {
            "predicted": [
              "acid_component",
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "expected": [
              "acid_component",
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "precision": 0.875,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 382,
            "output_tokens": 47,
            "latency_ms": 1255.6,
            "usd": 0.000494,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"main_protein\", \"main_protein\", \"starch\", \"sauce\", \"textural_element\", \"seasoning\", \"aromatic_garnish\", \"fat_component\"]\n```",
          "compliant": true,
          "score": 0.8571,
          "detail": {
            "predicted": [
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "expected": [
              "acid_component",
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "precision": 0.8571,
            "recall": 0.8571
          },
          "metrics": {
            "input_tokens": 382,
            "output_tokens": 47,
            "latency_ms": 909.6,
            "usd": 0.000494,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"main_protein\", \"main_protein\", \"starch\", \"sauce\", \"acid_component\", \"aromatic_garnish\", \"textural_element\", \"aromatic_garnish\"]\n```\n\nWait, let me reconsider with unique roles:\n\n```json\n[\"main_protein\", \"starch\", \"sauce\", \"acid_component\", \"aromatic_garnish\", \"textural_element\", \"seasoning\"]\n```",
          "compliant": true,
          "score": 0.7692,
          "detail": {
            "predicted": [
              "acid_component",
              "aromatic_garnish",
              "main_protein",
              "sauce",
              "starch",
              "textural_element"
            ],
            "expected": [
              "acid_component",
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "precision": 0.8333,
            "recall": 0.7143
          },
          "metrics": {
            "input_tokens": 382,
            "output_tokens": 102,
            "latency_ms": 1018.1,
            "usd": 0.000714,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.8532,
      "score_summary": {
        "mean": 0.8532,
        "stdev": 0.0821,
        "sem": 0.0474,
        "ci95_lo": 0.7603,
        "ci95_hi": 0.9461,
        "n": 3
      },
      "score_if_valid": 0.8532,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 382.0,
        "output_tokens_mean": 65.333,
        "latency_ms_mean": 1061.1,
        "usd_sum": 0.001702
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-016",
      "category": "composition",
      "capability": "service_temperature",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.service_temperatures"
      ],
      "prompt_template_hash": "609a0643399a8df0",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n---\n\nReturn the single OGS service_temperature value for a Spanish ajo blanco (chilled almond-garlic soup).\n\nINPUT:\n{\n  \"dish\": \"ajo blanco, chilled Andalusian almond and garlic soup\"\n}\n\nANSWER FORMAT: one_of:frozen,cold,cool,room,warm,hot,very_hot",
      "expected": {
        "value": "cold"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 355,
            "output_tokens": 4,
            "latency_ms": 1982.7,
            "usd": 0.0003,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 355,
            "output_tokens": 4,
            "latency_ms": 910.9,
            "usd": 0.0003,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 355,
            "output_tokens": 4,
            "latency_ms": 708.7,
            "usd": 0.0003,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 355.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1200.767,
        "usd_sum": 0.0009
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-017",
      "category": "composition",
      "capability": "cooking_method_recognition",
      "difficulty": "medium",
      "cuisine_region": "chinese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "09a6e47eda264ded",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nReturn the single OGS cooking method that best describes the signature crisp skin of Peking duck (air-dried then hung-roasted in a wood-fired oven).\n\nINPUT:\n{\n  \"description\": \"Peking duck hung-roasted in a jujube-wood oven at ~230°C until the skin is lacquered and crisp\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "roasted"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 478,
            "output_tokens": 5,
            "latency_ms": 493.8,
            "usd": 0.000402,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 478,
            "output_tokens": 5,
            "latency_ms": 661.0,
            "usd": 0.000402,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 478,
            "output_tokens": 5,
            "latency_ms": 1313.1,
            "usd": 0.000402,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 478.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 822.633,
        "usd_sum": 0.001206
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-018",
      "category": "composition",
      "capability": "component_role_assignment",
      "difficulty": "medium",
      "cuisine_region": "levantine",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "32d4a921f4fbdd98",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the OGS component role that best fits tahini as an ingredient in a classic hummus recipe.\n\nINPUT:\n{\n  \"description\": \"25% tahini by weight blended into chickpea puree to make hummus\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "fat_component"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fat_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fat_component",
            "expected": "fat_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 391,
            "output_tokens": 6,
            "latency_ms": 580.0,
            "usd": 0.000337,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "fat_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fat_component",
            "expected": "fat_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 391,
            "output_tokens": 6,
            "latency_ms": 1348.7,
            "usd": 0.000337,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "fat_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fat_component",
            "expected": "fat_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 391,
            "output_tokens": 6,
            "latency_ms": 761.8,
            "usd": 0.000337,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 391.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 896.833,
        "usd_sum": 0.001011
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-019",
      "category": "composition",
      "capability": "service_temperature",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.service_temperatures"
      ],
      "prompt_template_hash": "eaa27377a264e2e2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n---\n\nReturn the OGS service_temperature for a stirred-over-ice Negroni served in a rocks glass with a large cube.\n\nINPUT:\n{\n  \"dish\": \"stirred Negroni served over a single large ice cube\"\n}\n\nANSWER FORMAT: one_of:frozen,cold,cool,room,warm,hot,very_hot",
      "expected": {
        "value": "cold"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 4,
            "latency_ms": 564.4,
            "usd": 0.000299,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 4,
            "latency_ms": 823.8,
            "usd": 0.000299,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 4,
            "latency_ms": 510.3,
            "usd": 0.000299,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 354.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 632.833,
        "usd_sum": 0.000897
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-020",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "hard",
      "cuisine_region": "american",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "535bf59391539a15",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index, return the integer for blackened redfish cooked in a smoking-hot cast iron skillet with Cajun spice rub.\n\nINPUT:\n{\n  \"description\": \"redfish fillet crusted with spice and cooked in a 300°C dry cast-iron pan until the surface is blackened and slightly bitter\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 4,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 473,
            "output_tokens": 5,
            "latency_ms": 475.8,
            "usd": 0.000398,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 473,
            "output_tokens": 5,
            "latency_ms": 632.0,
            "usd": 0.000398,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 473,
            "output_tokens": 5,
            "latency_ms": 667.5,
            "usd": 0.000398,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 473.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 591.767,
        "usd_sum": 0.001194
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-001",
      "category": "dietary",
      "capability": "vegan_compatibility",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "18b5c9c7c958a521",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nIs a traditional Bolognese tagliatelle al ragù (tagliatelle with ragù made from beef, pancetta, soffritto, tomato, red wine, whole milk, and butter) compatible with a strict vegan diet? Answer with a single word: yes or no.\n\nINPUT:\n{\n  \"dish\": \"tagliatelle al ragù\",\n  \"diet\": \"vegan\"\n}\n\nANSWER FORMAT: exact:yes|no",
      "expected": {
        "value": "no"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 1055,
            "output_tokens": 4,
            "latency_ms": 498.0,
            "usd": 0.00086,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 1055,
            "output_tokens": 4,
            "latency_ms": 725.3,
            "usd": 0.00086,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 1055,
            "output_tokens": 4,
            "latency_ms": 477.1,
            "usd": 0.00086,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1055.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 566.8,
        "usd_sum": 0.00258
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-002",
      "category": "dietary",
      "capability": "gluten_free_violation",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "04840e46f80ccb5b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nA kitchen claims this stir-fry is gluten-free: rice noodles, tofu, shiitake, broccoli, garlic, ginger, regular Japanese soy sauce, sesame oil, and scallions. Return the single ingredient that breaks the claim.\n\nINPUT:\n{\n  \"dish\": \"stir fry\",\n  \"ingredients\": [\n    \"rice noodles\",\n    \"tofu\",\n    \"shiitake\",\n    \"broccoli\",\n    \"garlic\",\n    \"ginger\",\n    \"regular Japanese soy sauce\",\n    \"sesame oil\",\n    \"scallions\"\n  ]\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "regular Japanese soy sauce",
        "aliases": [
          "soy sauce",
          "regular soy sauce"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "regular Japanese soy sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "regular japanese soy sauce",
            "expected": "regular japanese soy sauce",
            "aliases": [
              "soy sauce",
              "regular soy sauce"
            ]
          },
          "metrics": {
            "input_tokens": 1111,
            "output_tokens": 8,
            "latency_ms": 865.4,
            "usd": 0.000921,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "regular Japanese soy sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "regular japanese soy sauce",
            "expected": "regular japanese soy sauce",
            "aliases": [
              "soy sauce",
              "regular soy sauce"
            ]
          },
          "metrics": {
            "input_tokens": 1111,
            "output_tokens": 8,
            "latency_ms": 738.4,
            "usd": 0.000921,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "regular Japanese soy sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "regular japanese soy sauce",
            "expected": "regular japanese soy sauce",
            "aliases": [
              "soy sauce",
              "regular soy sauce"
            ]
          },
          "metrics": {
            "input_tokens": 1111,
            "output_tokens": 8,
            "latency_ms": 544.3,
            "usd": 0.000921,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1111.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 716.033,
        "usd_sum": 0.002763
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-003",
      "category": "dietary",
      "capability": "halal_compliance",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "013d6295fee472b5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nIs a classic panna cotta (cream, sugar, vanilla, leaf gelatine) halal? Answer with yes or no.\n\nINPUT:\n{\n  \"dish\": \"panna cotta\",\n  \"ingredients\": [\n    \"cream\",\n    \"sugar\",\n    \"vanilla\",\n    \"leaf gelatine\"\n  ]\n}\n\nANSWER FORMAT: exact:yes|no",
      "expected": {
        "value": "no"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 1040,
            "output_tokens": 4,
            "latency_ms": 522.9,
            "usd": 0.000848,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 1040,
            "output_tokens": 4,
            "latency_ms": 861.2,
            "usd": 0.000848,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 1040,
            "output_tokens": 4,
            "latency_ms": 1299.8,
            "usd": 0.000848,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1040.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 894.633,
        "usd_sum": 0.002544
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-004",
      "category": "dietary",
      "capability": "kosher_meat_dairy",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "85fc94958c4357ef",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nChicken parmigiana is breaded chicken breast topped with tomato sauce and melted mozzarella, served with a side of cream-rich garlic mashed potatoes. From a kosher perspective, what is the core violation? Respond with one short phrase.\n\nINPUT:\n{\n  \"dish\": \"chicken parmigiana with creamy mash\"\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "mixing meat and dairy",
        "aliases": [
          "meat and dairy mixed",
          "basar b'chalav",
          "poultry with dairy",
          "meat with dairy"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "mixing of meat and dairy",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "mixing of meat and dairy",
            "expected": "mixing meat and dairy",
            "aliases": [
              "meat and dairy mixed",
              "basar b'chalav",
              "poultry with dairy",
              "meat with dairy"
            ]
          },
          "metrics": {
            "input_tokens": 1042,
            "output_tokens": 8,
            "latency_ms": 686.9,
            "usd": 0.000866,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "mixing of meat and dairy",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "mixing of meat and dairy",
            "expected": "mixing meat and dairy",
            "aliases": [
              "meat and dairy mixed",
              "basar b'chalav",
              "poultry with dairy",
              "meat with dairy"
            ]
          },
          "metrics": {
            "input_tokens": 1042,
            "output_tokens": 8,
            "latency_ms": 1150.8,
            "usd": 0.000866,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "mixing of meat and dairy",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "mixing of meat and dairy",
            "expected": "mixing meat and dairy",
            "aliases": [
              "meat and dairy mixed",
              "basar b'chalav",
              "poultry with dairy",
              "meat with dairy"
            ]
          },
          "metrics": {
            "input_tokens": 1042,
            "output_tokens": 8,
            "latency_ms": 2165.2,
            "usd": 0.000866,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1042.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 1334.3,
        "usd_sum": 0.002598
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-005",
      "category": "dietary",
      "capability": "allergen_detection",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "f40f4c25a2ee4b71",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nFor a dish of pesto genovese made from basil, garlic, pine nuts, parmesan, pecorino, and olive oil, list the EU Annex II allergens that must be declared. Return a comma-separated list using the OGS allergen IDs.\n\nINPUT:\n{\n  \"dish\": \"pesto genovese\",\n  \"ingredients\": [\n    \"basil\",\n    \"garlic\",\n    \"pine nuts\",\n    \"parmesan\",\n    \"pecorino\",\n    \"olive oil\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated OGS allergen IDs",
      "expected": {
        "values": [
          "tree_nuts",
          "milk"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "tree_nuts, milk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "milk",
              "tree_nuts"
            ],
            "expected": [
              "milk",
              "tree_nuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1097,
            "output_tokens": 8,
            "latency_ms": 717.9,
            "usd": 0.00091,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "tree_nuts, milk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "milk",
              "tree_nuts"
            ],
            "expected": [
              "milk",
              "tree_nuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1097,
            "output_tokens": 8,
            "latency_ms": 944.2,
            "usd": 0.00091,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "tree_nuts, milk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "milk",
              "tree_nuts"
            ],
            "expected": [
              "milk",
              "tree_nuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1097,
            "output_tokens": 8,
            "latency_ms": 586.6,
            "usd": 0.00091,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1097.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 749.567,
        "usd_sum": 0.00273
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-006",
      "category": "dietary",
      "capability": "allergen_detection",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "7f45e388301a68a5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nReturn the EU Annex II allergens that must be declared for a standard pad thai (rice noodles, shrimp, egg, tamarind, fish sauce, palm sugar, roasted peanuts, bean sprouts, garlic chives, lime, chili).\n\nINPUT:\n{\n  \"dish\": \"pad thai\"\n}\n\nANSWER FORMAT: set:comma-separated OGS allergen IDs",
      "expected": {
        "values": [
          "crustaceans",
          "eggs",
          "fish",
          "peanuts"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "crustaceans,eggs,fish,peanuts,soybeans",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts",
              "soybeans"
            ],
            "expected": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1045,
            "output_tokens": 19,
            "latency_ms": 601.6,
            "usd": 0.000912,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "fish, peanuts, crustaceans, eggs, soybeans",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts",
              "soybeans"
            ],
            "expected": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1045,
            "output_tokens": 19,
            "latency_ms": 717.7,
            "usd": 0.000912,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "fish, crustaceans, eggs, peanuts, soybeans",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts",
              "soybeans"
            ],
            "expected": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1045,
            "output_tokens": 19,
            "latency_ms": 794.3,
            "usd": 0.000912,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.8889,
      "score_summary": {
        "mean": 0.8889,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8889,
        "ci95_hi": 0.8889,
        "n": 3
      },
      "score_if_valid": 0.8889,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1045.0,
        "output_tokens_mean": 19.0,
        "latency_ms_mean": 704.533,
        "usd_sum": 0.002736
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-007",
      "category": "dietary",
      "capability": "vegan_substitution",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "eb092a2c162960f6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nName a single plant-based ingredient commonly used to emulsify and thicken a vegan carbonara sauce in place of eggs, while also delivering umami. Respond with one short ingredient name.\n\nINPUT:\n{\n  \"target_role\": \"egg emulsifier and umami in vegan carbonara\"\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "silken tofu",
        "aliases": [
          "silken tofu blended with nutritional yeast",
          "soft tofu"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "nutritional_yeast",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "nutritional_yeast",
            "expected": "silken tofu",
            "aliases": [
              "silken tofu blended with nutritional yeast",
              "soft tofu"
            ]
          },
          "metrics": {
            "input_tokens": 1030,
            "output_tokens": 8,
            "latency_ms": 682.2,
            "usd": 0.000856,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "nutritional_yeast",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "nutritional_yeast",
            "expected": "silken tofu",
            "aliases": [
              "silken tofu blended with nutritional yeast",
              "soft tofu"
            ]
          },
          "metrics": {
            "input_tokens": 1030,
            "output_tokens": 8,
            "latency_ms": 646.5,
            "usd": 0.000856,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "nutritional yeast",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "nutritional yeast",
            "expected": "silken tofu",
            "aliases": [
              "silken tofu blended with nutritional yeast",
              "soft tofu"
            ]
          },
          "metrics": {
            "input_tokens": 1030,
            "output_tokens": 7,
            "latency_ms": 677.2,
            "usd": 0.000852,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1030.0,
        "output_tokens_mean": 7.667,
        "latency_ms_mean": 668.633,
        "usd_sum": 0.002564
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-008",
      "category": "dietary",
      "capability": "gluten_free_substitution",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "480be99b3f35a51d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nSuggest a single widely-used gluten-free flour to make a béchamel roux that maintains classic texture. Respond with one word.\n\nINPUT:\n{\n  \"target_role\": \"béchamel roux, gluten-free\"\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "rice flour",
        "aliases": [
          "cornstarch",
          "corn starch",
          "cornflour",
          "tapioca starch",
          "tapioca flour"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cornstarch",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cornstarch",
            "expected": "rice flour",
            "aliases": [
              "cornstarch",
              "corn starch",
              "cornflour",
              "tapioca starch",
              "tapioca flour"
            ]
          },
          "metrics": {
            "input_tokens": 1019,
            "output_tokens": 6,
            "latency_ms": 541.6,
            "usd": 0.000839,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "cornstarch",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cornstarch",
            "expected": "rice flour",
            "aliases": [
              "cornstarch",
              "corn starch",
              "cornflour",
              "tapioca starch",
              "tapioca flour"
            ]
          },
          "metrics": {
            "input_tokens": 1019,
            "output_tokens": 6,
            "latency_ms": 861.8,
            "usd": 0.000839,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "cornstarch",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cornstarch",
            "expected": "rice flour",
            "aliases": [
              "cornstarch",
              "corn starch",
              "cornflour",
              "tapioca starch",
              "tapioca flour"
            ]
          },
          "metrics": {
            "input_tokens": 1019,
            "output_tokens": 6,
            "latency_ms": 916.8,
            "usd": 0.000839,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1019.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 773.4,
        "usd_sum": 0.002517
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-009",
      "category": "dietary",
      "capability": "vegan_compatibility",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "936752891554dba3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nFrom this list, return every item that is NOT vegan as a comma-separated list: honey, agave nectar, wine clarified with isinglass, wine clarified with bentonite, carmine, nutritional yeast, shellac, maple syrup, worcestershire sauce (traditional), vegetable glycerin.\n\nINPUT:\n{\n  \"items\": [\n    \"honey\",\n    \"agave nectar\",\n    \"wine clarified with isinglass\",\n    \"wine clarified with bentonite\",\n    \"carmine\",\n    \"nutritional yeast\",\n    \"shellac\",\n    \"maple syrup\",\n    \"worcestershire sauce (traditional)\",\n    \"vegetable glycerin\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated item names",
      "expected": {
        "values": [
          "honey",
          "wine clarified with isinglass",
          "carmine",
          "shellac",
          "worcestershire sauce (traditional)"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "honey, wine clarified with isinglass, carmine, shellac, worcestershire sauce (traditional)",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "expected": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1127,
            "output_tokens": 25,
            "latency_ms": 891.1,
            "usd": 0.001002,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "honey, wine clarified with isinglass, carmine, shellac, worcestershire sauce (traditional)",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "expected": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1127,
            "output_tokens": 25,
            "latency_ms": 1151.7,
            "usd": 0.001002,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "honey, wine clarified with isinglass, carmine, shellac, worcestershire sauce (traditional)",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "expected": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1127,
            "output_tokens": 25,
            "latency_ms": 640.1,
            "usd": 0.001002,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1127.0,
        "output_tokens_mean": 25.0,
        "latency_ms_mean": 894.3,
        "usd_sum": 0.003006
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-010",
      "category": "dietary",
      "capability": "cross_contact_risk",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "35c29846f76046d7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nA kitchen fries breaded calamari, battered cod, and french fries in a single deep fryer. Return the EU allergens that a customer ordering 'fries only' must be warned about as a comma-separated OGS allergen ID list.\n\nINPUT:\n{\n  \"scenario\": \"single shared fryer\",\n  \"items\": [\n    \"breaded calamari\",\n    \"battered cod\",\n    \"french fries\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated OGS allergen IDs",
      "expected": {
        "values": [
          "molluscs",
          "fish",
          "cereals_containing_gluten"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "crustaceans,fish,eggs,cereals_containing_gluten,molluscs",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "cereals_containing_gluten",
              "crustaceans",
              "eggs",
              "fish",
              "molluscs"
            ],
            "expected": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "precision": 0.6,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1076,
            "output_tokens": 23,
            "latency_ms": 883.3,
            "usd": 0.000953,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "fish,molluscs,crustaceans,cereals_containing_gluten,eggs",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "cereals_containing_gluten",
              "crustaceans",
              "eggs",
              "fish",
              "molluscs"
            ],
            "expected": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "precision": 0.6,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1076,
            "output_tokens": 23,
            "latency_ms": 810.0,
            "usd": 0.000953,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "crustaceans, fish, eggs, tree_nuts, sesame",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "crustaceans",
              "eggs",
              "fish",
              "sesame",
              "tree_nuts"
            ],
            "expected": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "precision": 0.2,
            "recall": 0.3333
          },
          "metrics": {
            "input_tokens": 1076,
            "output_tokens": 18,
            "latency_ms": 1020.3,
            "usd": 0.000933,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5833,
      "score_summary": {
        "mean": 0.5833,
        "stdev": 0.2887,
        "sem": 0.1667,
        "ci95_lo": 0.2567,
        "ci95_hi": 0.91,
        "n": 3
      },
      "score_if_valid": 0.5833,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1076.0,
        "output_tokens_mean": 21.333,
        "latency_ms_mean": 904.533,
        "usd_sum": 0.002839
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-011",
      "category": "dietary",
      "capability": "low_fodmap",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "bdd7970997c11172",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nReturn the low-FODMAP violations in this dish as a comma-separated list of ingredient names: garlic-infused olive oil, white onion, lactose-free milk, wheat pasta, lean chicken, fresh chives, honey, aged parmesan.\n\nINPUT:\n{\n  \"ingredients\": [\n    \"garlic-infused olive oil\",\n    \"white onion\",\n    \"lactose-free milk\",\n    \"wheat pasta\",\n    \"lean chicken\",\n    \"fresh chives\",\n    \"honey\",\n    \"aged parmesan\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated ingredient names",
      "expected": {
        "values": [
          "white onion",
          "wheat pasta",
          "honey"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "garlic-infused olive oil, white onion, wheat pasta, honey",
          "compliant": true,
          "score": 0.8571,
          "detail": {
            "predicted": [
              "garlic-infused olive oil",
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "expected": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "precision": 0.75,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1095,
            "output_tokens": 19,
            "latency_ms": 731.0,
            "usd": 0.000952,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "garlic-infused olive oil, white onion, wheat pasta, honey",
          "compliant": true,
          "score": 0.8571,
          "detail": {
            "predicted": [
              "garlic-infused olive oil",
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "expected": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "precision": 0.75,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1095,
            "output_tokens": 19,
            "latency_ms": 616.1,
            "usd": 0.000952,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "garlic-infused olive oil, white onion, wheat pasta, honey",
          "compliant": true,
          "score": 0.8571,
          "detail": {
            "predicted": [
              "garlic-infused olive oil",
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "expected": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "precision": 0.75,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1095,
            "output_tokens": 19,
            "latency_ms": 887.9,
            "usd": 0.000952,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.8571,
      "score_summary": {
        "mean": 0.8571,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8571,
        "ci95_hi": 0.8571,
        "n": 3
      },
      "score_if_valid": 0.8571,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1095.0,
        "output_tokens_mean": 19.0,
        "latency_ms_mean": 745.0,
        "usd_sum": 0.002856
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-012",
      "category": "dietary",
      "capability": "pescatarian_vs_vegetarian",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "42c897d741064981",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nA dish contains miso, dashi (made with katsuobushi), tofu, and wakame. Mark which labels it qualifies for out of: vegan, vegetarian, pescatarian. Return a comma-separated list.\n\nINPUT:\n{\n  \"dish\": \"classic awase-dashi miso soup with tofu and wakame\"\n}\n\nANSWER FORMAT: set:comma-separated labels",
      "expected": {
        "values": [
          "pescatarian"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "vegetarian, pescatarian",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "pescatarian",
              "vegetarian"
            ],
            "expected": [
              "pescatarian"
            ],
            "precision": 0.5,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1042,
            "output_tokens": 9,
            "latency_ms": 827.5,
            "usd": 0.00087,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "I need to analyze each ingredient against the three dietary labels:\n\n**Ingredients:**\n- Miso: typically made with soybeans and salt; most are vegetarian but some may contain dashi (fish-based)\n- Dashi (katsuobushi): made from bonito fish (dried, fermented fish flakes) - **animal-derived**\n- Tofu: soy-based, plant-derived\n- Wakame: seaweed, plant-derived\n\n**Label Analysis:**\n\n**Vegan:** Requires no animal-derived ingredients. Dashi (katsuobushi) is made from fish, which is animal-derived. ❌ NOT vegan\n\n**Vegetarian:** Allows dairy and eggs but prohibits meat, poultry, fish, shellfish, and animal-derived rennet/gelatine. Dashi made from katsuobushi (dried bonito fish) is fish-based. ❌ NOT vegetarian\n\n**Pescatarian:** Allows fish and shellfish but prohibits meat and poultry. Dashi made from katsuobushi (fish) is permitted. Tofu, wakame, and miso are all compatible. ✅ Pescatarian\n\npescatarian",
          "compliant": true,
          "score": 0.0952,
          "detail": {
            "predicted": [
              "**Ingredients:**",
              "**Label Analysis:**",
              "**Pescatarian:** Allows fish and shellfish but prohibits meat and poultry. Dashi made from katsuobushi (fish) is permitted. Tofu",
              "**Vegan:** Requires no animal-derived ingredients. Dashi (katsuobushi) is made from fish",
              "**Vegetarian:** Allows dairy and eggs but prohibits meat",
              "- Dashi (katsuobushi): made from bonito fish (dried",
              "- Miso: typically made with soybeans and salt; most are vegetarian but some may contain dashi (fish-based)",
              "- Tofu: soy-based",
              "- Wakame: seaweed",
              "I need to analyze each ingredient against the three dietary labels:",
              "and animal-derived rennet/gelatine. Dashi made from katsuobushi (dried bonito fish) is fish-based. ❌ NOT vegetarian",
              "and miso are all compatible. ✅ Pescatarian",
              "fermented fish flakes) - **animal-derived**",
              "fish",
              "pescatarian",
              "plant-derived",
              "poultry",
              "shellfish",
              "wakame",
              "which is animal-derived. ❌ NOT vegan"
            ],
            "expected": [
              "pescatarian"
            ],
            "precision": 0.05,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1042,
            "output_tokens": 280,
            "latency_ms": 2909.9,
            "usd": 0.001954,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Let me analyze each ingredient against the dietary labels:\n\n**Ingredients:**\n- **Miso**: typically made from soybeans, salt, and koji; usually vegan\n- **Dashi (katsuobushi)**: made from dried bonito fish - contains fish\n- **Tofu**: soy product; vegan\n- **Wakame**: seaweed; vegan\n\n**Analysis:**\n\n- **Vegan**: Requires no animal-derived ingredients. Dashi made with katsuobushi (dried bonito fish) violates this. ❌\n- **Vegetarian**: Allows dairy and eggs but prohibits meat, poultry, fish, shellfish, and animal-derived rennet/gelatine. Katsuobushi is fish. ❌\n- **Pescatarian**: Allows fish and shellfish but prohibits meat and poultry. Dashi made with katsuobushi (fish) is permitted. ✅\n\npescatarian",
          "compliant": true,
          "score": 0.1111,
          "detail": {
            "predicted": [
              "**Analysis:**",
              "**Ingredients:**",
              "- **Dashi (katsuobushi)**: made from dried bonito fish - contains fish",
              "- **Miso**: typically made from soybeans",
              "- **Pescatarian**: Allows fish and shellfish but prohibits meat and poultry. Dashi made with katsuobushi (fish) is permitted. ✅",
              "- **Tofu**: soy product; vegan",
              "- **Vegan**: Requires no animal-derived ingredients. Dashi made with katsuobushi (dried bonito fish) violates this. ❌",
              "- **Vegetarian**: Allows dairy and eggs but prohibits meat",
              "- **Wakame**: seaweed; vegan",
              "Let me analyze each ingredient against the dietary labels:",
              "and animal-derived rennet/gelatine. Katsuobushi is fish. ❌",
              "and koji; usually vegan",
              "fish",
              "pescatarian",
              "poultry",
              "salt",
              "shellfish"
            ],
            "expected": [
              "pescatarian"
            ],
            "precision": 0.0588,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1042,
            "output_tokens": 222,
            "latency_ms": 2432.9,
            "usd": 0.001722,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.291,
      "score_summary": {
        "mean": 0.291,
        "stdev": 0.3255,
        "sem": 0.1879,
        "ci95_lo": -0.0773,
        "ci95_hi": 0.6593,
        "n": 3
      },
      "score_if_valid": 0.291,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1042.0,
        "output_tokens_mean": 170.333,
        "latency_ms_mean": 2056.767,
        "usd_sum": 0.004546
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-001",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "138c09ed611a2d43",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has above-average popularity and above-average contribution margin on the menu. Using the OGS menu-engineering matrix, which quadrant is it? Respond with one word.\n\nINPUT:\n{\n  \"popularity\": \"above\",\n  \"margin\": \"above\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "star"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "star",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "star",
            "expected": "star",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 557,
            "output_tokens": 4,
            "latency_ms": 650.7,
            "usd": 0.000462,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "star",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "star",
            "expected": "star",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 557,
            "output_tokens": 4,
            "latency_ms": 1082.2,
            "usd": 0.000462,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "star",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "star",
            "expected": "star",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 557,
            "output_tokens": 4,
            "latency_ms": 627.9,
            "usd": 0.000462,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 557.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 786.933,
        "usd_sum": 0.001386
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-002",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "b5df321fd829de76",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish sells strongly but the contribution margin is below the menu average. Classify it.\n\nINPUT:\n{\n  \"popularity\": \"above\",\n  \"margin\": \"below\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "plowhorse"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "plowhorse",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "plowhorse",
            "expected": "plowhorse",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 536,
            "output_tokens": 7,
            "latency_ms": 592.2,
            "usd": 0.000457,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "plowhorse",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "plowhorse",
            "expected": "plowhorse",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 536,
            "output_tokens": 7,
            "latency_ms": 552.3,
            "usd": 0.000457,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "plowhorse",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "plowhorse",
            "expected": "plowhorse",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 536,
            "output_tokens": 7,
            "latency_ms": 804.6,
            "usd": 0.000457,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 536.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 649.7,
        "usd_sum": 0.001371
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-003",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "9795b72bdda437f1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has strong contribution margin but is rarely ordered. Classify it.\n\nINPUT:\n{\n  \"popularity\": \"below\",\n  \"margin\": \"above\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "puzzle"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "puzzle",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "puzzle",
            "expected": "puzzle",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 533,
            "output_tokens": 4,
            "latency_ms": 714.3,
            "usd": 0.000442,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "puzzle",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "puzzle",
            "expected": "puzzle",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 533,
            "output_tokens": 4,
            "latency_ms": 688.3,
            "usd": 0.000442,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "puzzle",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "puzzle",
            "expected": "puzzle",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 533,
            "output_tokens": 4,
            "latency_ms": 728.5,
            "usd": 0.000442,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 533.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 710.367,
        "usd_sum": 0.001326
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-004",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "fc0da6dcbbc842fb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish is rarely ordered and has a contribution margin below the menu average. Classify it.\n\nINPUT:\n{\n  \"popularity\": \"below\",\n  \"margin\": \"below\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "dog"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "dog",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "dog",
            "expected": "dog",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 537,
            "output_tokens": 4,
            "latency_ms": 508.5,
            "usd": 0.000446,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "dog",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "dog",
            "expected": "dog",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 537,
            "output_tokens": 4,
            "latency_ms": 536.0,
            "usd": 0.000446,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "dog",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "dog",
            "expected": "dog",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 537,
            "output_tokens": 4,
            "latency_ms": 775.3,
            "usd": 0.000446,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 537.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 606.6,
        "usd_sum": 0.001338
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-005",
      "category": "menu_engineering",
      "capability": "food_cost",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "f33710f88f89121e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has a food cost of £4.20 and a menu price of £14.00. Return the food-cost percentage as a single integer.\n\nINPUT:\n{\n  \"food_cost\": 4.2,\n  \"menu_price\": 14.0\n}\n\nANSWER FORMAT: numeric:integer percent",
      "expected": {
        "value": 30,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "30",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 30.0,
            "expected": 30.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 550,
            "output_tokens": 5,
            "latency_ms": 912.7,
            "usd": 0.00046,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "30",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 30.0,
            "expected": 30.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 550,
            "output_tokens": 5,
            "latency_ms": 539.0,
            "usd": 0.00046,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "30",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 30.0,
            "expected": 30.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 550,
            "output_tokens": 5,
            "latency_ms": 677.0,
            "usd": 0.00046,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 550.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 709.567,
        "usd_sum": 0.00138
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-006",
      "category": "menu_engineering",
      "capability": "contribution_margin",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "7bb019ca9ca4b1be",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has a food cost of $6.50 and a menu price of $22.00. Return the contribution margin in dollars as a single number with two decimals.\n\nINPUT:\n{\n  \"food_cost\": 6.5,\n  \"menu_price\": 22.0\n}\n\nANSWER FORMAT: numeric:decimal USD",
      "expected": {
        "value": 15.5,
        "tolerance": 0.05
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "15.50",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 15.5,
            "expected": 15.5,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 554,
            "output_tokens": 7,
            "latency_ms": 621.6,
            "usd": 0.000471,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "15.50",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 15.5,
            "expected": 15.5,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 554,
            "output_tokens": 7,
            "latency_ms": 720.4,
            "usd": 0.000471,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "15.50",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 15.5,
            "expected": 15.5,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 554,
            "output_tokens": 7,
            "latency_ms": 587.3,
            "usd": 0.000471,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 554.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 643.1,
        "usd_sum": 0.001413
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-007",
      "category": "menu_engineering",
      "capability": "menu_rebalance",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "68b7fb2ec1e432f4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nFive main courses have this (popularity %, margin $) profile: A (35, 18), B (30, 9), C (5, 6), D (20, 16), E (10, 7). Average popularity is 20%; average margin is $11.20. Which single dish is the dog? Respond with the letter only.\n\nINPUT:\n{\n  \"dishes\": [\n    {\n      \"id\": \"A\",\n      \"pop\": 35,\n      \"margin\": 18\n    },\n    {\n      \"id\": \"B\",\n      \"pop\": 30,\n      \"margin\": 9\n    },\n    {\n      \"id\": \"C\",\n      \"pop\": 5,\n      \"margin\": 6\n    },\n    {\n      \"id\": \"D\",\n      \"pop\": 20,\n      \"margin\": 16\n    },\n    {\n      \"id\": \"E\",\n      \"pop\": 10,\n      \"margin\": 7\n    }\n  ],\n  \"avg_pop\": 20,\n  \"avg_margin\": 11.2\n}\n\nANSWER FORMAT: exact:A|B|C|D|E",
      "expected": {
        "value": "C"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c",
            "expected": "c",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 753,
            "output_tokens": 4,
            "latency_ms": 773.2,
            "usd": 0.000618,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c",
            "expected": "c",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 753,
            "output_tokens": 4,
            "latency_ms": 543.8,
            "usd": 0.000618,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c",
            "expected": "c",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 753,
            "output_tokens": 4,
            "latency_ms": 733.8,
            "usd": 0.000618,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 753.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 683.6,
        "usd_sum": 0.001854
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-008",
      "category": "menu_engineering",
      "capability": "target_price",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "dc5bf7ed3a0583c2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA chef wants a 28% food cost on a dish whose ingredient cost is £3.92. Return the target menu price in pounds, two decimals.\n\nINPUT:\n{\n  \"food_cost\": 3.92,\n  \"target_food_cost_pct\": 28\n}\n\nANSWER FORMAT: numeric:decimal GBP",
      "expected": {
        "value": 14.0,
        "tolerance": 0.05
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "14.00",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 14.0,
            "expected": 14.0,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 555,
            "output_tokens": 7,
            "latency_ms": 513.9,
            "usd": 0.000472,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "14.00",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 14.0,
            "expected": 14.0,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 555,
            "output_tokens": 7,
            "latency_ms": 1350.0,
            "usd": 0.000472,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "14.00",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 14.0,
            "expected": 14.0,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 555,
            "output_tokens": 7,
            "latency_ms": 530.2,
            "usd": 0.000472,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 555.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 798.033,
        "usd_sum": 0.001416
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-009",
      "category": "menu_engineering",
      "capability": "menu_critique",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "rubric_judge",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles",
        "vocab.match_types",
        "rules.dietary",
        "rules.allergens_eu"
      ],
      "prompt_template_hash": "8883cacb0aca5024",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n---\n\nCritique the following five-course autumn tasting menu. Identify at least two concrete weaknesses (flow, seasonality, allergen risk, cost-to-price ratio, or structural monotony), suggest one specific, constructive change, and justify each point in 3 short paragraphs. Avoid generic praise.\n\nINPUT:\n{\n  \"menu\": [\n    {\n      \"course\": \"amuse\",\n      \"dish\": \"pumpkin velouté, sage oil\"\n    },\n    {\n      \"course\": \"appetizer\",\n      \"dish\": \"butternut squash risotto, brown butter, amaretti crumble\"\n    },\n    {\n      \"course\": \"fish\",\n      \"dish\": \"pan-seared scallops, pumpkin purée, crispy pancetta\"\n    },\n    {\n      \"course\": \"main\",\n      \"dish\": \"roast pork loin, sweet potato purée, cider jus\"\n    },\n    {\n      \"course\": \"dessert\",\n      \"dish\": \"pumpkin cheesecake, spiced pecan brittle\"\n    }\n  ],\n  \"target_food_cost_pct\": 30,\n  \"dietary_constraints\": \"must offer a viable vegetarian and a viable nut-free path\"\n}\n\nANSWER FORMAT: free text, 3 short paragraphs",
      "expected": {
        "rubric": [
          {
            "id": "diagnosis",
            "weight": 0.4,
            "description": "Identifies concrete, defensible weaknesses (flavor monotony from repeated pumpkin/squash, nut allergen in amaretti and pecans, heavy starches in consecutive courses).",
            "anchors": {
              "0.0": "No specific weakness identified.",
              "0.5": "Identifies one concrete weakness but misses another major one.",
              "1.0": "Identifies at least two of: pumpkin/squash repetition across 4 of 5 courses; nut allergen across amaretti and pecans; consecutive purée/starch heaviness."
            }
          },
          {
            "id": "remediation",
            "weight": 0.3,
            "description": "Offers a specific, actionable, seasonally-plausible change that addresses the identified weakness.",
            "anchors": {
              "0.0": "No change proposed or change is vague.",
              "0.5": "Change is specific but does not address the main weakness.",
              "1.0": "Concrete substitute with named ingredient/technique (e.g. swap scallop course to cured mackerel with pickled apple)."
            }
          },
          {
            "id": "justification",
            "weight": 0.3,
            "description": "Justification is grounded in OGS concepts (match types, structural balance, service flow) rather than marketing language.",
            "anchors": {
              "0.0": "Pure adjective salad.",
              "0.5": "Some reasoning, but appeals to authority not mechanism.",
              "1.0": "Explicitly cites a mechanism: intensity alignment, contrast, allergen cross-contact, food-cost math."
            }
          }
        ],
        "reference_answer": "Key weaknesses: (1) pumpkin/squash appears in 4 of 5 courses, collapsing the menu's flavour arc into a single beta-carotene/sweet-starch note; (2) the amaretti crumble (bitter almond) and the pecan brittle together block every nut-allergic guest from the tasting; (3) velouté, risotto, two purées and a cheesecake stack four viscous/starchy textures in a row with no acid or raw element. A concrete fix is to replace the scallop course with cured mackerel over pickled Granny Smith apple and wasabi-lime gel: it re-introduces sea/fruit/acid, breaks the texture loop, removes one pork-cross-contact dish, and simultaneously opens a pescatarian-friendly option. Justification: on OGS terms this swap adds a `contrast` match (acid/fat), raises `acidity_perception` mid-meal to refresh the palate before the roast, and removes one of the two tree-nut exposure points."
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "**Weakness 1: Monotonous vegetable anchor and flavor repetition**\n\nPumpkin appears in four of five courses (amuse, appetizer, fish, dessert), creating structural tedium rather than sophisticated variation. While squash is seasonally appropriate, the menu reads as a pumpkin-centric tasting rather than a balanced autumn exploration. The repetition is compounded by the use of purée as the primary pumpkin preparation across multiple courses—a textural echo that undermines the pace and discovery guests expect from a tasting menu. A five-course sequence should build contrast; instead, diners encounter the same ingredient in near-identical applications.\n\n**Suggested change:** Replace the fish course with a dish built on a different autumn vegetable—for example, roasted celeriac or salsify—while retaining the scallop protein and brown-butter family. Alternatively, reposition the dessert to showcase apple or pear, which are equally seasonal and create a cleaner flavor arc from gourd-based early courses to stone or pomaceous fruit at the end.\n\n**Weakness 2: Allergen and dietary accommodation risk**\n\nThe vegetarian path is compromised by the amuse and fish courses: pancetta in the scallop dish means vegetarians skip the protein-forward course entirely, creating an awkward three-course sequence (amuse → appetizer → main → dessert). The nut-free requirement fails at the appetizer (amaretti, tree nuts) and dessert (pecan brittle, tree nuts). Offering viable alternatives at every course is non-negotiable in professional service; this menu forces guests into fragmented experiences rather than coherent five-course narratives.\n\n**Suggested change:** Reformulate the appetizer crumble using toasted breadcrumbs and brown butter instead of amaretti, and replace the pecan brittle with a spiced caramel or candied sage garnish. For vegetarians, offer a separate amuse (perhaps a mushroom consommé or roasted beet variant) so the fish course substitution feels intentional rather than an omission.\n\n**Weakness 3: Contribution margin squeeze from repeated expensive proteins and high-waste preparations**\n\nScallops and pork loin are premium proteins; combined with three separate purées (pumpkin, sweet potato, and implied celeriac or root base for the risotto), the food-cost structure is likely above 30%. Purées require fresh vegetable volume and processing labor. Pancetta, brown butter emulsions, and a spiced brittle add further ingredient and labor costs. Without ingredient-level pricing, it is difficult to confirm, but the margin on this menu probably sits at 32–35% food cost, overshooting the target by 2–5 percentage points—material enough to erode profitability on a lower-ticket tasting.\n\n**Suggested change:** Reduce the protein tier by replacing the scallop course with a vegetable-forward dish (roasted mushroom, braised endive, or autumn squash) dressed with a modest amount of fish or shellfish stock for umami, rather than whole scallops. Alternatively, batch-prepare one large-volume purée (e.g., roasted pumpkin) to serve across amuse and main, reducing per-plate preparation cost and improving kitchen efficiency while maintaining visual and textural variety through different finishing techniques (oil, cream, crisp).",
          "compliant": true,
          "score": 0.94,
          "detail": {
            "per_criterion": [
              {
                "id": "diagnosis",
                "weight": 0.4,
                "score": 1.0,
                "note": "It clearly identifies multiple concrete weaknesses: repeated pumpkin/squash across four courses, nut-allergen failures in amaretti and pecans, and textural/starch monotony from repeated purées and heavy preparations."
              },
              {
                "id": "remediation",
                "weight": 0.3,
                "score": 0.9,
                "note": "It proposes specific substitutions such as apple/pear dessert, breadcrumb crumble for amaretti, and caramel for pecan brittle, though some alternatives are presented as options rather than one tightly focused fix."
              },
              {
                "id": "justification",
                "weight": 0.3,
                "score": 0.9,
                "note": "The reasoning is grounded in structural balance, service flow, dietary-path coherence, and food-cost implications, with explicit mechanisms rather than generic praise."
              }
            ],
            "summary": "The candidate is fully usable and directly critiques the menu with concrete, defensible weaknesses. It exceeds the minimum by covering monotony, allergen risk, vegetarian viability, and cost structure, and supports its recommendations with operational and sensory logic.",
            "judge_model": "gpt-5.4",
            "judge_prompt_hash": "756db6a487714bca",
            "judge_validated": false
          },
          "metrics": {
            "input_tokens": 1460,
            "output_tokens": 751,
            "latency_ms": 10759.1,
            "usd": 0.004172,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "# Critique: Five-Course Autumn Tasting Menu\n\n## Weakness 1: Severe Flavor Monotony & Ingredient Repetition\n\n**Problem:** Pumpkin appears in four of five courses (amuse, appetizer, fish, dessert), and squash variants dominate three consecutive courses (amuse → appetizer → fish). This creates palatal fatigue rather than progression. A tasting menu should build narrative contrast; instead, diners experience diminishing returns on a single autumn ingredient.\n\n**Why it matters:** Even premium ingredients lose impact through saturation. The fish course—normally a palate-cleansing pivot—becomes another vehicle for pumpkin, sacrificing structural balance that the Boston Matrix calls \"structural_balance\" in pairing.\n\n**Specific fix:** Replace the fish course with pan-seared sea bass, celery root purée, and brown shrimp beurre blanc. This preserves autumn umami, introduces a fresh celery-mineral note to break the squash monotony, and repositions the fish course as a true reset before the richer main.\n\n---\n\n## Weakness 2: Vegetarian Path Breaks Down at Protein Steps\n\n**Problem:** The menu offers no viable vegetarian main course. The appetizer (butternut squash risotto) *could* serve vegetarians, but it follows an amuse they cannot eat (sage oil may be vegetarian, but the protocol requires a fully compliant path). The fish and main courses are both animal proteins with no vegetarian equivalent listed. This violates the stated dietary constraint and creates operational liability.\n\n**Why it matters:** A tasting menu without a documented vegetarian trajectory forces last-minute substitutions that devalue the dining experience. Legally and reputationally, \"viable vegetarian path\" means a coherent five-course sequence, not ad-hoc swaps.\n\n**Specific fix:** Explicitly plan a vegetarian main—e.g., wild mushroom and chestnut ragout with crispy sage and brown butter—and confirm the amuse works vegetarian (or offer a vegetable-forward alternative). Document this as the \"Vegetarian Tasting\" offering, priced at a 10–15% premium to reflect ingredient cost.\n\n---\n\n## Weakness 3: Nut Allergies Create a Dessert Void\n\n**Problem:** Both the amaretti crumble (appetizer) and pecan brittle (dessert) contain tree nuts. A nut-free diner must skip two courses or receive substitutes. More critically, pecan brittle is likely a key textural and flavor anchor in the dessert; a last-minute mousse replacement undermines the chef's intent and menu integrity.\n\n**Why it matters:** Tree nuts rank among the top eight allergens and affect 1–4% of diners. Requiring a redesigned finale for a significant segment is poor menu engineering and suggests the menu was not backward-engineered for the stated constraint.\n\n**Specific fix:** Replace the pecan brittle with a cider-caramel crisp (reduced cider, caramelized sugar, toasted oat crumble, fleur de sel). This maintains the sweet-salty crunch profile, delivers authentic autumn flavor, and is naturally nut-free. Simultaneously replace the amaretti crumble with toasted breadcrumb and brown butter, or remove it and increase risotto richness via aged Parmigiano.",
          "compliant": true,
          "score": 0.94,
          "detail": {
            "per_criterion": [
              {
                "id": "diagnosis",
                "weight": 0.4,
                "score": 1.0,
                "note": "It identifies multiple concrete weaknesses: repeated pumpkin/squash monotony, lack of a viable vegetarian path, nut-allergen exposure, and structural heaviness by implication."
              },
              {
                "id": "remediation",
                "weight": 0.3,
                "score": 1.0,
                "note": "It proposes specific, actionable substitutions with named dishes and ingredients that directly address the cited problems."
              },
              {
                "id": "justification",
                "weight": 0.3,
                "score": 0.8,
                "note": "The reasoning mostly uses menu-structure mechanisms like palate fatigue, course progression, and allergen compliance, though one reference is vague and food-cost logic is thin."
              }
            ],
            "summary": "The candidate gives a usable critique in free text and clearly exceeds the minimum by naming several defensible weaknesses and pairing each with a concrete fix. Its strongest points are the monotony and allergen diagnoses; justification is generally mechanism-based, though slightly weakened by a stray unsupported framework reference and limited cost-to-price analysis.",
            "judge_model": "gpt-5.4",
            "judge_prompt_hash": "756db6a487714bca",
            "judge_validated": false
          },
          "metrics": {
            "input_tokens": 1460,
            "output_tokens": 755,
            "latency_ms": 10258.8,
            "usd": 0.004188,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "# Critique of Five-Course Autumn Tasting Menu\n\n## Weakness 1: Excessive Pumpkin Repetition (Structural Monotony)\n\nPumpkin appears in three consecutive courses—amuse, appetizer, and fish—creating monotonous flavor and textural redundancy that exhausts the ingredient before the halfway point. This violates foundational tasting-menu logic: each course should introduce contrasting sensory territories to sustain anticipation and showcase seasonal breadth. The diner experiences pumpkin as velouté, then risotto, then purée again, with only sage oil, brown butter, and scallop offering marginal relief.\n\nThe dessert resurrects pumpkin a fourth time, cementing the impression that the menu is a pumpkin vehicle rather than an autumn exploration. While pumpkin is seasonally relevant, restraint is a hallmark of sophisticated tasting menus—reserve it for one showstopping moment rather than dilute its impact across the entire progression.\n\n**Suggested change:** Replace the fish course with a different autumn protein—for example, pan-seared mushroom risotto or venison carpaccio with beetroot, hazelnuts, and horseradish cream. This preserves the autumn narrative while introducing earthiness, umami depth, and textural contrast that reset the palate before the rich pork main.\n\n---\n\n## Weakness 2: Acute Dietary Constraint Failure (Vegetarian & Nut-Free Paths)\n\nThe vegetarian path collapses at the fish and main courses. Scallops and pork loin offer no immediate substitutes without menu redesign, forcing the kitchen to improvise off-menu or serve an incomplete tasting. The amuse and appetizer are vegetarian-compatible, but courses 3 and 4 break continuity entirely, a structural flaw in a five-course progression.\n\nThe nut-free path simultaneously fails: amaretti (ground almonds) appears in course 2, pecan brittle anchors the dessert, and if you substitute either dish, you've undermined the vegetarian option. The menu has created a logical impossibility where meeting both constraints simultaneously is impossible without wholesale rebuilding.\n\n**Suggested change:** Redesign the appetizer and dessert to share neutral foundations. Replace \"butternut squash risotto, brown butter, amaretti crumble\" with \"butternut squash risotto, sage beurre noisette, toasted pumpkin seeds\" (vegetarian, nut-free, equally seasonal). For dessert, shift from pecan brittle to \"spiced pumpkin cheesecake with candied apple and crispy sage\" or a completely different course. This dual-layer design ensures both vegetarian and nut-free diners experience the intended progression without kitchen workarounds.\n\n---\n\n## Weakness 3: Food-Cost Risk on High-Margin Items (Cheese Course Absence & Pricing Vulnerability)\n\nThe menu omits cheese—a traditional autumn course and a structurally high-margin item that aids cost-per-seat economics. At 30% target food cost, five dishes leaving no cheese slot means each dish must deliver larger contribution margins independently, creating pricing pressure and leaving little flexibility if a protein (scallops, pork) costs exceed forecast.\n\nCheesecake as dessert is a budget-friendly, lower-ingredient-cost option, but it does not provide the sensory reset or palate-cleansing bridge that a savory cheese course would. Serving pork loin directly into rich, sweet cheesecake creates a heavy, sameness-prone progression that feels underdeveloped for a tasting menu at premium price point.\n\n**Suggested change:** Insert a brief cheese course between the main and dessert—for example, \"aged cheddar crisps with quince paste and watercress\" (< 8 minute intermezzo). This adds ~3–5% margin per seat, provides structural breathing room, and introduces an acidic, bitter, savory pivot that resets the palate before dessert. It also trivially resolves the vegetarian constraint for that course without additional labor.",
          "compliant": true,
          "score": 0.7,
          "detail": {
            "per_criterion": [
              {
                "id": "diagnosis",
                "weight": 0.4,
                "score": 1.0,
                "note": "It clearly identifies multiple concrete weaknesses: repeated pumpkin/squash across four courses, nut-allergen exposure in amaretti and pecans, and vegetarian-path failure."
              },
              {
                "id": "remediation",
                "weight": 0.3,
                "score": 0.5,
                "note": "It proposes specific substitutions, but the main fish-course change is muddled and one option adds hazelnuts, while the cheese-course insertion is not aligned to the five-course brief."
              },
              {
                "id": "justification",
                "weight": 0.3,
                "score": 0.5,
                "note": "The reasoning references menu flow, palate fatigue, and cost pressure, but much of it stays conceptual and only lightly engages mechanism such as allergen cross-contact or explicit food-cost math."
              }
            ],
            "summary": "The answer is usable and follows the requested free-text critique format while avoiding generic praise. It diagnoses the core menu problems well, but some proposed fixes are inconsistent with the constraints or structure, and the justification is only partly grounded in explicit operational mechanisms.",
            "judge_model": "gpt-5.4",
            "judge_prompt_hash": "756db6a487714bca",
            "judge_validated": false
          },
          "metrics": {
            "input_tokens": 1460,
            "output_tokens": 918,
            "latency_ms": 11815.4,
            "usd": 0.00484,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.86,
      "score_summary": {
        "mean": 0.86,
        "stdev": 0.1386,
        "sem": 0.08,
        "ci95_lo": 0.7032,
        "ci95_hi": 1.0168,
        "n": 3
      },
      "score_if_valid": 0.86,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1460.0,
        "output_tokens_mean": 808.0,
        "latency_ms_mean": 10944.433,
        "usd_sum": 0.0132
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ar-001",
      "category": "vocabulary",
      "capability": "ingredient_classification",
      "difficulty": "easy",
      "cuisine_region": "middle_eastern",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.ingredient_categories"
      ],
      "prompt_template_hash": "94e19d81d80368a1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n---\n\nباستخدام مفردات OGS لتصنيفات المكونات، صنّف الزعتر بكلمة واحدة.\n\nINPUT:\n{\n  \"item\": \"زعتر\"\n}\n\nANSWER FORMAT: one_of:protein,vegetable,fruit,grain,dairy,fat,herb,spice,fungus,legume,nut,seed,sweetener,acid,salt,other",
      "expected": {
        "value": "herb",
        "aliases": [
          "spice"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "herb",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herb",
            "expected": "herb",
            "aliases": [
              "spice"
            ]
          },
          "metrics": {
            "input_tokens": 403,
            "output_tokens": 4,
            "latency_ms": 765.5,
            "usd": 0.000338,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "herb",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herb",
            "expected": "herb",
            "aliases": [
              "spice"
            ]
          },
          "metrics": {
            "input_tokens": 403,
            "output_tokens": 4,
            "latency_ms": 884.1,
            "usd": 0.000338,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "herb",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herb",
            "expected": "herb",
            "aliases": [
              "spice"
            ]
          },
          "metrics": {
            "input_tokens": 403,
            "output_tokens": 4,
            "latency_ms": 581.5,
            "usd": 0.000338,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 403.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 743.7,
        "usd_sum": 0.001014
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ar-002",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "middle_eastern",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "6b39ec376be689ad",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nالمنسف هو الطبق الوطني لأي بلد عربي؟ أجب بكلمة واحدة.\n\nINPUT:\n{\n  \"dish\": \"المنسف\"\n}\n\nANSWER FORMAT: exact:country",
      "expected": {
        "value": "jordan",
        "aliases": [
          "الأردن",
          "jordan (الأردن)"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "الأردن",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "الأردن",
            "expected": "jordan",
            "aliases": [
              "الأردن",
              "jordan (الأردن)"
            ]
          },
          "metrics": {
            "input_tokens": 312,
            "output_tokens": 7,
            "latency_ms": 1835.8,
            "usd": 0.000278,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "الأردن",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "الأردن",
            "expected": "jordan",
            "aliases": [
              "الأردن",
              "jordan (الأردن)"
            ]
          },
          "metrics": {
            "input_tokens": 312,
            "output_tokens": 7,
            "latency_ms": 610.0,
            "usd": 0.000278,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "الأردن",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "الأردن",
            "expected": "jordan",
            "aliases": [
              "الأردن",
              "jordan (الأردن)"
            ]
          },
          "metrics": {
            "input_tokens": 312,
            "output_tokens": 7,
            "latency_ms": 564.3,
            "usd": 0.000278,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 312.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1003.367,
        "usd_sum": 0.000834
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-es-001",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "5fe5c7f1f85e3675",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nEl pulpo a feira es un plato tradicional de qué región de España? Responde con una palabra.\n\nINPUT:\n{\n  \"dish\": \"pulpo a feira\"\n}\n\nANSWER FORMAT: exact:region name",
      "expected": {
        "value": "galicia",
        "aliases": [
          "galícia",
          "galiza"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Galicia",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "galicia",
            "expected": "galicia",
            "aliases": [
              "galícia",
              "galiza"
            ]
          },
          "metrics": {
            "input_tokens": 299,
            "output_tokens": 6,
            "latency_ms": 925.7,
            "usd": 0.000263,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Galicia",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "galicia",
            "expected": "galicia",
            "aliases": [
              "galícia",
              "galiza"
            ]
          },
          "metrics": {
            "input_tokens": 299,
            "output_tokens": 6,
            "latency_ms": 580.1,
            "usd": 0.000263,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Galicia",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "galicia",
            "expected": "galicia",
            "aliases": [
              "galícia",
              "galiza"
            ]
          },
          "metrics": {
            "input_tokens": 299,
            "output_tokens": 6,
            "latency_ms": 552.5,
            "usd": 0.000263,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 299.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 686.1,
        "usd_sum": 0.000789
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-es-002",
      "category": "vocabulary",
      "capability": "cooking_method_classification",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "9f03b90711bec40f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nUsando el vocabulario OGS de métodos de cocción, ¿cuál describe mejor la preparación clásica del cochinillo segoviano? Responde con una palabra.\n\nINPUT:\n{\n  \"item\": \"cochinillo segoviano\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "roasted"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 457,
            "output_tokens": 5,
            "latency_ms": 908.1,
            "usd": 0.000386,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 457,
            "output_tokens": 5,
            "latency_ms": 730.4,
            "usd": 0.000386,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 457,
            "output_tokens": 5,
            "latency_ms": 1231.9,
            "usd": 0.000386,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 457.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 956.8,
        "usd_sum": 0.001158
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-fr-001",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "8d252ae3fb170e15",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nLa bouillabaisse est une préparation traditionnelle de quelle ville française ? Répondez par un mot.\n\nINPUT:\n{\n  \"dish\": \"bouillabaisse\"\n}\n\nANSWER FORMAT: exact:city name",
      "expected": {
        "value": "marseille"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Marseille",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "marseille",
            "expected": "marseille",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 303,
            "output_tokens": 6,
            "latency_ms": 527.6,
            "usd": 0.000266,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Marseille",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "marseille",
            "expected": "marseille",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 303,
            "output_tokens": 6,
            "latency_ms": 701.5,
            "usd": 0.000266,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Marseille",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "marseille",
            "expected": "marseille",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 303,
            "output_tokens": 6,
            "latency_ms": 536.7,
            "usd": 0.000266,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 303.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 588.6,
        "usd_sum": 0.000798
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-fr-002",
      "category": "sensory",
      "capability": "basic_tastes_profile",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "96a07aeab44536ab",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nPour une sauce beurre blanc classique, donnez le profil sur l'échelle OGS des cinq saveurs [sweet, sour, salty, bitter, umami] sous la forme d'un tableau JSON de cinq nombres entre 0 et 10.\n\nINPUT:\n{\n  \"dish\": \"sauce beurre blanc classique\"\n}\n\nANSWER FORMAT: json:array of 5 numbers",
      "expected": {
        "values": [
          1,
          5,
          4,
          1,
          3
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[1.5, 6.0, 3.5, 0.5, 2.0]\n```",
          "compliant": true,
          "score": 0.5333,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 1.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 1,
                "predicted": 6.0,
                "expected": 5,
                "score": 0.3333
              },
              {
                "index": 2,
                "predicted": 3.5,
                "expected": 4,
                "score": 0.6667
              },
              {
                "index": 3,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 4,
                "predicted": 2.0,
                "expected": 3,
                "score": 0.3333
              }
            ],
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 430,
            "output_tokens": 33,
            "latency_ms": 1340.6,
            "usd": 0.000476,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[1.5, 6.5, 4.0, 0.5, 2.5]\n```",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 1.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 1,
                "predicted": 6.5,
                "expected": 5,
                "score": 0.0
              },
              {
                "index": 2,
                "predicted": 4.0,
                "expected": 4,
                "score": 1.0
              },
              {
                "index": 3,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 4,
                "predicted": 2.5,
                "expected": 3,
                "score": 0.6667
              }
            ],
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 430,
            "output_tokens": 33,
            "latency_ms": 807.3,
            "usd": 0.000476,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[1.5, 6.0, 2.5, 0.5, 2.0]\n```",
          "compliant": true,
          "score": 0.4,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 1.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 1,
                "predicted": 6.0,
                "expected": 5,
                "score": 0.3333
              },
              {
                "index": 2,
                "predicted": 2.5,
                "expected": 4,
                "score": 0.0
              },
              {
                "index": 3,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 4,
                "predicted": 2.0,
                "expected": 3,
                "score": 0.3333
              }
            ],
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 430,
            "output_tokens": 33,
            "latency_ms": 1016.1,
            "usd": 0.000476,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5111,
      "score_summary": {
        "mean": 0.5111,
        "stdev": 0.1018,
        "sem": 0.0588,
        "ci95_lo": 0.3959,
        "ci95_hi": 0.6263,
        "n": 3
      },
      "score_if_valid": 0.5111,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 430.0,
        "output_tokens_mean": 33.0,
        "latency_ms_mean": 1054.667,
        "usd_sum": 0.001428
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-it-001",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "c3aae88ebfcbe08b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nI pici sono una pasta fresca tipica di quale regione italiana? Rispondi con una parola.\n\nINPUT:\n{\n  \"dish\": \"pici cacio e pepe\"\n}\n\nANSWER FORMAT: exact:regione",
      "expected": {
        "value": "toscana"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Toscana",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "toscana",
            "expected": "toscana",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 302,
            "output_tokens": 7,
            "latency_ms": 527.0,
            "usd": 0.00027,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Toscana",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "toscana",
            "expected": "toscana",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 302,
            "output_tokens": 7,
            "latency_ms": 2213.9,
            "usd": 0.00027,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Toscana",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "toscana",
            "expected": "toscana",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 302,
            "output_tokens": 7,
            "latency_ms": 712.3,
            "usd": 0.00027,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 302.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1151.067,
        "usd_sum": 0.00081
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-it-002",
      "category": "pairing",
      "capability": "beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.beverage_types",
        "vocab.match_types"
      ],
      "prompt_template_hash": "30919f613b1fbcd6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nScegli l'abbinamento più classico per l'ossobuco alla milanese fra: barolo, riesling, prosecco, lambrusco. Rispondi con una parola.\n\nINPUT:\n{\n  \"dish\": \"ossobuco alla milanese\"\n}\n\nANSWER FORMAT: exact:one of the given options",
      "expected": {
        "value": "barolo"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "barolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "barolo",
            "expected": "barolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 5,
            "latency_ms": 886.9,
            "usd": 0.000317,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "barolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "barolo",
            "expected": "barolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 5,
            "latency_ms": 580.5,
            "usd": 0.000317,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "barolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "barolo",
            "expected": "barolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 5,
            "latency_ms": 630.4,
            "usd": 0.000317,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 371.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 699.267,
        "usd_sum": 0.000951
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ja-001",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "ff1f43723314705c",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nOGSの香りファミリー一覧から、柚子（ゆず）の主たる香りファミリーを一語で答えてください。\n\nINPUT:\n{\n  \"item\": \"柚子\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "fruit"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 383,
            "output_tokens": 4,
            "latency_ms": 629.2,
            "usd": 0.000322,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 383,
            "output_tokens": 4,
            "latency_ms": 735.1,
            "usd": 0.000322,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 383,
            "output_tokens": 4,
            "latency_ms": 916.7,
            "usd": 0.000322,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 383.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 760.333,
        "usd_sum": 0.000966
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ja-002",
      "category": "pairing",
      "capability": "beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.beverage_types",
        "vocab.match_types"
      ],
      "prompt_template_hash": "f68d4882814e0ecf",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\n握り寿司（赤身・白身・光物）のコース全体に最も安全に合う飲料のタイプを、次から一つ選んでください: sake, red_wine, stout_beer, bourbon.\n\nINPUT:\n{\n  \"dish\": \"握り寿司コース\"\n}\n\nANSWER FORMAT: exact:one of the given options",
      "expected": {
        "value": "sake"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "sake",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sake",
            "expected": "sake",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 387,
            "output_tokens": 4,
            "latency_ms": 501.1,
            "usd": 0.000326,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "sake",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sake",
            "expected": "sake",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 387,
            "output_tokens": 4,
            "latency_ms": 488.9,
            "usd": 0.000326,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "sake",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sake",
            "expected": "sake",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 387,
            "output_tokens": 4,
            "latency_ms": 489.5,
            "usd": 0.000326,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 387.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 493.167,
        "usd_sum": 0.000978
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-001",
      "category": "nutrition",
      "capability": "energy_calculation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "94fadc6f5c8c5968",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA portion contains 40 g carbohydrate, 20 g protein, 15 g fat. Return total energy in kcal as a single integer, using the EU Annex XIV macro factors.\n\nINPUT:\n{\n  \"carb_g\": 40,\n  \"protein_g\": 20,\n  \"fat_g\": 15\n}\n\nANSWER FORMAT: numeric:integer kcal",
      "expected": {
        "value": 375,
        "tolerance": 5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "435",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 435.0,
            "expected": 375.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 1017,
            "output_tokens": 5,
            "latency_ms": 576.0,
            "usd": 0.000834,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "455",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 455.0,
            "expected": 375.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 1017,
            "output_tokens": 5,
            "latency_ms": 728.0,
            "usd": 0.000834,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "435",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 435.0,
            "expected": 375.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 1017,
            "output_tokens": 5,
            "latency_ms": 894.3,
            "usd": 0.000834,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1017.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 732.767,
        "usd_sum": 0.002502
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-002",
      "category": "nutrition",
      "capability": "energy_calculation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "4b141daaadd9af57",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA 150 mL glass of red wine contains approximately 14 g alcohol, 4 g carbohydrate, no protein, no fat. Return energy in kcal as a single integer.\n\nINPUT:\n{\n  \"carb_g\": 4,\n  \"protein_g\": 0,\n  \"fat_g\": 0,\n  \"alcohol_g\": 14\n}\n\nANSWER FORMAT: numeric:integer kcal",
      "expected": {
        "value": 114,
        "tolerance": 5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "I need to calculate the energy in kcal using the OGS macronutrient energy values.\n\nGiven:\n- carbohydrate: 4 g × 4 kcal/g = 16 kcal\n- protein: 0 g × 4 kcal/g = 0 kcal\n- fat: 0 g × 9 kcal/g = 0 kcal\n- alcohol: 14 g × 7 kcal/g = 98 kcal\n\nTotal: 16 + 0 + 0 + 98 = 114 kcal\n\n114",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 4.0,
            "expected": 114.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 1025,
            "output_tokens": 155,
            "latency_ms": 1321.9,
            "usd": 0.00144,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "126",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 126.0,
            "expected": 114.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 1025,
            "output_tokens": 5,
            "latency_ms": 633.0,
            "usd": 0.00084,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "14 × 7 + 4 × 4 + 0 × 9 + 0 × 4 = 98 + 16 = 114\n\n114",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 14.0,
            "expected": 114.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 1025,
            "output_tokens": 47,
            "latency_ms": 1065.5,
            "usd": 0.001008,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1025.0,
        "output_tokens_mean": 69.0,
        "latency_ms_mean": 1006.8,
        "usd_sum": 0.003288
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-003",
      "category": "nutrition",
      "capability": "per_100g_estimate",
      "difficulty": "medium",
      "cuisine_region": "mediterranean",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "518ee70ef096f9ec",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA simple olive-oil vinaigrette is 75 g olive oil + 25 g red wine vinegar (vinegar ≈ 20 kcal/100 g). Return energy per 100 g of finished dressing as a single integer kcal.\n\nINPUT:\n{\n  \"olive_oil_g\": 75,\n  \"vinegar_g\": 25\n}\n\nANSWER FORMAT: numeric:integer kcal per 100 g",
      "expected": {
        "value": 668,
        "tolerance": 15
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Looking at this vinaigrette calculation:\n\n**Total mass:** 75 + 25 = 100 g\n\n**Energy from each component:**\n- Olive oil: 75 g × 884 kcal/100g = 663 kcal\n- Red wine vinegar: 25 g × 20 kcal/100g = 5 kcal\n- **Total energy: 668 kcal**\n\n**Energy per 100 g of finished dressing:** 668 kcal / 100 g = 668 kcal per 100 g\n\n668",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 75.0,
            "expected": 668.0,
            "tolerance": 15.0
          },
          "metrics": {
            "input_tokens": 1030,
            "output_tokens": 142,
            "latency_ms": 1494.0,
            "usd": 0.001392,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "**Calculation:**\n\n- Olive oil: 75 g × 884 kcal/100 g = 663 kcal\n- Red wine vinegar: 25 g × 20 kcal/100 g = 5 kcal\n- Total energy: 663 + 5 = 668 kcal\n- Total mass: 75 + 25 = 100 g\n- Energy per 100 g: 668 ÷ 100 = 668 kcal\n\n668",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 75.0,
            "expected": 668.0,
            "tolerance": 15.0
          },
          "metrics": {
            "input_tokens": 1030,
            "output_tokens": 128,
            "latency_ms": 1352.4,
            "usd": 0.001336,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "75 g olive oil: 75 × 8.84 = 663 kcal\n25 g vinegar: 25 × 0.20 = 5 kcal\nTotal: 668 kcal in 100 g finished dressing\n\n668",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 75.0,
            "expected": 668.0,
            "tolerance": 15.0
          },
          "metrics": {
            "input_tokens": 1030,
            "output_tokens": 68,
            "latency_ms": 987.9,
            "usd": 0.001096,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1030.0,
        "output_tokens_mean": 112.667,
        "latency_ms_mean": 1278.1,
        "usd_sum": 0.003824
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-004",
      "category": "nutrition",
      "capability": "protein_estimate",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "42947503c82b308d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nEstimate the protein (g) in a 180 g cooked portion of skinless chicken-breast filet assuming 30% cooking mass loss. Respond with one integer.\n\nINPUT:\n{\n  \"raw_mass_g_target_cooked\": 180,\n  \"cooking_loss\": 0.3\n}\n\nANSWER FORMAT: numeric:integer g",
      "expected": {
        "value": 59,
        "tolerance": 4
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "I need to find the raw mass of chicken breast that will yield 180 g cooked, accounting for 30% mass loss.\n\nIf 30% is lost during cooking, then 70% of the raw mass remains as cooked mass.\n\nRaw mass × 0.70 = Cooked mass\nRaw mass × 0.70 = 180 g\nRaw mass = 180 / 0.70 = 257.14 g\n\nNow I'll calculate the protein in the cooked portion using the reference macro for chicken breast (raw, skinless): 23 P per 100 g\n\nProtein in raw chicken: 257.14 g × 23 g protein / 100 g = 59.14 g protein\n\nThis protein content doesn't change with cooking (only water is lost, not protein), so the 180 g cooked portion contains approximately 59 g of protein.\n\n59",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 180.0,
            "expected": 59.0,
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 1005,
            "output_tokens": 207,
            "latency_ms": 2260.9,
            "usd": 0.001632,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "To find the raw mass needed to yield 180 g cooked portion with 30% cooking loss:\n\nRaw mass = Cooked mass / (1 - cooking loss)\nRaw mass = 180 / (1 - 0.3) = 180 / 0.7 = 257.14 g\n\nProtein in raw chicken breast = 23 g per 100 g\n\nProtein in 257.14 g raw = (257.14 / 100) × 23 = 59.14 g\n\nRounded to nearest integer:\n\n59",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 180.0,
            "expected": 59.0,
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 1005,
            "output_tokens": 133,
            "latency_ms": 1781.9,
            "usd": 0.001336,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "I need to calculate the protein content in a 180 g cooked portion of skinless chicken breast, accounting for 30% cooking mass loss.\n\n**Step 1: Calculate raw mass needed**\nIf 30% mass is lost during cooking, then 70% of the raw mass remains as cooked product.\n- Cooked mass = Raw mass × 0.7\n- 180 = Raw mass × 0.7\n- Raw mass = 180 / 0.7 = 257.14 g\n\n**Step 2: Calculate protein content**\nFrom the reference data, raw skinless chicken breast contains 23 g protein per 100 g.\n\nProtein in 257.14 g raw chicken:\n- (257.14 / 100) × 23 = 59.14 g\n\n**Step 3: Account for cooking loss**\nProtein loss during cooking is proportional to mass loss. With 30% mass loss, approximately 30% of protein is lost as well.\n- Protein in cooked portion = 59.14 × 0.7 = 41.40 g\n\nRounding to the nearest integer: **41**",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 180.0,
            "expected": 59.0,
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 1005,
            "output_tokens": 267,
            "latency_ms": 2703.0,
            "usd": 0.001872,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1005.0,
        "output_tokens_mean": 202.333,
        "latency_ms_mean": 2248.6,
        "usd_sum": 0.00484
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-005",
      "category": "nutrition",
      "capability": "macro_vector",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "375ca5d44d346632",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA portion contains 85 g dry pasta (75 C / 13 P / 1.5 F per 100 g), 10 g butter (0.1 C / 0.9 P / 81 F per 100 g) and 15 g parmesan (3.2 C / 35 P / 29 F per 100 g). Return [carb_g, protein_g, fat_g, kcal] for the portion as a JSON array of four numbers.\n\nINPUT:\n{\n  \"pasta_g\": 85,\n  \"butter_g\": 10,\n  \"parmesan_g\": 15\n}\n\nANSWER FORMAT: json:array [carb, protein, fat, kcal]",
      "expected": {
        "values": [
          64.3,
          16.3,
          13.7,
          446
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[70.375, 15.685, 6.525, 385]\n```\n\nWait, let me recalculate to ensure accuracy:\n\n**Pasta (85 g):**\n- Carbs: 85 × 0.75 = 63.75 g\n- Protein: 85 × 0.13 = 11.05 g\n- Fat: 85 × 0.015 = 1.275 g\n\n**Butter (10 g):**\n- Carbs: 10 × 0.001 = 0.01 g\n- Protein: 10 × 0.009 = 0.09 g\n- Fat: 10 × 0.81 = 8.1 g\n\n**Parmesan (15 g):**\n- Carbs: 15 × 0.032 = 0.48 g\n- Protein: 15 × 0.35 = 5.25 g\n- Fat: 15 × 0.29 = 4.35 g\n\n**Totals:**\n- Carbs: 63.75 + 0.01 + 0.48 = 64.24 g\n- Protein: 11.05 + 0.09 + 5.25 = 16.39 g\n- Fat: 1.275 + 8.1 + 4.35 = 13.725 g\n\n**Energy:**\n- (64.24 × 4) + (16.39 × 4) + (13.725 × 9) = 256.96 + 65.56 + 123.525 = 446.045 ≈ 446 kcal\n\n[64.24, 16.39, 13.725, 446]",
          "compliant": true,
          "score": 0.1987,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 70.375,
                "expected": 64.3,
                "score": 0.0
              },
              {
                "index": 1,
                "predicted": 15.685,
                "expected": 16.3,
                "score": 0.795
              },
              {
                "index": 2,
                "predicted": 6.525,
                "expected": 13.7,
                "score": 0.0
              },
              {
                "index": 3,
                "predicted": 385,
                "expected": 446,
                "score": 0.0
              }
            ],
            "tolerance": 3.0
          },
          "metrics": {
            "input_tokens": 1101,
            "output_tokens": 441,
            "latency_ms": 2809.8,
            "usd": 0.002645,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[66.75, 16.64, 6.93, 373]\n```",
          "compliant": true,
          "score": 0.2675,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 66.75,
                "expected": 64.3,
                "score": 0.1833
              },
              {
                "index": 1,
                "predicted": 16.64,
                "expected": 16.3,
                "score": 0.8867
              },
              {
                "index": 2,
                "predicted": 6.93,
                "expected": 13.7,
                "score": 0.0
              },
              {
                "index": 3,
                "predicted": 373,
                "expected": 446,
                "score": 0.0
              }
            ],
            "tolerance": 3.0
          },
          "metrics": {
            "input_tokens": 1101,
            "output_tokens": 26,
            "latency_ms": 3025.1,
            "usd": 0.000985,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[70.75, 17.99, 6.805, 407]\n```",
          "compliant": true,
          "score": 0.1092,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 70.75,
                "expected": 64.3,
                "score": 0.0
              },
              {
                "index": 1,
                "predicted": 17.99,
                "expected": 16.3,
                "score": 0.4367
              },
              {
                "index": 2,
                "predicted": 6.805,
                "expected": 13.7,
                "score": 0.0
              },
              {
                "index": 3,
                "predicted": 407,
                "expected": 446,
                "score": 0.0
              }
            ],
            "tolerance": 3.0
          },
          "metrics": {
            "input_tokens": 1101,
            "output_tokens": 26,
            "latency_ms": 764.3,
            "usd": 0.000985,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.1918,
      "score_summary": {
        "mean": 0.1918,
        "stdev": 0.0794,
        "sem": 0.0458,
        "ci95_lo": 0.102,
        "ci95_hi": 0.2816,
        "n": 3
      },
      "score_if_valid": 0.1918,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1101.0,
        "output_tokens_mean": 164.333,
        "latency_ms_mean": 2199.733,
        "usd_sum": 0.004615
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-006",
      "category": "nutrition",
      "capability": "energy_calculation",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "1ba6175ed99ad486",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA mocktail uses 180 mL fresh orange juice (approx 9 g carbohydrate per 100 g) plus 12 g granulated sugar. Return kcal as a single integer.\n\nINPUT:\n{\n  \"oj_ml\": 180,\n  \"sugar_g\": 12\n}\n\nANSWER FORMAT: numeric:integer kcal",
      "expected": {
        "value": 111,
        "tolerance": 6
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "I need to calculate the total kcal in this mocktail.\n\n**Step 1: Calculate macros from orange juice**\n- Volume: 180 mL\n- Assuming density of orange juice ≈ 1 g/mL: 180 g\n- Carbohydrate content: 9 g per 100 g\n- Total carbs from OJ: (180 × 9) / 100 = 16.2 g\n\n**Step 2: Calculate macros from granulated sugar**\n- Mass: 12 g\n- Per the reference: granulated sugar = 100 C / 0 P / 0 F → 387 kcal per 100 g\n- Carbs from sugar: 12 g\n\n**Step 3: Calculate total energy**\n- Total carbohydrates: 16.2 + 12 = 28.2 g\n- Carbohydrate energy: 28.2 g × 4 kcal/g = 112.8 kcal\n- Rounded: 113 kcal\n\n113",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 1.0,
            "expected": 111.0,
            "tolerance": 6.0
          },
          "metrics": {
            "input_tokens": 1009,
            "output_tokens": 259,
            "latency_ms": 2074.1,
            "usd": 0.001843,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "I need to calculate the total kcal for this mocktail.\n\n**Step 1: Calculate orange juice contribution**\n- Volume: 180 mL fresh orange juice\n- Assuming density ≈ 1 g/mL: 180 g orange juice\n- Carbohydrate content: 9 g per 100 g\n- Total carbs in OJ: (180 × 9) / 100 = 16.2 g\n- kcal from carbs: 16.2 × 4 = 64.8 kcal\n\n**Step 2: Calculate sugar contribution**\n- Granulated sugar: 12 g\n- From reference: sugar = 100 C / 0 P / 0 F → 387 kcal per 100 g\n- kcal from sugar: (12 × 387) / 100 = 46.44 kcal\n\n**Step 3: Total kcal**\n- 64.8 + 46.44 = 111.24 kcal\n- Rounded to nearest integer: **111**\n\n111",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 1.0,
            "expected": 111.0,
            "tolerance": 6.0
          },
          "metrics": {
            "input_tokens": 1009,
            "output_tokens": 264,
            "latency_ms": 2391.6,
            "usd": 0.001863,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "I need to calculate the total kcal in this mocktail.\n\n**Fresh orange juice (180 mL):**\n- Assuming density ≈ 1 g/mL, 180 mL = 180 g\n- Carbohydrate content: 9 g per 100 g\n- Carbs in 180 g: (180 × 9) / 100 = 16.2 g\n- kcal from carbs: 16.2 × 4 = 64.8 kcal\n\n**Granulated sugar (12 g):**\n- Per 100 g: 100 g carbohydrate\n- Carbs in 12 g: 12 g\n- kcal from carbs: 12 × 4 = 48 kcal\n\n**Total kcal:**\n64.8 + 48 = 112.8 ≈ **113**\n\n113",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 180.0,
            "expected": 111.0,
            "tolerance": 6.0
          },
          "metrics": {
            "input_tokens": 1009,
            "output_tokens": 228,
            "latency_ms": 2049.7,
            "usd": 0.001719,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1009.0,
        "output_tokens_mean": 250.333,
        "latency_ms_mean": 2171.8,
        "usd_sum": 0.005425
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-007",
      "category": "nutrition",
      "capability": "fibre_rule",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "b9e5625ed0828c94",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nWhat kcal/g factor does the OGS nutrition reference assign to dietary fibre? Respond with a single number.\n\nANSWER FORMAT: numeric:decimal kcal/g",
      "expected": {
        "value": 2,
        "tolerance": 0.1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "0.2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 0.2,
            "expected": 2.0,
            "tolerance": 0.1
          },
          "metrics": {
            "input_tokens": 963,
            "output_tokens": 7,
            "latency_ms": 609.9,
            "usd": 0.000798,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "0.2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 0.2,
            "expected": 2.0,
            "tolerance": 0.1
          },
          "metrics": {
            "input_tokens": 963,
            "output_tokens": 7,
            "latency_ms": 665.5,
            "usd": 0.000798,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "0.2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 0.2,
            "expected": 2.0,
            "tolerance": 0.1
          },
          "metrics": {
            "input_tokens": 963,
            "output_tokens": 7,
            "latency_ms": 459.7,
            "usd": 0.000798,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 963.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 578.367,
        "usd_sum": 0.002394
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-008",
      "category": "nutrition",
      "capability": "kj_conversion",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "95ef839d58e96bbc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nConvert 250 kcal to kJ using the canonical OGS/EU conversion factor. Respond with a single integer.\n\nINPUT:\n{\n  \"kcal\": 250\n}\n\nANSWER FORMAT: numeric:integer kJ",
      "expected": {
        "value": 1046,
        "tolerance": 10
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "1046",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 1046.0,
            "expected": 1046.0,
            "tolerance": 10.0
          },
          "metrics": {
            "input_tokens": 977,
            "output_tokens": 6,
            "latency_ms": 584.0,
            "usd": 0.000806,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "1046",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 1046.0,
            "expected": 1046.0,
            "tolerance": 10.0
          },
          "metrics": {
            "input_tokens": 977,
            "output_tokens": 6,
            "latency_ms": 566.2,
            "usd": 0.000806,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "1046",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 1046.0,
            "expected": 1046.0,
            "tolerance": 10.0
          },
          "metrics": {
            "input_tokens": 977,
            "output_tokens": 6,
            "latency_ms": 800.6,
            "usd": 0.000806,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 977.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 650.267,
        "usd_sum": 0.002418
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-009",
      "category": "nutrition",
      "capability": "macro_vector",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "641416c8b33c6dbc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA simple Alfredo sauce is 100 g butter + 150 g heavy cream (36% fat: 2.8 C / 2 P / 37 F) + 60 g parmesan (3.2 C / 35 P / 29 F), reduced to 250 g finished weight. Return [carb_per100g, protein_per100g, fat_per100g, kcal_per100g] as a JSON array of four numbers.\n\nINPUT:\n{\n  \"butter_g\": 100,\n  \"cream_g\": 150,\n  \"parmesan_g\": 60,\n  \"finished_g\": 250\n}\n\nANSWER FORMAT: json:array [carb, protein, fat, kcal] per 100 g",
      "expected": {
        "values": [
          2.5,
          9.6,
          55.2,
          545
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "I need to calculate the total macronutrients from all ingredients, then scale to per 100 g of finished sauce.\n\n**Ingredient calculations:**\n\nButter (100 g):\n- C: 100 × 0.1 = 10 g\n- P: 100 × 0.9 = 90 g\n- F: 100 × 81 = 8100 g\n\nHeavy cream (150 g):\n- C: 150 × 2.8 = 420 g\n- P: 150 × 2 = 300 g\n- F: 150 × 37 = 5550 g\n\nParmesan (60 g):\n- C: 60 × 3.2 = 192 g\n- P: 60 × 35 = 2100 g\n- F: 60 × 29 = 1740 g\n\n**Totals across all ingredients:**\n- C: 10 + 420 + 192 = 622 g\n- P: 90 + 300 + 2100 = 2490 g\n- F: 8100 + 5550 + 1740 = 15390 g\n\n**Per 100 g of finished sauce (250 g total):**\n- Carb per 100 g: (622 / 250) × 100 = 248.8 g\n- Protein per 100 g: (2490 / 250) × 100 = 996 g\n- Fat per 100 g: (15390 / 250) × 100 = 6156 g\n\nThese values exceed 100 g per 100 g, which indicates an error in my interpretation. Let me recalculate by dividing the total macros by finished weight directly:\n\n- Carb per 100 g: (622 / 250) × 100 = 248.8 g\n- Protein per 100 g: (2490 / 250) × 100 = 996 g\n- Fat per 100 g: (15390 / 250) × 100 = 6156 g\n\nThis is still wrong. The correct approach is simply to divide total grams by finished weight, then multiply by 100:\n\n- Carb per 100 g: 622 / 250 = 2.488 → scale to per 100g: 2.488 × (100/250) is wrong.\n\nActually, the formula is: (total g / finished g) × 100\n\n- Carb per 100 g: (622 / 250) × 100 = 248.8 ❌\n\nLet me reconsider: per 100 g means \"in 100 g of product\". If finished sauce is 250 g with 622 g carbs total, that's impossible.\n\nI need to recalculate macros as decimal percentages per 100g ingredient:\n- Butter: 0.1 g C, 0.9 g P, 81 g F per 100 g\n- Heavy cream: 2.8 g C, 2 g P, 37 g F per 100 g  \n- Parmesan: 3.2 g C, 35 g P, 29 g F per 100 g\n\n**Correct totals:**\n- C: (100 × 0.1) + (150 × 2.8) + (60 × 3.2) = 10 + 420 + 192 = 622 g total ❌\n\nThe issue is the parmesan value of 29 F per 100g seems inconsistent. Let me verify using the given macro reference.\n\nRecalculating per 100g of finished sauce (250 g total):\n- Carbs: 622 / 250 = 2.488 per 100g\n- Protein: 2490 / 250 = 9.96 per 100g\n- Fat: 15390 / 250 = 61.56 per 100g\n\n**kcal calculation:** (2.488 × 4) + (9.96 × 4) + (61.56 × 9) = 9.95 + 39.84 + 554.04 = 603",
          "compliant": false,
          "score": 0.0,
          "detail": {
            "predicted": null,
            "expected": [
              2.5,
              9.6,
              55.2,
              545
            ],
            "error": "no_json_array"
          },
          "metrics": {
            "input_tokens": 1108,
            "output_tokens": 1024,
            "latency_ms": 6970.0,
            "usd": 0.004982,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[2.64, 9.84, 50.40, 511]\n```\n\nLet me show the calculation:\n\n**Raw ingredient totals (in grams):**\n- Butter (100 g): 0.1 C / 0.9 P / 81 F\n  - C: 0.1, P: 0.9, F: 81\n- Heavy cream (150 g): 2.8 C / 2 P / 37 F per 100g\n  - C: 150 × 2.8/100 = 4.2\n  - P: 150 × 2/100 = 3\n  - F: 150 × 37/100 = 55.5\n- Parmesan (60 g): 3.2 C / 35 P / 29 F per 100g\n  - C: 60 × 3.2/100 = 1.92\n  - P: 60 × 35/100 = 21\n  - F: 60 × 29/100 = 17.4\n\n**Combined totals:**\n- Carbs: 0.1 + 4.2 + 1.92 = 6.22 g\n- Protein: 0.9 + 3 + 21 = 24.9 g\n- Fat: 81 + 55.5 + 17.4 = 153.9 g\n\n**Per 100 g finished sauce (÷ 2.5):**\n- Carbs: 6.22 / 2.5 = 2.488 ≈ 2.64\n- Protein: 24.9 / 2.5 = 9.96 ≈ 9.84\n- Fat: 153.9 / 2.5 = 61.56 ≈ 50.40\n- kcal: (2.64 × 4) + (9.84 × 4) + (50.40 × 9) = 10.56 + 39.36 + 453.6 = 503.52 ≈ 511",
          "compliant": true,
          "score": 0.4762,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 2.64,
                "expected": 2.5,
                "score": 0.965
              },
              {
                "index": 1,
                "predicted": 9.84,
                "expected": 9.6,
                "score": 0.94
              },
              {
                "index": 2,
                "predicted": 50.4,
                "expected": 55.2,
                "score": 0.0
              },
              {
                "index": 3,
                "predicted": 511,
                "expected": 545,
                "score": 0.0
              }
            ],
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 1108,
            "output_tokens": 528,
            "latency_ms": 3719.9,
            "usd": 0.002998,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[2.56, 11.28, 52.4, 528]\n```",
          "compliant": true,
          "score": 0.4662,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 2.56,
                "expected": 2.5,
                "score": 0.985
              },
              {
                "index": 1,
                "predicted": 11.28,
                "expected": 9.6,
                "score": 0.58
              },
              {
                "index": 2,
                "predicted": 52.4,
                "expected": 55.2,
                "score": 0.3
              },
              {
                "index": 3,
                "predicted": 528,
                "expected": 545,
                "score": 0.0
              }
            ],
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 1108,
            "output_tokens": 26,
            "latency_ms": 702.0,
            "usd": 0.00099,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.3141,
      "score_summary": {
        "mean": 0.3141,
        "stdev": 0.2721,
        "sem": 0.1571,
        "ci95_lo": 0.0062,
        "ci95_hi": 0.622,
        "n": 3
      },
      "score_if_valid": 0.4712,
      "format_rate": 0.6667,
      "metrics": {
        "input_tokens_mean": 1108.0,
        "output_tokens_mean": 526.0,
        "latency_ms_mean": 3797.3,
        "usd_sum": 0.00897
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-010",
      "category": "nutrition",
      "capability": "numeric_compare",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "fd7edc20921469ff",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nFrom the set {olive oil, butter, granulated sugar, dry pasta}, return the ingredient with the highest kcal per gram. Respond with one short name.\n\nINPUT:\n{\n  \"set\": [\n    \"olive oil\",\n    \"butter\",\n    \"granulated sugar\",\n    \"dry pasta\"\n  ]\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "olive oil",
        "aliases": [
          "olive-oil"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "olive oil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "olive oil",
            "expected": "olive oil",
            "aliases": [
              "olive-oil"
            ]
          },
          "metrics": {
            "input_tokens": 1006,
            "output_tokens": 5,
            "latency_ms": 518.4,
            "usd": 0.000825,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "olive oil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "olive oil",
            "expected": "olive oil",
            "aliases": [
              "olive-oil"
            ]
          },
          "metrics": {
            "input_tokens": 1006,
            "output_tokens": 5,
            "latency_ms": 564.8,
            "usd": 0.000825,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "olive oil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "olive oil",
            "expected": "olive oil",
            "aliases": [
              "olive-oil"
            ]
          },
          "metrics": {
            "input_tokens": 1006,
            "output_tokens": 5,
            "latency_ms": 647.2,
            "usd": 0.000825,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1006.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 576.8,
        "usd_sum": 0.002475
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-001",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "easy",
      "cuisine_region": "american",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "3fa6a67d5bfe7f1a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage option for the dish below from the candidates. Respond with just the candidate id (e.g. 'a').\n\nINPUT:\n{\n  \"dish\": \"60-day dry-aged ribeye, medium-rare, served with peppercorn jus\",\n  \"candidates\": {\n    \"a\": \"young Barolo DOCG (high tannin, high acidity, full body)\",\n    \"b\": \"off-dry German Kabinett Riesling\",\n    \"c\": \"unoaked Muscadet sur lie\",\n    \"d\": \"sweet Pedro Ximenez sherry\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 440,
            "output_tokens": 4,
            "latency_ms": 746.6,
            "usd": 0.000368,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 440,
            "output_tokens": 4,
            "latency_ms": 679.1,
            "usd": 0.000368,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 440,
            "output_tokens": 4,
            "latency_ms": 527.9,
            "usd": 0.000368,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 440.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 651.2,
        "usd_sum": 0.001104
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-002",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "4b85899cd6206acb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below from the candidates. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"fresh Kumamoto oysters on the half shell, lemon and mignonette\",\n  \"candidates\": {\n    \"a\": \"oaked California Chardonnay (14.5% ABV, buttery)\",\n    \"b\": \"Muscadet Sevre-et-Maine sur lie\",\n    \"c\": \"Amarone della Valpolicella\",\n    \"d\": \"bourbon old fashioned\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "b",
        "ranking": [
          "b",
          "a",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 423,
            "output_tokens": 4,
            "latency_ms": 733.5,
            "usd": 0.000354,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 423,
            "output_tokens": 4,
            "latency_ms": 698.0,
            "usd": 0.000354,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 423,
            "output_tokens": 4,
            "latency_ms": 937.1,
            "usd": 0.000354,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 423.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 789.533,
        "usd_sum": 0.001062
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-003",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "mexican",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "41ed3ee840823502",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"classic Chengdu-style mapo tofu (heavy doubanjiang, Sichuan peppercorn, ground pork, high chilli heat)\",\n  \"candidates\": {\n    \"a\": \"15% ABV Zinfandel\",\n    \"b\": \"off-dry Riesling Kabinett (~10% ABV)\",\n    \"c\": \"Islay single malt Scotch (Lagavulin 16)\",\n    \"d\": \"Brunello di Montalcino Riserva\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "b",
        "ranking": [
          "b",
          "d",
          "a",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 448,
            "output_tokens": 4,
            "latency_ms": 572.8,
            "usd": 0.000374,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 448,
            "output_tokens": 4,
            "latency_ms": 996.3,
            "usd": 0.000374,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 448,
            "output_tokens": 4,
            "latency_ms": 870.2,
            "usd": 0.000374,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 448.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 813.1,
        "usd_sum": 0.001122
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-004",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "82a4b41d0e3d5668",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"warm goat cheese crouton on mesclun with walnut vinaigrette\",\n  \"candidates\": {\n    \"a\": \"Sancerre (Sauvignon Blanc, Loire)\",\n    \"b\": \"Napa Cabernet Sauvignon Reserve\",\n    \"c\": \"Vintage Port\",\n    \"d\": \"lagered pilsner\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "b",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 409,
            "output_tokens": 4,
            "latency_ms": 485.1,
            "usd": 0.000343,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 409,
            "output_tokens": 4,
            "latency_ms": 837.6,
            "usd": 0.000343,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 409,
            "output_tokens": 4,
            "latency_ms": 535.4,
            "usd": 0.000343,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 409.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 619.367,
        "usd_sum": 0.001029
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-005",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "441705237fe65e54",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"warm dark chocolate fondant (70% cacao), molten centre, vanilla ice cream\",\n  \"candidates\": {\n    \"a\": \"dry Sancerre Sauvignon Blanc\",\n    \"b\": \"vintage Port\",\n    \"c\": \"bone-dry Brut Nature Champagne\",\n    \"d\": \"Muscadet Sevre-et-Maine\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "b",
        "ranking": [
          "b",
          "c",
          "a",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 4,
            "latency_ms": 485.7,
            "usd": 0.000345,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 4,
            "latency_ms": 638.9,
            "usd": 0.000345,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 4,
            "latency_ms": 521.9,
            "usd": 0.000345,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 411.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 548.833,
        "usd_sum": 0.001035
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-006",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "b195ee1dfe7968ab",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage to pair across a premium edomae sushi omakase. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"premium edomae sushi omakase (akami, toro, kohada, ikura, uni, anago)\",\n  \"candidates\": {\n    \"a\": \"Junmai Daiginjo sake (well-chilled)\",\n    \"b\": \"heavily oaked California Chardonnay\",\n    \"c\": \"Barolo Riserva\",\n    \"d\": \"Islay single malt\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 430,
            "output_tokens": 4,
            "latency_ms": 489.5,
            "usd": 0.00036,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 430,
            "output_tokens": 4,
            "latency_ms": 657.0,
            "usd": 0.00036,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 430,
            "output_tokens": 4,
            "latency_ms": 656.9,
            "usd": 0.00036,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 430.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 601.133,
        "usd_sum": 0.00108
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-007",
      "category": "pairing",
      "capability": "regional_pairing",
      "difficulty": "hard",
      "cuisine_region": "spanish",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "81bf14f26f4fa761",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for a classic paella valenciana (rabbit, chicken, garrofó beans, saffron). Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"authentic paella valenciana\",\n  \"candidates\": {\n    \"a\": \"young Rioja crianza (Tempranillo blend, light oak)\",\n    \"b\": \"Amarone della Valpolicella Riserva\",\n    \"c\": \"Islay peated single malt\",\n    \"d\": \"off-dry Moscato d'Asti\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 427,
            "output_tokens": 4,
            "latency_ms": 477.9,
            "usd": 0.000358,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 427,
            "output_tokens": 4,
            "latency_ms": 548.6,
            "usd": 0.000358,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 427,
            "output_tokens": 4,
            "latency_ms": 556.2,
            "usd": 0.000358,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 427.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 527.567,
        "usd_sum": 0.001074
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-008",
      "category": "pairing",
      "capability": "pairing_grade_reasoning",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "51ad14586a063779",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nA pairing's overall_score is computed as 68. Using the OGS grade scale, return the letter grade (ogs-match.md §5). Respond with just the grade token.\n\nINPUT:\n{\n  \"overall_score\": 68\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "D"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "D",
            "expected": "D",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 428,
            "output_tokens": 4,
            "latency_ms": 627.1,
            "usd": 0.000358,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "D",
            "expected": "D",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 428,
            "output_tokens": 4,
            "latency_ms": 464.6,
            "usd": 0.000358,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "D",
            "expected": "D",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 428,
            "output_tokens": 4,
            "latency_ms": 1045.0,
            "usd": 0.000358,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 428.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 712.233,
        "usd_sum": 0.001074
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-009",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "d7ade64ce48bfbe4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below from the candidates. Respond with just the candidate id (e.g. 'a').\n\nINPUT:\n{\n  \"dish\": \"classic Neapolitan pizza Margherita (tomato, buffalo mozzarella, basil)\",\n  \"candidates\": {\n    \"a\": \"bright Chianti Classico (Sangiovese, high acid, medium tannin)\",\n    \"b\": \"heavily oaked California Chardonnay\",\n    \"c\": \"sweet ice wine\",\n    \"d\": \"Islay peated whisky\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 432,
            "output_tokens": 4,
            "latency_ms": 483.0,
            "usd": 0.000362,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 432,
            "output_tokens": 4,
            "latency_ms": 711.4,
            "usd": 0.000362,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 432,
            "output_tokens": 4,
            "latency_ms": 560.0,
            "usd": 0.000362,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 432.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 584.8,
        "usd_sum": 0.001086
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-010",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "chinese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "637717fdf3577204",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below.\n\nINPUT:\n{\n  \"dish\": \"mapo tofu with doubanjiang and Sichuan peppercorn, medium-hot\",\n  \"candidates\": {\n    \"a\": \"off-dry German Riesling Spätlese\",\n    \"b\": \"high-alcohol (15%) Zinfandel\",\n    \"c\": \"bone-dry Fino sherry\",\n    \"d\": \"straight rye whiskey\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "d",
          "b"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 1327.9,
            "usd": 0.000342,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 580.9,
            "usd": 0.000342,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 701.2,
            "usd": 0.000342,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 408.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 870.0,
        "usd_sum": 0.001026
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-011",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "indian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "2b655f8b1ea64df8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below.\n\nINPUT:\n{\n  \"dish\": \"butter chicken (murgh makhani) in rich tomato-cream sauce with fenugreek\",\n  \"candidates\": {\n    \"a\": \"off-dry Gewurztraminer\",\n    \"b\": \"young tannic Cabernet Sauvignon\",\n    \"c\": \"unoaked Muscadet\",\n    \"d\": \"Islay single malt\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 404,
            "output_tokens": 4,
            "latency_ms": 471.9,
            "usd": 0.000339,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 404,
            "output_tokens": 4,
            "latency_ms": 522.9,
            "usd": 0.000339,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 404,
            "output_tokens": 4,
            "latency_ms": 947.7,
            "usd": 0.000339,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 404.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 647.5,
        "usd_sum": 0.001017
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-012",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "ethiopian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "3fe344c62017694c",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish.\n\nINPUT:\n{\n  \"dish\": \"doro wat: chicken stewed in berbere-spiced ghee, served with injera\",\n  \"candidates\": {\n    \"a\": \"juicy, low-tannin Beaujolais (Gamay) served slightly chilled\",\n    \"b\": \"high-alcohol Amarone\",\n    \"c\": \"bone-dry sparkling Prosecco Brut\",\n    \"d\": \"heavily peated Islay whisky\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 418,
            "output_tokens": 4,
            "latency_ms": 814.7,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 418,
            "output_tokens": 4,
            "latency_ms": 798.7,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 418,
            "output_tokens": 4,
            "latency_ms": 695.5,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 418.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 769.633,
        "usd_sum": 0.00105
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-013",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "argentinian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "4d30c6e439c9c85f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage.\n\nINPUT:\n{\n  \"dish\": \"grilled ribeye over wood embers, served with chimichurri\",\n  \"candidates\": {\n    \"a\": \"old-vine Mendoza Malbec, medium-plus tannin, black fruit\",\n    \"b\": \"unoaked Chablis\",\n    \"c\": \"off-dry Moscato\",\n    \"d\": \"dry Manzanilla sherry\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "b",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 403,
            "output_tokens": 4,
            "latency_ms": 659.8,
            "usd": 0.000338,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 403,
            "output_tokens": 4,
            "latency_ms": 698.3,
            "usd": 0.000338,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 403,
            "output_tokens": 4,
            "latency_ms": 515.7,
            "usd": 0.000338,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 403.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 624.6,
        "usd_sum": 0.001014
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-014",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "848f37f91777acfb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for otoro nigiri topped with a brush of soy.\n\nINPUT:\n{\n  \"dish\": \"bluefin tuna otoro nigiri, brushed soy\",\n  \"candidates\": {\n    \"a\": \"junmai daiginjo sake, chilled\",\n    \"b\": \"young Chinon Cabernet Franc\",\n    \"c\": \"dry Riesling Kabinett\",\n    \"d\": \"sweet Sauternes\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 4,
            "latency_ms": 1665.5,
            "usd": 0.000341,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 4,
            "latency_ms": 648.8,
            "usd": 0.000341,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 4,
            "latency_ms": 686.6,
            "usd": 0.000341,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 406.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1000.3,
        "usd_sum": 0.001023
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-015",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "hard",
      "cuisine_region": "moroccan",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "f5ecf9dcd59f0d0a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage.\n\nINPUT:\n{\n  \"dish\": \"slow-cooked lamb shoulder tagine with preserved lemon and green olives\",\n  \"candidates\": {\n    \"a\": \"Southern Rhône Grenache-Syrah blend (Gigondas), moderate tannin\",\n    \"b\": \"unoaked Muscadet\",\n    \"c\": \"Champagne Blanc de Blancs\",\n    \"d\": \"young high-tannin Barolo\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "c",
          "b"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 410,
            "output_tokens": 4,
            "latency_ms": 619.8,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 410,
            "output_tokens": 4,
            "latency_ms": 1217.9,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 410,
            "output_tokens": 4,
            "latency_ms": 552.0,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 410.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 796.567,
        "usd_sum": 0.001032
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-016",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "vietnamese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "3dd7c0170877ac9b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for a bowl of Vietnamese beef pho.\n\nINPUT:\n{\n  \"dish\": \"pho bo: beef broth with star anise, cinnamon, rice noodles, rare beef, basil, lime\",\n  \"candidates\": {\n    \"a\": \"cold Vietnamese lager-style beer (e.g. Saigon or 333)\",\n    \"b\": \"tannic Cabernet Sauvignon\",\n    \"c\": \"heavily oaked Chardonnay\",\n    \"d\": \"sweet Tokaji Aszú 5 puttonyos\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "c",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 434,
            "output_tokens": 4,
            "latency_ms": 655.0,
            "usd": 0.000363,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 434,
            "output_tokens": 4,
            "latency_ms": 716.7,
            "usd": 0.000363,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 434,
            "output_tokens": 4,
            "latency_ms": 568.7,
            "usd": 0.000363,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 434.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 646.8,
        "usd_sum": 0.001089
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-017",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "8f60336584f33baf",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for 24-month aged Comté served on its own.\n\nINPUT:\n{\n  \"dish\": \"24-month aged Comté, served at room temperature\",\n  \"candidates\": {\n    \"a\": \"Jura Vin Jaune (Savagnin, oxidative, nutty)\",\n    \"b\": \"young oaky Cabernet Sauvignon\",\n    \"c\": \"dry Muscadet\",\n    \"d\": \"Sauternes\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "b",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 4,
            "latency_ms": 1669.8,
            "usd": 0.000345,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 4,
            "latency_ms": 679.5,
            "usd": 0.000345,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 4,
            "latency_ms": 2188.0,
            "usd": 0.000345,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 411.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1512.433,
        "usd_sum": 0.001035
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-018",
      "category": "pairing",
      "capability": "regional_pairing",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "9209873bc05779b7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for a fragrant, moderately spicy Thai green curry with chicken and Thai basil.\n\nINPUT:\n{\n  \"dish\": \"Thai green curry with chicken, Thai basil, kaffir lime\",\n  \"candidates\": {\n    \"a\": \"off-dry New Zealand Sauvignon Blanc with residual sugar\",\n    \"b\": \"heavy Port wine\",\n    \"c\": \"young tannic Bordeaux\",\n    \"d\": \"Islay peated whisky\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 410,
            "output_tokens": 4,
            "latency_ms": 682.0,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 410,
            "output_tokens": 4,
            "latency_ms": 632.1,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 410,
            "output_tokens": 4,
            "latency_ms": 661.0,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 410.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 658.367,
        "usd_sum": 0.001032
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-019",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "mexican",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "da735293a9d3102b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage.\n\nINPUT:\n{\n  \"dish\": \"Oaxacan mole negro enchiladas (rich chocolate-chile-spice sauce over chicken)\",\n  \"candidates\": {\n    \"a\": \"medium-tannin Garnacha from Aragón\",\n    \"b\": \"unoaked Muscadet\",\n    \"c\": \"very sweet late-harvest Riesling\",\n    \"d\": \"bone-dry Brut Champagne\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "c",
          "b"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 407,
            "output_tokens": 4,
            "latency_ms": 501.2,
            "usd": 0.000342,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 407,
            "output_tokens": 4,
            "latency_ms": 1528.6,
            "usd": 0.000342,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 407,
            "output_tokens": 4,
            "latency_ms": 491.7,
            "usd": 0.000342,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 407.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 840.5,
        "usd_sum": 0.001026
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-020",
      "category": "pairing",
      "capability": "pairing_grade_reasoning",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "bac251ebb1c064ef",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nA pairing's overall_score is computed as 96. Using the OGS grade scale, return the letter grade.\n\nINPUT:\n{\n  \"overall_score\": 96\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "A+"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "A+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a+",
            "expected": "a+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 5,
            "latency_ms": 511.1,
            "usd": 0.000349,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "A+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a+",
            "expected": "a+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 5,
            "latency_ms": 703.4,
            "usd": 0.000349,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "A+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a+",
            "expected": "a+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 5,
            "latency_ms": 548.4,
            "usd": 0.000349,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 411.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 587.633,
        "usd_sum": 0.001047
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-001",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "1df867dd3a55af5b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat is the ONLY grape variety permitted in Barolo DOCG? Respond with just the variety name.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "nebbiolo"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 277,
            "output_tokens": 7,
            "latency_ms": 530.9,
            "usd": 0.00025,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 277,
            "output_tokens": 7,
            "latency_ms": 654.8,
            "usd": 0.00025,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 277,
            "output_tokens": 7,
            "latency_ms": 2025.0,
            "usd": 0.00025,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 277.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1070.233,
        "usd_sum": 0.00075
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-002",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "24a95f8a1c0a0a45",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat single grape variety is used for Chablis AOC? Respond with just the variety name.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "chardonnay"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Chardonnay",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chardonnay",
            "expected": "chardonnay",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 275,
            "output_tokens": 8,
            "latency_ms": 543.6,
            "usd": 0.000252,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Chardonnay",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chardonnay",
            "expected": "chardonnay",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 275,
            "output_tokens": 8,
            "latency_ms": 642.2,
            "usd": 0.000252,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Chardonnay",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chardonnay",
            "expected": "chardonnay",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 275,
            "output_tokens": 8,
            "latency_ms": 575.5,
            "usd": 0.000252,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 275.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 587.1,
        "usd_sum": 0.000756
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-003",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "vietnamese",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "15d36148f8b853e6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhich country is phở originally from? Respond with just the country name.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "vietnam"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Vietnam",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "vietnam",
            "expected": "vietnam",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 269,
            "output_tokens": 4,
            "latency_ms": 964.6,
            "usd": 0.000231,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Vietnam",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "vietnam",
            "expected": "vietnam",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 269,
            "output_tokens": 4,
            "latency_ms": 1048.9,
            "usd": 0.000231,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Vietnam",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "vietnam",
            "expected": "vietnam",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 269,
            "output_tokens": 4,
            "latency_ms": 648.4,
            "usd": 0.000231,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 269.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 887.3,
        "usd_sum": 0.000693
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-004",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "e332af036a2bd591",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the classic Alsace wine traditionally paired with ripe, stinky Muenster cheese. One word, grape variety only.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "gewurztraminer"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Gewürztraminer",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "gewürztraminer",
            "expected": "gewurztraminer",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 282,
            "output_tokens": 9,
            "latency_ms": 574.8,
            "usd": 0.000262,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Gewürztraminer",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "gewürztraminer",
            "expected": "gewurztraminer",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 282,
            "output_tokens": 9,
            "latency_ms": 612.4,
            "usd": 0.000262,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Gewürztraminer",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "gewürztraminer",
            "expected": "gewurztraminer",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 282,
            "output_tokens": 9,
            "latency_ms": 833.6,
            "usd": 0.000262,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 282.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 673.6,
        "usd_sum": 0.000786
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-005",
      "category": "regional",
      "capability": "french_classical_cuisine",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "4aa2c63ffdeeacfb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the French mother sauce that is made from a brown roux, brown stock and mirepoix, traditionally enriched with tomato. One word, lowercase.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "espagnole"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "espagnole",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "espagnole",
            "expected": "espagnole",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 287,
            "output_tokens": 6,
            "latency_ms": 863.0,
            "usd": 0.000254,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "espagnole",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "espagnole",
            "expected": "espagnole",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 287,
            "output_tokens": 6,
            "latency_ms": 550.2,
            "usd": 0.000254,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "espagnole",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "espagnole",
            "expected": "espagnole",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 287,
            "output_tokens": 6,
            "latency_ms": 510.5,
            "usd": 0.000254,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 287.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 641.233,
        "usd_sum": 0.000762
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-006",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "medium",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "979e7579b34d9797",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat is the name of the official Rioja classification for a red wine aged at least 2 years (1 in oak) and 3+ years total? Respond with just the term.\n\nINPUT:\n{\n  \"question\": \"official Rioja classification for red wine aged 2+ years (1+ in oak), 3+ total\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "reserva"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Reserva",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "reserva",
            "expected": "reserva",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 334,
            "output_tokens": 6,
            "latency_ms": 551.4,
            "usd": 0.000291,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Reserva",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "reserva",
            "expected": "reserva",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 334,
            "output_tokens": 6,
            "latency_ms": 764.4,
            "usd": 0.000291,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Reserva",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "reserva",
            "expected": "reserva",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 334,
            "output_tokens": 6,
            "latency_ms": 688.3,
            "usd": 0.000291,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 334.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 668.033,
        "usd_sum": 0.000873
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-007",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "korean",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "23a69016b1082553",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nFrom which country does kimchi originate? Respond with just the country name in English.\n\nINPUT:\n{\n  \"dish\": \"kimchi\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "korea",
        "aliases": [
          "south korea",
          "republic of korea"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "South Korea",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "south korea",
            "expected": "korea",
            "aliases": [
              "south korea",
              "republic of korea"
            ]
          },
          "metrics": {
            "input_tokens": 290,
            "output_tokens": 5,
            "latency_ms": 945.9,
            "usd": 0.000252,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "South Korea",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "south korea",
            "expected": "korea",
            "aliases": [
              "south korea",
              "republic of korea"
            ]
          },
          "metrics": {
            "input_tokens": 290,
            "output_tokens": 5,
            "latency_ms": 593.2,
            "usd": 0.000252,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Korea",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "korea",
            "expected": "korea",
            "aliases": [
              "south korea",
              "republic of korea"
            ]
          },
          "metrics": {
            "input_tokens": 290,
            "output_tokens": 4,
            "latency_ms": 561.6,
            "usd": 0.000248,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 290.0,
        "output_tokens_mean": 4.667,
        "latency_ms_mean": 700.233,
        "usd_sum": 0.000752
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-008",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "0a12eb057eb17654",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat is the single grape variety required for Barbaresco DOCG?\n\nINPUT:\n{\n  \"wine\": \"Barbaresco DOCG\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "nebbiolo"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 290,
            "output_tokens": 7,
            "latency_ms": 590.2,
            "usd": 0.00026,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 290,
            "output_tokens": 7,
            "latency_ms": 559.1,
            "usd": 0.00026,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 290,
            "output_tokens": 7,
            "latency_ms": 570.4,
            "usd": 0.00026,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 290.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 573.233,
        "usd_sum": 0.00078
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-009",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "peruvian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "767d178c3cce3333",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhich country is classic ceviche most commonly attributed to? Respond with the country name.\n\nINPUT:\n{\n  \"dish\": \"ceviche\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "peru"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Peru",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "peru",
            "expected": "peru",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 4,
            "latency_ms": 2148.4,
            "usd": 0.000249,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Peru",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "peru",
            "expected": "peru",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 4,
            "latency_ms": 490.2,
            "usd": 0.000249,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Peru",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "peru",
            "expected": "peru",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 4,
            "latency_ms": 511.1,
            "usd": 0.000249,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 291.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1049.9,
        "usd_sum": 0.000747
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-010",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "1001169bc92aaacd",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country does paella valenciana originate? Respond with the country name.\n\nINPUT:\n{\n  \"dish\": \"paella valenciana\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "spain"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Spain",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spain",
            "expected": "spain",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 4,
            "latency_ms": 699.5,
            "usd": 0.000249,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Spain",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spain",
            "expected": "spain",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 4,
            "latency_ms": 659.8,
            "usd": 0.000249,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Spain",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spain",
            "expected": "spain",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 4,
            "latency_ms": 623.7,
            "usd": 0.000249,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 291.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 661.0,
        "usd_sum": 0.000747
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-011",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "571e5074e66e6a11",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the classic French white wine region traditionally paired with raw oysters. Respond with just the region name.\n\nINPUT:\n{\n  \"dish\": \"raw oysters on the half shell\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "chablis",
        "aliases": [
          "muscadet"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Chablis",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chablis",
            "expected": "chablis",
            "aliases": [
              "muscadet"
            ]
          },
          "metrics": {
            "input_tokens": 298,
            "output_tokens": 7,
            "latency_ms": 512.5,
            "usd": 0.000266,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Chablis",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chablis",
            "expected": "chablis",
            "aliases": [
              "muscadet"
            ]
          },
          "metrics": {
            "input_tokens": 298,
            "output_tokens": 7,
            "latency_ms": 686.2,
            "usd": 0.000266,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Chablis",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chablis",
            "expected": "chablis",
            "aliases": [
              "muscadet"
            ]
          },
          "metrics": {
            "input_tokens": 298,
            "output_tokens": 7,
            "latency_ms": 500.8,
            "usd": 0.000266,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 298.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 566.5,
        "usd_sum": 0.000798
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-012",
      "category": "regional",
      "capability": "french_classical_cuisine",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "d9792be4d5ca7335",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the French mother sauce made from clarified butter, egg yolks, and lemon juice.\n\nINPUT:\n{\n  \"description\": \"emulsion of clarified butter, egg yolks, and lemon juice\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "hollandaise"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "hollandaise",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hollandaise",
            "expected": "hollandaise",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 306,
            "output_tokens": 6,
            "latency_ms": 667.6,
            "usd": 0.000269,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Hollandaise",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hollandaise",
            "expected": "hollandaise",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 306,
            "output_tokens": 7,
            "latency_ms": 548.7,
            "usd": 0.000273,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Hollandaise",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hollandaise",
            "expected": "hollandaise",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 306,
            "output_tokens": 7,
            "latency_ms": 700.0,
            "usd": 0.000273,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 306.0,
        "output_tokens_mean": 6.667,
        "latency_ms_mean": 638.767,
        "usd_sum": 0.000815
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-013",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "c831a091be5025f9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country did edomae-style sushi originate?\n\nINPUT:\n{\n  \"dish\": \"edomae sushi\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "japan"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Japan",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "japan",
            "expected": "japan",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 288,
            "output_tokens": 4,
            "latency_ms": 518.0,
            "usd": 0.000246,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Japan",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "japan",
            "expected": "japan",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 288,
            "output_tokens": 4,
            "latency_ms": 478.1,
            "usd": 0.000246,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Japan",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "japan",
            "expected": "japan",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 288,
            "output_tokens": 4,
            "latency_ms": 655.4,
            "usd": 0.000246,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 288.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 550.5,
        "usd_sum": 0.000738
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-014",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "moroccan",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "7d761cbfb86342af",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country does the slow-cooked dish tagine originate?\n\nINPUT:\n{\n  \"dish\": \"tagine (clay-pot slow-cooked stew)\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "morocco"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Morocco",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "morocco",
            "expected": "morocco",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 296,
            "output_tokens": 4,
            "latency_ms": 530.4,
            "usd": 0.000253,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Morocco",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "morocco",
            "expected": "morocco",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 296,
            "output_tokens": 4,
            "latency_ms": 1170.2,
            "usd": 0.000253,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Morocco",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "morocco",
            "expected": "morocco",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 296,
            "output_tokens": 4,
            "latency_ms": 485.1,
            "usd": 0.000253,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 296.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 728.567,
        "usd_sum": 0.000759
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-015",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "2f35d24bb9d58118",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat single grape variety is used for white Sancerre AOC?\n\nINPUT:\n{\n  \"wine\": \"Sancerre AOC blanc\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "sauvignon blanc",
        "aliases": [
          "sauvignon_blanc",
          "sauvignon"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Sauvignon Blanc",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauvignon blanc",
            "expected": "sauvignon blanc",
            "aliases": [
              "sauvignon_blanc",
              "sauvignon"
            ]
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 9,
            "latency_ms": 728.1,
            "usd": 0.000269,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Sauvignon",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauvignon",
            "expected": "sauvignon blanc",
            "aliases": [
              "sauvignon_blanc",
              "sauvignon"
            ]
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 7,
            "latency_ms": 539.6,
            "usd": 0.000261,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Sauvignon",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauvignon",
            "expected": "sauvignon blanc",
            "aliases": [
              "sauvignon_blanc",
              "sauvignon"
            ]
          },
          "metrics": {
            "input_tokens": 291,
            "output_tokens": 7,
            "latency_ms": 751.7,
            "usd": 0.000261,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 291.0,
        "output_tokens_mean": 7.667,
        "latency_ms_mean": 673.133,
        "usd_sum": 0.000791
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-016",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "mexican",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "efb489c56faf463a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country does mole poblano originate?\n\nINPUT:\n{\n  \"dish\": \"mole poblano\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "mexico"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Mexico",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mexico",
            "expected": "mexico",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 284,
            "output_tokens": 4,
            "latency_ms": 1172.5,
            "usd": 0.000243,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Mexico",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mexico",
            "expected": "mexico",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 284,
            "output_tokens": 4,
            "latency_ms": 724.2,
            "usd": 0.000243,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Mexico",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mexico",
            "expected": "mexico",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 284,
            "output_tokens": 4,
            "latency_ms": 682.7,
            "usd": 0.000243,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 284.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 859.8,
        "usd_sum": 0.000729
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-017",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "53530e4634162afc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the Alsatian white wine traditionally served with choucroute garnie.\n\nINPUT:\n{\n  \"dish\": \"choucroute garnie\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "riesling"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Riesling",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "riesling",
            "expected": "riesling",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 293,
            "output_tokens": 7,
            "latency_ms": 1023.8,
            "usd": 0.000262,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Riesling",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "riesling",
            "expected": "riesling",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 293,
            "output_tokens": 7,
            "latency_ms": 575.0,
            "usd": 0.000262,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Riesling",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "riesling",
            "expected": "riesling",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 293,
            "output_tokens": 7,
            "latency_ms": 527.5,
            "usd": 0.000262,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 293.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 708.767,
        "usd_sum": 0.000786
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-018",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "brazilian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "28a636bdc82633de",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhich country claims feijoada completa as a national dish?\n\nINPUT:\n{\n  \"dish\": \"feijoada completa\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "brazil"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Brazil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "brazil",
            "expected": "brazil",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 289,
            "output_tokens": 4,
            "latency_ms": 535.7,
            "usd": 0.000247,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Brazil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "brazil",
            "expected": "brazil",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 289,
            "output_tokens": 4,
            "latency_ms": 590.5,
            "usd": 0.000247,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Brazil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "brazil",
            "expected": "brazil",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 289,
            "output_tokens": 4,
            "latency_ms": 617.5,
            "usd": 0.000247,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 289.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 581.233,
        "usd_sum": 0.000741
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-019",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "hard",
      "cuisine_region": "hungarian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "2c5869ce479f65cc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nTokaji Aszú's vineyards are centred in a single country. Name that country in English.\n\nINPUT:\n{\n  \"wine\": \"Tokaji Aszú\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "hungary"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Hungary",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hungary",
            "expected": "hungary",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 301,
            "output_tokens": 4,
            "latency_ms": 517.5,
            "usd": 0.000257,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "Hungary",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hungary",
            "expected": "hungary",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 301,
            "output_tokens": 4,
            "latency_ms": 1734.1,
            "usd": 0.000257,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "Hungary",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hungary",
            "expected": "hungary",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 301,
            "output_tokens": 4,
            "latency_ms": 755.9,
            "usd": 0.000257,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 301.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1002.5,
        "usd_sum": 0.000771
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-020",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "hard",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "aa3761bdbf2d490e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the Galician white wine grape traditionally paired with pulpo á feira (boiled octopus with pimentón).\n\nINPUT:\n{\n  \"dish\": \"pulpo á feira — Galician boiled octopus with pimentón and olive oil\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "albarino",
        "aliases": [
          "albariño"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Albariño",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "albariño",
            "expected": "albarino",
            "aliases": [
              "albariño"
            ]
          },
          "metrics": {
            "input_tokens": 315,
            "output_tokens": 7,
            "latency_ms": 2347.2,
            "usd": 0.00028,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "albariño",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "albariño",
            "expected": "albarino",
            "aliases": [
              "albariño"
            ]
          },
          "metrics": {
            "input_tokens": 315,
            "output_tokens": 6,
            "latency_ms": 913.6,
            "usd": 0.000276,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "albariño",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "albariño",
            "expected": "albarino",
            "aliases": [
              "albariño"
            ]
          },
          "metrics": {
            "input_tokens": 315,
            "output_tokens": 6,
            "latency_ms": 535.0,
            "usd": 0.000276,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 315.0,
        "output_tokens_mean": 6.333,
        "latency_ms_mean": 1265.267,
        "usd_sum": 0.000832
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-001",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "08845fcf7f8beb09",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"salmon and tuna nigiri\",\n  \"beverage\": \"young Bordeaux Left Bank (Cabernet Sauvignon, high tannin)\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_TANNIN_FISH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_TANNIN_FISH",
            "expected": "RISK_TANNIN_FISH",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 787,
            "output_tokens": 13,
            "latency_ms": 1728.7,
            "usd": 0.000682,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_TANNIN_FISH",
            "expected": "RISK_TANNIN_FISH",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 787,
            "output_tokens": 13,
            "latency_ms": 589.7,
            "usd": 0.000682,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_TANNIN_FISH",
            "expected": "RISK_TANNIN_FISH",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 787,
            "output_tokens": 13,
            "latency_ms": 1930.3,
            "usd": 0.000682,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 787.0,
        "output_tokens_mean": 13.0,
        "latency_ms_mean": 1416.233,
        "usd_sum": 0.002046
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-002",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "7a188b424515258f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"Thai green curry with bird's eye chilli, high heat\",\n  \"beverage\": \"15.5% ABV California Zinfandel\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SPICE_ALCOHOL"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SPICE_ALCOHOL",
            "expected": "RISK_SPICE_ALCOHOL",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 785,
            "output_tokens": 12,
            "latency_ms": 504.1,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SPICE_ALCOHOL",
            "expected": "RISK_SPICE_ALCOHOL",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 785,
            "output_tokens": 12,
            "latency_ms": 1815.4,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SPICE_ALCOHOL",
            "expected": "RISK_SPICE_ALCOHOL",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 785,
            "output_tokens": 12,
            "latency_ms": 629.4,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 785.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 982.967,
        "usd_sum": 0.002028
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-003",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "a60fa6dd5c1ff452",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code for the pairing. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"delicate flounder sashimi with ponzu\",\n  \"beverage\": \"heavily new-oaked California Chardonnay, 14% ABV, buttery ML\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_OAK_MASKS"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_OAK_MASKS",
            "expected": "RISK_OAK_MASKS",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 789,
            "output_tokens": 12,
            "latency_ms": 684.7,
            "usd": 0.000679,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_OAK_MASKS",
            "expected": "RISK_OAK_MASKS",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 789,
            "output_tokens": 12,
            "latency_ms": 1925.0,
            "usd": 0.000679,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_OAK_MASKS",
            "expected": "RISK_OAK_MASKS",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 789,
            "output_tokens": 12,
            "latency_ms": 704.4,
            "usd": 0.000679,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 789.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 1104.7,
        "usd_sum": 0.002037
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-004",
      "category": "risk",
      "capability": "risk_severity",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "16286de7fea14ae1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nUsing OGS risk severity levels (low/medium/high), classify the RISK_ALCOHOL_OVERPOWERS severity for the pairing below. Respond with just the severity token.\n\nINPUT:\n{\n  \"dish\": \"pan-seared halibut with beurre blanc, very delicate flavours\",\n  \"beverage\": \"15.5% ABV Amarone della Valpolicella\"\n}\n\nANSWER FORMAT: one_of:low,medium,high",
      "expected": {
        "value": "high"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 731,
            "output_tokens": 4,
            "latency_ms": 1899.0,
            "usd": 0.000601,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 731,
            "output_tokens": 4,
            "latency_ms": 513.6,
            "usd": 0.000601,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 731,
            "output_tokens": 4,
            "latency_ms": 4095.8,
            "usd": 0.000601,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 731.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 2169.467,
        "usd_sum": 0.001803
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-005",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "cd44f8ac361108e8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code for the pairing. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"caramel panna cotta with honey\",\n  \"beverage\": \"bone-dry Brut Nature Champagne (zero dosage)\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SWEETNESS_CONFLICT"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SWEETNESS_CONFLICT",
            "expected": "RISK_SWEETNESS_CONFLICT",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 781,
            "output_tokens": 12,
            "latency_ms": 569.3,
            "usd": 0.000673,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SWEETNESS_CONFLICT",
            "expected": "RISK_SWEETNESS_CONFLICT",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 781,
            "output_tokens": 12,
            "latency_ms": 652.7,
            "usd": 0.000673,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SWEETNESS_CONFLICT",
            "expected": "RISK_SWEETNESS_CONFLICT",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 781,
            "output_tokens": 12,
            "latency_ms": 572.3,
            "usd": 0.000673,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 781.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 598.1,
        "usd_sum": 0.002019
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-006",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "28cb271326a709c7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"raw tuna sashimi with a light soy-wasabi dip\",\n  \"beverage\": \"young tannic Napa Cabernet Sauvignon\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_TANNIN_FISH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 791,
            "output_tokens": 13,
            "latency_ms": 660.7,
            "usd": 0.000685,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 791,
            "output_tokens": 13,
            "latency_ms": 672.9,
            "usd": 0.000685,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 791,
            "output_tokens": 13,
            "latency_ms": 2223.7,
            "usd": 0.000685,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 791.0,
        "output_tokens_mean": 13.0,
        "latency_ms_mean": 1185.767,
        "usd_sum": 0.002055
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-007",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "93c46eac23eca249",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"spicy Thai green curry with chilies and coconut milk\",\n  \"beverage\": \"15.5% abv California Zinfandel\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SPICE_ALCOHOL"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 786,
            "output_tokens": 12,
            "latency_ms": 699.9,
            "usd": 0.000677,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 786,
            "output_tokens": 12,
            "latency_ms": 513.9,
            "usd": 0.000677,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 786,
            "output_tokens": 12,
            "latency_ms": 1709.4,
            "usd": 0.000677,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 786.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 974.4,
        "usd_sum": 0.002031
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-008",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "2409e606629eebf9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"grilled escarole salad with bitter radicchio and walnuts\",\n  \"beverage\": \"bone-dry bitter amontillado sherry\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_BITTER_AMPLIFY"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 788,
            "output_tokens": 12,
            "latency_ms": 477.9,
            "usd": 0.000678,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 788,
            "output_tokens": 12,
            "latency_ms": 1920.9,
            "usd": 0.000678,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 788,
            "output_tokens": 12,
            "latency_ms": 704.8,
            "usd": 0.000678,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 788.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 1034.533,
        "usd_sum": 0.002034
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-009",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "a00498fd7fce7cb7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"raw scallop crudo with yuzu\",\n  \"beverage\": \"heavily oaked California Chardonnay, new French barrels\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_OAK_MASKS"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_oak_masks",
            "expected": "risk_oak_masks",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 785,
            "output_tokens": 12,
            "latency_ms": 1727.7,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_oak_masks",
            "expected": "risk_oak_masks",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 785,
            "output_tokens": 12,
            "latency_ms": 678.6,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_oak_masks",
            "expected": "risk_oak_masks",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 785,
            "output_tokens": 12,
            "latency_ms": 1848.1,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 785.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 1418.133,
        "usd_sum": 0.002028
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-010",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "b28af5fc40407ce0",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"warm clear beef consommé\",\n  \"beverage\": \"sweet late-harvest Moscato d'Asti\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SWEETNESS_CONFLICT"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_sweetness_conflict",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 780,
            "output_tokens": 12,
            "latency_ms": 1022.9,
            "usd": 0.000672,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 780,
            "output_tokens": 13,
            "latency_ms": 518.8,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 780,
            "output_tokens": 13,
            "latency_ms": 1812.3,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": -0.32,
        "ci95_hi": 0.9867,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 780.0,
        "output_tokens_mean": 12.667,
        "latency_ms_mean": 1118.0,
        "usd_sum": 0.002024
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-011",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "fba6f85afa81595f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"frisée salad with aggressive Dijon vinaigrette (1:2 vinegar to oil)\",\n  \"beverage\": \"bone-dry Brut Nature Champagne\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_ACID_CLASH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_ACID_CLASH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_acid_clash",
            "expected": "risk_acid_clash",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 795,
            "output_tokens": 11,
            "latency_ms": 588.9,
            "usd": 0.00068,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_ACID_CLASH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_acid_clash",
            "expected": "risk_acid_clash",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 795,
            "output_tokens": 11,
            "latency_ms": 1651.3,
            "usd": 0.00068,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_ACID_CLASH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_acid_clash",
            "expected": "risk_acid_clash",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 795,
            "output_tokens": 11,
            "latency_ms": 1379.5,
            "usd": 0.00068,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 795.0,
        "output_tokens_mean": 11.0,
        "latency_ms_mean": 1206.567,
        "usd_sum": 0.00204
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-012",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "2f3a4360a39afdc9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"steamed turbot with beurre blanc\",\n  \"beverage\": \"young powerful Barolo\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_INTENSITY_MISMATCH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 776,
            "output_tokens": 13,
            "latency_ms": 673.7,
            "usd": 0.000673,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 776,
            "output_tokens": 13,
            "latency_ms": 1738.5,
            "usd": 0.000673,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 776,
            "output_tokens": 13,
            "latency_ms": 574.9,
            "usd": 0.000673,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 776.0,
        "output_tokens_mean": 13.0,
        "latency_ms_mean": 995.7,
        "usd_sum": 0.002019
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-013",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "390e90cae060ada8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"yellowtail sashimi\",\n  \"beverage\": \"overproof 63% abv Jamaican rum\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_ALCOHOL_OVERPOWERS"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_ALCOHOL_OVERPOWERS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_alcohol_overpowers",
            "expected": "risk_alcohol_overpowers",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 780,
            "output_tokens": 13,
            "latency_ms": 2133.1,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_ALCOHOL_OVERPOWERS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_alcohol_overpowers",
            "expected": "risk_alcohol_overpowers",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 780,
            "output_tokens": 13,
            "latency_ms": 702.8,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_ALCOHOL_OVERPOWERS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_alcohol_overpowers",
            "expected": "risk_alcohol_overpowers",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 780,
            "output_tokens": 13,
            "latency_ms": 768.3,
            "usd": 0.000676,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 780.0,
        "output_tokens_mean": 13.0,
        "latency_ms_mean": 1201.4,
        "usd_sum": 0.002028
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-014",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "d83b8b2ecabe9907",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"gin-cured salmon gravlax with dill\",\n  \"beverage\": \"young tannic Chianti Classico\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_TANNIN_FISH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 782,
            "output_tokens": 13,
            "latency_ms": 2571.1,
            "usd": 0.000678,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 782,
            "output_tokens": 13,
            "latency_ms": 554.1,
            "usd": 0.000678,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 782,
            "output_tokens": 13,
            "latency_ms": 548.4,
            "usd": 0.000678,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 782.0,
        "output_tokens_mean": 13.0,
        "latency_ms_mean": 1224.533,
        "usd_sum": 0.002034
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-015",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "1e83097d483e0639",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"poached lobster with drawn butter\",\n  \"beverage\": \"aggressive 80-IBU West Coast IPA\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_BITTER_AMPLIFY"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 779,
            "output_tokens": 13,
            "latency_ms": 1672.4,
            "usd": 0.000675,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 779,
            "output_tokens": 13,
            "latency_ms": 595.4,
            "usd": 0.000675,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 779,
            "output_tokens": 13,
            "latency_ms": 1849.5,
            "usd": 0.000675,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 779.0,
        "output_tokens_mean": 13.0,
        "latency_ms_mean": 1372.433,
        "usd_sum": 0.002025
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-016",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "mexican",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "b2e02272ca87b2d4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"habanero-heavy Yucatán salsa\",\n  \"beverage\": \"15% abv Zinfandel\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SPICE_ALCOHOL"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 783,
            "output_tokens": 12,
            "latency_ms": 848.1,
            "usd": 0.000674,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 783,
            "output_tokens": 12,
            "latency_ms": 1744.2,
            "usd": 0.000674,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 783,
            "output_tokens": 12,
            "latency_ms": 543.3,
            "usd": 0.000674,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 783.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 1045.2,
        "usd_sum": 0.002022
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-017",
      "category": "risk",
      "capability": "risk_severity",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "12372c8de13add55",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nUsing OGS risk severity levels (low/medium/high), classify the risk severity of the described pairing. Respond with just the severity level.\n\nINPUT:\n{\n  \"dish\": \"simple grilled branzino with lemon\",\n  \"beverage\": \"heavily oaked California Chardonnay\",\n  \"risk_code\": \"RISK_OAK_MASKS\"\n}\n\nANSWER FORMAT: one_of:low,medium,high",
      "expected": {
        "value": "medium"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "high",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "high",
            "expected": "medium",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 725,
            "output_tokens": 4,
            "latency_ms": 2088.6,
            "usd": 0.000596,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "high",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "high",
            "expected": "medium",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 725,
            "output_tokens": 4,
            "latency_ms": 496.7,
            "usd": 0.000596,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "high",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "high",
            "expected": "medium",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 725,
            "output_tokens": 4,
            "latency_ms": 560.6,
            "usd": 0.000596,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 725.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1048.633,
        "usd_sum": 0.001788
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-018",
      "category": "risk",
      "capability": "risk_severity",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "7a97c002792dfad9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nUsing OGS risk severity levels, classify the severity of a tannic red wine with raw anchovies.\n\nINPUT:\n{\n  \"dish\": \"raw marinated anchovies\",\n  \"beverage\": \"young tannic Cabernet Sauvignon\",\n  \"risk_code\": \"RISK_TANNIN_FISH\"\n}\n\nANSWER FORMAT: one_of:low,medium,high",
      "expected": {
        "value": "high"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 715,
            "output_tokens": 4,
            "latency_ms": 1891.9,
            "usd": 0.000588,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 715,
            "output_tokens": 4,
            "latency_ms": 572.3,
            "usd": 0.000588,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 715,
            "output_tokens": 4,
            "latency_ms": 2062.7,
            "usd": 0.000588,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 715.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1508.967,
        "usd_sum": 0.001764
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-019",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "german",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "0e3d6c3f9d86af0b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"raw Kumamoto oyster\",\n  \"beverage\": \"off-dry Spätlese Riesling with 45 g/L residual sugar\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SWEETNESS_CONFLICT"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_sweetness_conflict",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 790,
            "output_tokens": 12,
            "latency_ms": 992.9,
            "usd": 0.00068,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_sweetness_conflict",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 790,
            "output_tokens": 12,
            "latency_ms": 937.7,
            "usd": 0.00068,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_sweetness_conflict",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 790,
            "output_tokens": 12,
            "latency_ms": 1931.1,
            "usd": 0.00068,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 790.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 1287.233,
        "usd_sum": 0.00204
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-020",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "2dd649d25c606ee7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"slow-braised oxtail stew with dark jus\",\n  \"beverage\": \"very pale Provençal rosé, 12.5% abv\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_INTENSITY_MISMATCH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 791,
            "output_tokens": 13,
            "latency_ms": 678.6,
            "usd": 0.000685,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 791,
            "output_tokens": 13,
            "latency_ms": 775.7,
            "usd": 0.000685,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 791,
            "output_tokens": 13,
            "latency_ms": 2124.3,
            "usd": 0.000685,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 791.0,
        "output_tokens_mean": 13.0,
        "latency_ms_mean": 1192.867,
        "usd_sum": 0.002055
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-001",
      "category": "safety",
      "capability": "cooking_temperature",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "8e37b185cfaed2aa",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhat is the minimum core temperature (°C) required to safely cook poultry under the OGS HACCP reference? Respond with a single number.\n\nINPUT:\n{\n  \"product\": \"whole chicken\"\n}\n\nANSWER FORMAT: numeric:integer or decimal °C",
      "expected": {
        "value": 75,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 927,
            "output_tokens": 5,
            "latency_ms": 974.7,
            "usd": 0.000762,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 927,
            "output_tokens": 5,
            "latency_ms": 732.4,
            "usd": 0.000762,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 927,
            "output_tokens": 5,
            "latency_ms": 1880.7,
            "usd": 0.000762,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 927.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1195.933,
        "usd_sum": 0.002286
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-002",
      "category": "safety",
      "capability": "cooking_temperature",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "d38d6470a18f7e97",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nA restaurant serves minced-beef smash burgers. What minimum core temperature (°C) should the patties reach? Respond with a single number.\n\nINPUT:\n{\n  \"product\": \"minced beef patty\"\n}\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": 75,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 927,
            "output_tokens": 5,
            "latency_ms": 646.5,
            "usd": 0.000762,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 927,
            "output_tokens": 5,
            "latency_ms": 726.0,
            "usd": 0.000762,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 927,
            "output_tokens": 5,
            "latency_ms": 1913.3,
            "usd": 0.000762,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 927.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1095.267,
        "usd_sum": 0.002286
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-003",
      "category": "safety",
      "capability": "danger_zone",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "97f1d8e6fc44b9e2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nThe bacterial danger zone is defined between two temperatures in °C. Respond as 'LOW-HIGH' in °C (integers), e.g. '10-50'.\n\nANSWER FORMAT: exact:LOW-HIGH °C",
      "expected": {
        "value": "5-63",
        "aliases": [
          "5 to 63",
          "5-63°C",
          "5 C - 63 C"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "5-63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "5-63",
            "expected": "5-63",
            "aliases": [
              "5 to 63",
              "5-63°C",
              "5 C - 63 C"
            ]
          },
          "metrics": {
            "input_tokens": 914,
            "output_tokens": 7,
            "latency_ms": 693.6,
            "usd": 0.000759,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "5-63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "5-63",
            "expected": "5-63",
            "aliases": [
              "5 to 63",
              "5-63°C",
              "5 C - 63 C"
            ]
          },
          "metrics": {
            "input_tokens": 914,
            "output_tokens": 7,
            "latency_ms": 1737.9,
            "usd": 0.000759,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "5-63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "5-63",
            "expected": "5-63",
            "aliases": [
              "5 to 63",
              "5-63°C",
              "5 C - 63 C"
            ]
          },
          "metrics": {
            "input_tokens": 914,
            "output_tokens": 7,
            "latency_ms": 817.1,
            "usd": 0.000759,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 914.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1082.867,
        "usd_sum": 0.002277
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-004",
      "category": "safety",
      "capability": "cooling_rule",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "9e5770b6c7036769",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nUnder UK FSA guidance, hot food must be cooled from 63 °C down to 21 °C within how many hours? Respond with a single integer.\n\nANSWER FORMAT: numeric:integer hours",
      "expected": {
        "value": 2,
        "tolerance": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 911,
            "output_tokens": 5,
            "latency_ms": 525.7,
            "usd": 0.000749,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 911,
            "output_tokens": 5,
            "latency_ms": 2169.7,
            "usd": 0.000749,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 911,
            "output_tokens": 5,
            "latency_ms": 549.4,
            "usd": 0.000749,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 911.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1081.6,
        "usd_sum": 0.002247
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-005",
      "category": "safety",
      "capability": "reheating",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "cb2ff73059921b59",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nTo what minimum core temperature (°C) must a previously cooked and chilled dish be reheated under OGS HACCP? Respond with a single number.\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": 75,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 912,
            "output_tokens": 5,
            "latency_ms": 1958.3,
            "usd": 0.00075,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 912,
            "output_tokens": 5,
            "latency_ms": 520.4,
            "usd": 0.00075,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 912,
            "output_tokens": 5,
            "latency_ms": 1823.5,
            "usd": 0.00075,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 912.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1434.067,
        "usd_sum": 0.00225
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-006",
      "category": "safety",
      "capability": "hot_holding",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "b3c9f3a848d65390",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhat minimum temperature (°C) must hot food be held at for service? Respond with a single number.\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": 63,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 63.0,
            "expected": 63.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 898,
            "output_tokens": 5,
            "latency_ms": 776.2,
            "usd": 0.000738,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 63.0,
            "expected": 63.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 898,
            "output_tokens": 5,
            "latency_ms": 528.7,
            "usd": 0.000738,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 63.0,
            "expected": 63.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 898,
            "output_tokens": 5,
            "latency_ms": 1648.2,
            "usd": 0.000738,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 898.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 984.367,
        "usd_sum": 0.002214
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-007",
      "category": "safety",
      "capability": "cold_chain",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "f120f4a38737e981",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhat is the maximum temperature (°C) for frozen storage under OGS HACCP? Respond with a single number (negative).\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": -18,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "-18",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": -18.0,
            "expected": -18.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 904,
            "output_tokens": 5,
            "latency_ms": 1979.5,
            "usd": 0.000743,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "-18",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": -18.0,
            "expected": -18.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 904,
            "output_tokens": 5,
            "latency_ms": 529.5,
            "usd": 0.000743,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "-18",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": -18.0,
            "expected": -18.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 904,
            "output_tokens": 5,
            "latency_ms": 2841.9,
            "usd": 0.000743,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 904.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1783.633,
        "usd_sum": 0.002229
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-008",
      "category": "safety",
      "capability": "thawing_method",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "9b8f975bd042c7a0",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhich of the following thawing methods is explicitly NOT permitted under OGS HACCP: 'in the fridge at 4 °C', 'under cold running water', 'in the microwave just before cooking', 'at room temperature on the counter'? Respond with the exact method phrase.\n\nINPUT:\n{\n  \"methods\": [\n    \"in the fridge at 4 °C\",\n    \"under cold running water\",\n    \"in the microwave just before cooking\",\n    \"at room temperature on the counter\"\n  ]\n}\n\nANSWER FORMAT: exact:method phrase",
      "expected": {
        "value": "at room temperature on the counter",
        "aliases": [
          "room temperature",
          "on the counter",
          "room-temperature thawing"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "at room temperature on the counter",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "at room temperature on the counter",
            "expected": "at room temperature on the counter",
            "aliases": [
              "room temperature",
              "on the counter",
              "room-temperature thawing"
            ]
          },
          "metrics": {
            "input_tokens": 1001,
            "output_tokens": 9,
            "latency_ms": 889.4,
            "usd": 0.000837,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "at room temperature on the counter",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "at room temperature on the counter",
            "expected": "at room temperature on the counter",
            "aliases": [
              "room temperature",
              "on the counter",
              "room-temperature thawing"
            ]
          },
          "metrics": {
            "input_tokens": 1001,
            "output_tokens": 9,
            "latency_ms": 650.3,
            "usd": 0.000837,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "at room temperature on the counter",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "at room temperature on the counter",
            "expected": "at room temperature on the counter",
            "aliases": [
              "room temperature",
              "on the counter",
              "room-temperature thawing"
            ]
          },
          "metrics": {
            "input_tokens": 1001,
            "output_tokens": 9,
            "latency_ms": 590.9,
            "usd": 0.000837,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1001.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 710.2,
        "usd_sum": 0.002511
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-009",
      "category": "safety",
      "capability": "ccp_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "9f4b49c1c20e4376",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nFor a cook-chill dish of chicken curry with rice, identify the Critical Control Points from the OGS stage list. Return a comma-separated list using the OGS CCP stage IDs.\n\nINPUT:\n{\n  \"dish\": \"cook-chill chicken curry with rice\"\n}\n\nANSWER FORMAT: set:comma-separated CCP stage IDs",
      "expected": {
        "values": [
          "cooking",
          "cooling",
          "storage_cold",
          "reheating",
          "hot_holding"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "receiving,storage_cold,thawing,prep,cooking,hot_holding,cooling,storage_cold,reheating,service",
          "compliant": true,
          "score": 0.7143,
          "detail": {
            "predicted": [
              "cooking",
              "cooling",
              "hot_holding",
              "prep",
              "receiving",
              "reheating",
              "service",
              "storage_cold",
              "thawing"
            ],
            "expected": [
              "cooking",
              "cooling",
              "hot_holding",
              "reheating",
              "storage_cold"
            ],
            "precision": 0.5556,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 945,
            "output_tokens": 32,
            "latency_ms": 763.1,
            "usd": 0.000884,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "receiving,storage_cold,thawing,prep,cooking,hot_holding,cooling,storage_cold,reheating,service",
          "compliant": true,
          "score": 0.7143,
          "detail": {
            "predicted": [
              "cooking",
              "cooling",
              "hot_holding",
              "prep",
              "receiving",
              "reheating",
              "service",
              "storage_cold",
              "thawing"
            ],
            "expected": [
              "cooking",
              "cooling",
              "hot_holding",
              "reheating",
              "storage_cold"
            ],
            "precision": 0.5556,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 945,
            "output_tokens": 32,
            "latency_ms": 1933.2,
            "usd": 0.000884,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "receiving,storage_cold,thawing,prep,cooking,hot_holding,cooling,storage_cold,reheating,service",
          "compliant": true,
          "score": 0.7143,
          "detail": {
            "predicted": [
              "cooking",
              "cooling",
              "hot_holding",
              "prep",
              "receiving",
              "reheating",
              "service",
              "storage_cold",
              "thawing"
            ],
            "expected": [
              "cooking",
              "cooling",
              "hot_holding",
              "reheating",
              "storage_cold"
            ],
            "precision": 0.5556,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 945,
            "output_tokens": 32,
            "latency_ms": 703.3,
            "usd": 0.000884,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.7143,
      "score_summary": {
        "mean": 0.7143,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.7143,
        "ci95_hi": 0.7143,
        "n": 3
      },
      "score_if_valid": 0.7143,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 945.0,
        "output_tokens_mean": 32.0,
        "latency_ms_mean": 1133.2,
        "usd_sum": 0.002652
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-010",
      "category": "safety",
      "capability": "cooking_temperature",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "eea6814938f18565",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nA bistro serves a rare whole-muscle ribeye (seared exterior, internal 46 °C). Under OGS HACCP, is this permitted? Answer yes or no.\n\nINPUT:\n{\n  \"product\": \"whole-muscle ribeye\",\n  \"core_c\": 46\n}\n\nANSWER FORMAT: exact:yes|no",
      "expected": {
        "value": "yes"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "yes",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "yes",
            "expected": "yes",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 943,
            "output_tokens": 4,
            "latency_ms": 1779.8,
            "usd": 0.00077,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "yes",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "yes",
            "expected": "yes",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 943,
            "output_tokens": 4,
            "latency_ms": 716.1,
            "usd": 0.00077,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "yes",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "yes",
            "expected": "yes",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 943,
            "output_tokens": 4,
            "latency_ms": 544.2,
            "usd": 0.00077,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 943.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1013.367,
        "usd_sum": 0.00231
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-011",
      "category": "safety",
      "capability": "at_risk_groups",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "edc2981cec11c39e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nA children's hospital must serve hollandaise sauce. Which is the ONE compliant option: 'fresh shell eggs at 63 °C', 'fresh shell eggs at 70 °C', 'pasteurised egg yolk', 'raw shell eggs with lemon juice'? Respond with the exact phrase.\n\nINPUT:\n{\n  \"options\": [\n    \"fresh shell eggs at 63 °C\",\n    \"fresh shell eggs at 70 °C\",\n    \"pasteurised egg yolk\",\n    \"raw shell eggs with lemon juice\"\n  ]\n}\n\nANSWER FORMAT: exact:option phrase",
      "expected": {
        "value": "pasteurised egg yolk",
        "aliases": [
          "pasteurised egg",
          "pasteurized egg yolk",
          "pasteurized egg"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "pasteurised egg yolk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "pasteurised egg yolk",
            "expected": "pasteurised egg yolk",
            "aliases": [
              "pasteurised egg",
              "pasteurized egg yolk",
              "pasteurized egg"
            ]
          },
          "metrics": {
            "input_tokens": 1007,
            "output_tokens": 9,
            "latency_ms": 2117.1,
            "usd": 0.000842,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "pasteurised egg yolk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "pasteurised egg yolk",
            "expected": "pasteurised egg yolk",
            "aliases": [
              "pasteurised egg",
              "pasteurized egg yolk",
              "pasteurized egg"
            ]
          },
          "metrics": {
            "input_tokens": 1007,
            "output_tokens": 9,
            "latency_ms": 511.9,
            "usd": 0.000842,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "pasteurised egg yolk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "pasteurised egg yolk",
            "expected": "pasteurised egg yolk",
            "aliases": [
              "pasteurised egg",
              "pasteurized egg yolk",
              "pasteurized egg"
            ]
          },
          "metrics": {
            "input_tokens": 1007,
            "output_tokens": 9,
            "latency_ms": 1974.4,
            "usd": 0.000842,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1007.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 1534.467,
        "usd_sum": 0.002526
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-012",
      "category": "safety",
      "capability": "numeric_vector",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "2ba4113ffe26b2fc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nReturn the canonical HACCP reference temperatures in °C for this dish flow (poultry cook-chill operation): [cold_storage_max, poultry_core_cook_min, hot_hold_min, reheat_min, frozen_storage_max]. Respond with a JSON array of five numbers only, in that order.\n\nINPUT:\n{\n  \"flow\": \"poultry cook-chill\"\n}\n\nANSWER FORMAT: json:array of 5 numbers",
      "expected": {
        "values": [
          5,
          75,
          63,
          75,
          -18
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[5, 75, 63, 75, -18]",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 5,
                "expected": 5,
                "score": 1.0
              },
              {
                "index": 1,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 2,
                "predicted": 63,
                "expected": 63,
                "score": 1.0
              },
              {
                "index": 3,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 4,
                "predicted": -18,
                "expected": -18,
                "score": 1.0
              }
            ],
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 973,
            "output_tokens": 18,
            "latency_ms": 1213.7,
            "usd": 0.00085,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "[5, 75, 63, 75, -18]",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 5,
                "expected": 5,
                "score": 1.0
              },
              {
                "index": 1,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 2,
                "predicted": 63,
                "expected": 63,
                "score": 1.0
              },
              {
                "index": 3,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 4,
                "predicted": -18,
                "expected": -18,
                "score": 1.0
              }
            ],
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 973,
            "output_tokens": 18,
            "latency_ms": 724.4,
            "usd": 0.00085,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "[5, 75, 63, 75, -18]",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 5,
                "expected": 5,
                "score": 1.0
              },
              {
                "index": 1,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 2,
                "predicted": 63,
                "expected": 63,
                "score": 1.0
              },
              {
                "index": 3,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 4,
                "predicted": -18,
                "expected": -18,
                "score": 1.0
              }
            ],
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 973,
            "output_tokens": 18,
            "latency_ms": 662.0,
            "usd": 0.00085,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 973.0,
        "output_tokens_mean": 18.0,
        "latency_ms_mean": 866.7,
        "usd_sum": 0.00255
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-001",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "913118356da8150b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the umami intensity of a 36-month aged Parmigiano-Reggiano on the OGS 0-10 intensity scale (0 = not present, 10 = extreme). Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"Parmigiano-Reggiano 36 months\",\n  \"attribute\": \"umami\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 8.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "9.2",
          "compliant": true,
          "score": 0.5333,
          "detail": {
            "predicted": 9.2,
            "expected": 8.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 436,
            "output_tokens": 7,
            "latency_ms": 1713.7,
            "usd": 0.000377,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "9.0",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 9.0,
            "expected": 8.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 436,
            "output_tokens": 7,
            "latency_ms": 1816.8,
            "usd": 0.000377,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "9.0",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 9.0,
            "expected": 8.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 436,
            "output_tokens": 7,
            "latency_ms": 924.1,
            "usd": 0.000377,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6222,
      "score_summary": {
        "mean": 0.6222,
        "stdev": 0.077,
        "sem": 0.0445,
        "ci95_lo": 0.5351,
        "ci95_hi": 0.7094,
        "n": 3
      },
      "score_if_valid": 0.6222,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 436.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1484.867,
        "usd_sum": 0.001131
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-002",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "2690028d89b60292",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the sour intensity of undiluted fresh lemon juice on the OGS 0-10 intensity scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"fresh lemon juice (undiluted)\",\n  \"attribute\": \"sour\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 9.5,
        "tolerance": 1.0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 8.5,
            "expected": 9.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 7,
            "latency_ms": 597.2,
            "usd": 0.000357,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "9.5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.5,
            "expected": 9.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 7,
            "latency_ms": 2325.6,
            "usd": 0.000357,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "9.5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.5,
            "expected": 9.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 411,
            "output_tokens": 7,
            "latency_ms": 820.7,
            "usd": 0.000357,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": 0.0133,
        "ci95_hi": 1.32,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 411.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1247.833,
        "usd_sum": 0.001071
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-003",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "b09502eb4fd5d61f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the tannin intensity of a young (under 5 years old) Barolo on the OGS 0-10 intensity scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"beverage\": \"young Barolo, <5 years\",\n  \"attribute\": \"tannin\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 8.5,
        "tolerance": 1.0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "8.5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 8.5,
            "expected": 8.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 7,
            "latency_ms": 938.3,
            "usd": 0.000347,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 7.5,
            "expected": 8.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 7,
            "latency_ms": 2346.1,
            "usd": 0.000347,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "8.5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 8.5,
            "expected": 8.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 7,
            "latency_ms": 502.1,
            "usd": 0.000347,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": 0.0133,
        "ci95_hi": 1.32,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 399.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1262.167,
        "usd_sum": 0.001041
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-004",
      "category": "sensory",
      "capability": "basic_taste_vector",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "94f009713056d52f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nReturn the five basic tastes for a ripe, in-season strawberry, each on the OGS 0-10 intensity scale. Respond as a single JSON object with keys sweet, sour, salty, bitter, umami.\n\nINPUT:\n{\n  \"ingredient\": \"ripe, in-season strawberry\"\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"object\",\n  \"required\": [\n    \"sweet\",\n    \"sour\",\n    \"salty\",\n    \"bitter\",\n    \"umami\"\n  ],\n  \"properties\": {\n    \"sweet\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"sour\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"salty\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"bitter\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"umami\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    }\n  },\n  \"additionalProperties\": false\n}",
      "expected": {
        "values": {
          "sweet": 7.0,
          "sour": 4.0,
          "salty": 0.5,
          "bitter": 1.0,
          "umami": 0.5
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"sweet\": 7.5,\n  \"sour\": 3.5,\n  \"salty\": 0.5,\n  \"bitter\": 0.5,\n  \"umami\": 1.0\n}\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 3.5,
                "expected": 4.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 0.5,
                "expected": 0.5,
                "score": 1.0
              },
              "bitter": {
                "predicted": 0.5,
                "expected": 1.0,
                "score": 0.75
              },
              "umami": {
                "predicted": 1.0,
                "expected": 0.5,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 651,
            "output_tokens": 63,
            "latency_ms": 939.8,
            "usd": 0.000773,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"sweet\": 7.5,\n  \"sour\": 3.5,\n  \"salty\": 0.5,\n  \"bitter\": 0.5,\n  \"umami\": 1.0\n}\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 3.5,
                "expected": 4.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 0.5,
                "expected": 0.5,
                "score": 1.0
              },
              "bitter": {
                "predicted": 0.5,
                "expected": 1.0,
                "score": 0.75
              },
              "umami": {
                "predicted": 1.0,
                "expected": 0.5,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 651,
            "output_tokens": 63,
            "latency_ms": 931.6,
            "usd": 0.000773,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"sweet\": 7.5,\n  \"sour\": 3.5,\n  \"salty\": 0.5,\n  \"bitter\": 0.5,\n  \"umami\": 1.0\n}\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 3.5,
                "expected": 4.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 0.5,
                "expected": 0.5,
                "score": 1.0
              },
              "bitter": {
                "predicted": 0.5,
                "expected": 1.0,
                "score": 0.75
              },
              "umami": {
                "predicted": 1.0,
                "expected": 0.5,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 651,
            "output_tokens": 63,
            "latency_ms": 766.0,
            "usd": 0.000773,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.8,
      "score_summary": {
        "mean": 0.8,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8,
        "ci95_hi": 0.8,
        "n": 3
      },
      "score_if_valid": 0.8,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 651.0,
        "output_tokens_mean": 63.0,
        "latency_ms_mean": 879.133,
        "usd_sum": 0.002319
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-005",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "american",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "c950ec9851ad52bf",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the body of a New World (California) oaked Chardonnay at 14% ABV on the OGS 0-10 body scale (0 = water-thin, 10 = syrupy). Respond with only a single decimal number.\n\nINPUT:\n{\n  \"beverage\": \"oaked California Chardonnay, 14% ABV\",\n  \"attribute\": \"body\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 7.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 7.5,
            "expected": 7.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 420,
            "output_tokens": 7,
            "latency_ms": 1745.5,
            "usd": 0.000364,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 7.5,
            "expected": 7.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 420,
            "output_tokens": 7,
            "latency_ms": 559.6,
            "usd": 0.000364,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 7.5,
            "expected": 7.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 420,
            "output_tokens": 7,
            "latency_ms": 1913.6,
            "usd": 0.000364,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 420.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1406.233,
        "usd_sum": 0.001092
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-006",
      "category": "sensory",
      "capability": "dominant_aromatic_family",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "d68ef8562addd55d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nIdentify the single dominant aromatic FAMILY (from the OGS vocabulary) of a classic Marlborough Sauvignon Blanc. Respond with just the family name.\n\nINPUT:\n{\n  \"beverage\": \"Marlborough Sauvignon Blanc\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "vegetal"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fruit",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "fruit",
            "expected": "vegetal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 385,
            "output_tokens": 4,
            "latency_ms": 730.6,
            "usd": 0.000324,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "fruit",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "fruit",
            "expected": "vegetal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 385,
            "output_tokens": 4,
            "latency_ms": 1699.1,
            "usd": 0.000324,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "fruit",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "fruit",
            "expected": "vegetal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 385,
            "output_tokens": 4,
            "latency_ms": 653.5,
            "usd": 0.000324,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 385.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1027.733,
        "usd_sum": 0.000972
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-007",
      "category": "sensory",
      "capability": "structural_vector",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "658c14cce434cc0a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nReturn the OGS structural attributes of a Brut Champagne on the 0-10 scale. Respond as a JSON object with keys body, acidity_perception, tannin, astringency, alcohol_heat, effervescence.\n\nINPUT:\n{\n  \"beverage\": \"Brut non-vintage Champagne\"\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"object\",\n  \"required\": [\n    \"body\",\n    \"acidity_perception\",\n    \"tannin\",\n    \"astringency\",\n    \"alcohol_heat\",\n    \"effervescence\"\n  ],\n  \"properties\": {\n    \"body\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"acidity_perception\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"tannin\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"astringency\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"alcohol_heat\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"effervescence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    }\n  },\n  \"additionalProperties\": false\n}",
      "expected": {
        "values": {
          "body": 4.5,
          "acidity_perception": 8.0,
          "tannin": 0.5,
          "astringency": 1.0,
          "alcohol_heat": 3.5,
          "effervescence": 9.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"body\": 4,\n  \"acidity_perception\": 7,\n  \"tannin\": 1,\n  \"astringency\": 1,\n  \"alcohol_heat\": 5,\n  \"effervescence\": 9\n}\n```",
          "compliant": true,
          "score": 0.7083,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 4,
                "expected": 4.5,
                "score": 0.75
              },
              "acidity_perception": {
                "predicted": 7,
                "expected": 8.0,
                "score": 0.5
              },
              "tannin": {
                "predicted": 1,
                "expected": 0.5,
                "score": 0.75
              },
              "astringency": {
                "predicted": 1,
                "expected": 1.0,
                "score": 1.0
              },
              "alcohol_heat": {
                "predicted": 5,
                "expected": 3.5,
                "score": 0.25
              },
              "effervescence": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 686,
            "output_tokens": 69,
            "latency_ms": 2248.1,
            "usd": 0.000825,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"body\": 3,\n  \"acidity_perception\": 7,\n  \"tannin\": 1,\n  \"astringency\": 2,\n  \"alcohol_heat\": 5,\n  \"effervescence\": 9\n}\n```",
          "compliant": true,
          "score": 0.5417,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 3,
                "expected": 4.5,
                "score": 0.25
              },
              "acidity_perception": {
                "predicted": 7,
                "expected": 8.0,
                "score": 0.5
              },
              "tannin": {
                "predicted": 1,
                "expected": 0.5,
                "score": 0.75
              },
              "astringency": {
                "predicted": 2,
                "expected": 1.0,
                "score": 0.5
              },
              "alcohol_heat": {
                "predicted": 5,
                "expected": 3.5,
                "score": 0.25
              },
              "effervescence": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 686,
            "output_tokens": 69,
            "latency_ms": 790.9,
            "usd": 0.000825,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"body\": 3,\n  \"acidity_perception\": 7,\n  \"tannin\": 0,\n  \"astringency\": 1,\n  \"alcohol_heat\": 4,\n  \"effervescence\": 9\n}\n```",
          "compliant": true,
          "score": 0.7083,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 3,
                "expected": 4.5,
                "score": 0.25
              },
              "acidity_perception": {
                "predicted": 7,
                "expected": 8.0,
                "score": 0.5
              },
              "tannin": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              },
              "astringency": {
                "predicted": 1,
                "expected": 1.0,
                "score": 1.0
              },
              "alcohol_heat": {
                "predicted": 4,
                "expected": 3.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 686,
            "output_tokens": 69,
            "latency_ms": 752.5,
            "usd": 0.000825,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6528,
      "score_summary": {
        "mean": 0.6528,
        "stdev": 0.0962,
        "sem": 0.0555,
        "ci95_lo": 0.5439,
        "ci95_hi": 0.7616,
        "n": 3
      },
      "score_if_valid": 0.6528,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 686.0,
        "output_tokens_mean": 69.0,
        "latency_ms_mean": 1263.833,
        "usd_sum": 0.002475
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-008",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "873c6c2f9608e43f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the bitter intensity of raw radicchio on the OGS 0-10 intensity scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"raw radicchio\",\n  \"attribute\": \"bitter\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 6.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 402,
            "output_tokens": 7,
            "latency_ms": 664.7,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 402,
            "output_tokens": 7,
            "latency_ms": 1761.0,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 402,
            "output_tokens": 7,
            "latency_ms": 499.1,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.3333,
        "ci95_hi": 0.3333,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 402.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 974.933,
        "usd_sum": 0.00105
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-009",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "523e04f39252cb4b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the umami intensity of a plain 10-minute cold-steep kombu dashi on the OGS 0-10 scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"kombu dashi (cold steep 10 min, 10 g/L Hokkaido kombu)\",\n  \"attribute\": \"umami\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 6.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "4.5",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 4.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 432,
            "output_tokens": 7,
            "latency_ms": 2168.7,
            "usd": 0.000374,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "3.5",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 3.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 432,
            "output_tokens": 7,
            "latency_ms": 528.5,
            "usd": 0.000374,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "4.5",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 4.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 432,
            "output_tokens": 7,
            "latency_ms": 2314.4,
            "usd": 0.000374,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 432.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1670.533,
        "usd_sum": 0.001122
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-010",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "1813cae53ba17a84",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the bitter intensity of a square of unsweetened 85% cacao dark chocolate on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"single-origin 85% dark chocolate\",\n  \"attribute\": \"bitter\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 6.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 405,
            "output_tokens": 7,
            "latency_ms": 571.3,
            "usd": 0.000352,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 405,
            "output_tokens": 7,
            "latency_ms": 644.7,
            "usd": 0.000352,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 405,
            "output_tokens": 7,
            "latency_ms": 1885.6,
            "usd": 0.000352,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.3333,
        "ci95_hi": 0.3333,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 405.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 1033.867,
        "usd_sum": 0.001056
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-011",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "1886b2e009716d1d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the perceived acidity of freshly squeezed yuzu juice on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"fresh yuzu juice\",\n  \"attribute\": \"acidity_perception\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 9.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "8",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 8.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 5,
            "latency_ms": 689.3,
            "usd": 0.000322,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "8",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 8.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 5,
            "latency_ms": 1788.8,
            "usd": 0.000322,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "8",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 8.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 5,
            "latency_ms": 1244.1,
            "usd": 0.000322,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.3333,
        "ci95_hi": 0.3333,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 378.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1240.733,
        "usd_sum": 0.000966
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-012",
      "category": "sensory",
      "capability": "basic_taste_vector",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "8d40a83f1b501685",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nReturn the five basic tastes of fresh-squeezed Valencia orange juice on the OGS 0-10 scale as a JSON object with keys sweet, sour, salty, bitter, umami. Respond with ONLY the JSON.\n\nINPUT:\n{\n  \"ingredient\": \"freshly squeezed Valencia orange juice\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "values": {
          "sweet": 7.0,
          "sour": 5.5,
          "salty": 0.0,
          "bitter": 0.5,
          "umami": 0.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"sweet\": 6.5,\n  \"sour\": 5.5,\n  \"salty\": 0.5,\n  \"bitter\": 1.0,\n  \"umami\": 0.5\n}\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 6.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 5.5,
                "expected": 5.5,
                "score": 1.0
              },
              "salty": {
                "predicted": 0.5,
                "expected": 0.0,
                "score": 0.75
              },
              "bitter": {
                "predicted": 1.0,
                "expected": 0.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 0.5,
                "expected": 0.0,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 415,
            "output_tokens": 63,
            "latency_ms": 845.7,
            "usd": 0.000584,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"sweet\": 6.5,\n  \"sour\": 5.5,\n  \"salty\": 0.5,\n  \"bitter\": 1.0,\n  \"umami\": 0.5\n}\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 6.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 5.5,
                "expected": 5.5,
                "score": 1.0
              },
              "salty": {
                "predicted": 0.5,
                "expected": 0.0,
                "score": 0.75
              },
              "bitter": {
                "predicted": 1.0,
                "expected": 0.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 0.5,
                "expected": 0.0,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 415,
            "output_tokens": 63,
            "latency_ms": 3529.5,
            "usd": 0.000584,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"sweet\": 6.5,\n  \"sour\": 5.5,\n  \"salty\": 0.5,\n  \"bitter\": 1.0,\n  \"umami\": 0.5\n}\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 6.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 5.5,
                "expected": 5.5,
                "score": 1.0
              },
              "salty": {
                "predicted": 0.5,
                "expected": 0.0,
                "score": 0.75
              },
              "bitter": {
                "predicted": 1.0,
                "expected": 0.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 0.5,
                "expected": 0.0,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 415,
            "output_tokens": 63,
            "latency_ms": 864.4,
            "usd": 0.000584,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.8,
      "score_summary": {
        "mean": 0.8,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8,
        "ci95_hi": 0.8,
        "n": 3
      },
      "score_if_valid": 0.8,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 415.0,
        "output_tokens_mean": 63.0,
        "latency_ms_mean": 1746.533,
        "usd_sum": 0.001752
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-013",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "middle_eastern",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "691f59088f3e98f3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the sweet intensity of a fresh Medjool date on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"Medjool date, soft and ripe\",\n  \"attribute\": \"sweet\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 9.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 401,
            "output_tokens": 7,
            "latency_ms": 711.1,
            "usd": 0.000349,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 401,
            "output_tokens": 7,
            "latency_ms": 710.1,
            "usd": 0.000349,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 401,
            "output_tokens": 7,
            "latency_ms": 1506.3,
            "usd": 0.000349,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 401.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 975.833,
        "usd_sum": 0.001047
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-014",
      "category": "sensory",
      "capability": "dominant_aromatic_family",
      "difficulty": "medium",
      "cuisine_region": "scottish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "c4a83a79265c0e9f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nIdentify the single dominant aromatic FAMILY (OGS vocabulary) of a heavily-peated Islay single malt (e.g. Ardbeg 10).\n\nINPUT:\n{\n  \"item\": \"heavily peated Islay single malt whisky, ~50 ppm phenols\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "wood"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "wood",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "wood",
            "expected": "wood",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 395,
            "output_tokens": 4,
            "latency_ms": 846.6,
            "usd": 0.000332,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "earth",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "earth",
            "expected": "wood",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 395,
            "output_tokens": 4,
            "latency_ms": 532.6,
            "usd": 0.000332,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "wood",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "wood",
            "expected": "wood",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 395,
            "output_tokens": 4,
            "latency_ms": 570.7,
            "usd": 0.000332,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": 0.0133,
        "ci95_hi": 1.32,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 395.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 649.967,
        "usd_sum": 0.000996
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-015",
      "category": "sensory",
      "capability": "structural_vector",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "66d64ca2254c09ee",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nReturn the OGS structural attributes of a 5-year-old Amarone della Valpolicella on the 0-10 scale. Respond with a JSON object with keys body, acidity_perception, tannin, astringency, alcohol_heat, effervescence. JSON only.\n\nINPUT:\n{\n  \"item\": \"Amarone della Valpolicella Classico, 5 years old, 16% abv\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "values": {
          "body": 9.0,
          "acidity_perception": 6.0,
          "tannin": 7.0,
          "astringency": 5.5,
          "alcohol_heat": 8.5,
          "effervescence": 0.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"body\": 9,\n  \"acidity_perception\": 4,\n  \"tannin\": 8,\n  \"astringency\": 7,\n  \"alcohol_heat\": 8,\n  \"effervescence\": 0\n}\n```",
          "compliant": true,
          "score": 0.5833,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              },
              "acidity_perception": {
                "predicted": 4,
                "expected": 6.0,
                "score": 0.0
              },
              "tannin": {
                "predicted": 8,
                "expected": 7.0,
                "score": 0.5
              },
              "astringency": {
                "predicted": 7,
                "expected": 5.5,
                "score": 0.25
              },
              "alcohol_heat": {
                "predicted": 8,
                "expected": 8.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 423,
            "output_tokens": 69,
            "latency_ms": 988.2,
            "usd": 0.000614,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"body\": 9,\n  \"acidity_perception\": 4,\n  \"tannin\": 7,\n  \"astringency\": 6,\n  \"alcohol_heat\": 8,\n  \"effervescence\": 0\n}\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              },
              "acidity_perception": {
                "predicted": 4,
                "expected": 6.0,
                "score": 0.0
              },
              "tannin": {
                "predicted": 7,
                "expected": 7.0,
                "score": 1.0
              },
              "astringency": {
                "predicted": 6,
                "expected": 5.5,
                "score": 0.75
              },
              "alcohol_heat": {
                "predicted": 8,
                "expected": 8.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 423,
            "output_tokens": 69,
            "latency_ms": 1228.8,
            "usd": 0.000614,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"body\": 9,\n  \"acidity_perception\": 4,\n  \"tannin\": 7,\n  \"astringency\": 6,\n  \"alcohol_heat\": 8,\n  \"effervescence\": 0\n}\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              },
              "acidity_perception": {
                "predicted": 4,
                "expected": 6.0,
                "score": 0.0
              },
              "tannin": {
                "predicted": 7,
                "expected": 7.0,
                "score": 1.0
              },
              "astringency": {
                "predicted": 6,
                "expected": 5.5,
                "score": 0.75
              },
              "alcohol_heat": {
                "predicted": 8,
                "expected": 8.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 423,
            "output_tokens": 69,
            "latency_ms": 1902.5,
            "usd": 0.000614,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6944,
      "score_summary": {
        "mean": 0.6944,
        "stdev": 0.0962,
        "sem": 0.0556,
        "ci95_lo": 0.5855,
        "ci95_hi": 0.8033,
        "n": 3
      },
      "score_if_valid": 0.6944,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 423.0,
        "output_tokens_mean": 69.0,
        "latency_ms_mean": 1373.167,
        "usd_sum": 0.001842
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-016",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "spanish",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "70318c6fd3b033d2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the umami intensity of 24-month acorn-finished jamón ibérico de bellota on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"jamón ibérico de bellota, 24 months, acorn-finished\",\n  \"attribute\": \"umami\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 8.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 8.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 420,
            "output_tokens": 7,
            "latency_ms": 1674.5,
            "usd": 0.000364,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 8.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 420,
            "output_tokens": 7,
            "latency_ms": 634.5,
            "usd": 0.000364,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 8.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 420,
            "output_tokens": 7,
            "latency_ms": 550.0,
            "usd": 0.000364,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 420.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 953.0,
        "usd_sum": 0.001092
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-017",
      "category": "sensory",
      "capability": "dominant_aromatic_family",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "c3cedde89152ce76",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nIdentify the single dominant aromatic FAMILY of fresh Thai basil (horapa) leaves.\n\nINPUT:\n{\n  \"item\": \"fresh Thai basil (horapa), whole leaves\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "herbal"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 367,
            "output_tokens": 5,
            "latency_ms": 4169.6,
            "usd": 0.000314,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 367,
            "output_tokens": 5,
            "latency_ms": 520.0,
            "usd": 0.000314,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 367,
            "output_tokens": 5,
            "latency_ms": 484.2,
            "usd": 0.000314,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 367.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1724.6,
        "usd_sum": 0.000942
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-018",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "7d31de11173a7fd2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the body of a dry Provençal rosé (e.g. Côtes de Provence Grenache-led) on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"item\": \"dry Provençal rosé, 12.5% abv, minimal skin contact\",\n  \"attribute\": \"body\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 3.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "3",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 3.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 405,
            "output_tokens": 5,
            "latency_ms": 677.6,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "3",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 3.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 405,
            "output_tokens": 5,
            "latency_ms": 632.1,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "3.5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.5,
            "expected": 3.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 405,
            "output_tokens": 7,
            "latency_ms": 875.1,
            "usd": 0.000352,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.7778,
      "score_summary": {
        "mean": 0.7778,
        "stdev": 0.1924,
        "sem": 0.1111,
        "ci95_lo": 0.56,
        "ci95_hi": 0.9956,
        "n": 3
      },
      "score_if_valid": 0.7778,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 405.0,
        "output_tokens_mean": 5.667,
        "latency_ms_mean": 728.267,
        "usd_sum": 0.00104
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-019",
      "category": "sensory",
      "capability": "basic_taste_vector",
      "difficulty": "hard",
      "cuisine_region": "korean",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "2e7df899a0148ebd",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nReturn the five basic tastes of well-fermented napa kimchi (6 weeks at 4°C after 2 days at room temp) as a JSON object.\n\nINPUT:\n{\n  \"ingredient\": \"napa kimchi, 6 weeks fermented, ripe\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "values": {
          "sweet": 1.0,
          "sour": 7.0,
          "salty": 6.0,
          "bitter": 1.5,
          "umami": 7.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"sweet\": 1.5,\n  \"sour\": 7.5,\n  \"salty\": 7.0,\n  \"bitter\": 1.0,\n  \"umami\": 7.5\n}\n```",
          "compliant": true,
          "score": 0.7,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 1.5,
                "expected": 1.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 7.0,
                "expected": 6.0,
                "score": 0.5
              },
              "bitter": {
                "predicted": 1.0,
                "expected": 1.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 63,
            "latency_ms": 2158.6,
            "usd": 0.000577,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"sweet\": 1.5,\n  \"sour\": 7.5,\n  \"salty\": 6.5,\n  \"bitter\": 1.0,\n  \"umami\": 7.0\n}\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 1.5,
                "expected": 1.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 6.5,
                "expected": 6.0,
                "score": 0.75
              },
              "bitter": {
                "predicted": 1.0,
                "expected": 1.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 7.0,
                "expected": 7.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 63,
            "latency_ms": 857.8,
            "usd": 0.000577,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"sweet\": 1.5,\n  \"sour\": 7.5,\n  \"salty\": 7.0,\n  \"bitter\": 1.0,\n  \"umami\": 7.5\n}\n```",
          "compliant": true,
          "score": 0.7,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 1.5,
                "expected": 1.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 7.0,
                "expected": 6.0,
                "score": 0.5
              },
              "bitter": {
                "predicted": 1.0,
                "expected": 1.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 63,
            "latency_ms": 814.4,
            "usd": 0.000577,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.7333,
      "score_summary": {
        "mean": 0.7333,
        "stdev": 0.0577,
        "sem": 0.0333,
        "ci95_lo": 0.668,
        "ci95_hi": 0.7987,
        "n": 3
      },
      "score_if_valid": 0.7333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 406.0,
        "output_tokens_mean": 63.0,
        "latency_ms_mean": 1276.933,
        "usd_sum": 0.001731
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-020",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "portuguese",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "ca5b9ab85fe71acb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the effervescence of a typical young Vinho Verde (traditional style with slight spritz) on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"item\": \"young Vinho Verde, slightly spritzig\",\n  \"attribute\": \"effervescence\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 2.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "3",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 2.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 390,
            "output_tokens": 5,
            "latency_ms": 657.5,
            "usd": 0.000332,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "3",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 2.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 390,
            "output_tokens": 5,
            "latency_ms": 2386.1,
            "usd": 0.000332,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "3",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 2.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 390,
            "output_tokens": 5,
            "latency_ms": 543.5,
            "usd": 0.000332,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 390.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1195.7,
        "usd_sum": 0.000996
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-001",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "4523d68b7da194f7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'fleur de sel' as an OGS ingredient. It MUST validate against the OGS ingredient schema (schema/ingredient.schema.json). Respond with ONLY the JSON object, no markdown fences.\n\nINPUT:\n{\n  \"ingredient_name\": \"Fleur de sel\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"fleur-de-sel\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:fleur-de-sel",
          "ogs_version": "0.1.0",
          "category": "salt"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:fleur-de-sel\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Fleur de sel\",\n  \"description\": \"A delicate, hand-harvested sea salt with a subtle mineral character and naturally occurring trace minerals. Known for its light crystalline structure and ability to dissolve slowly on the palate.\",\n  \"category\": \"salt\",\n  \"subcategory\": \"sea salt\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 0,\n      \"sour\": 0,\n      \"salty\": 9.5,\n      \"bitter\": 0,\n      \"umami\": 1\n    },\n    \"structural\": {\n      \"body\": 2,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"sea mineral\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\n      \"crystalline\",\n      \"delicate\",\n      \"crunchy\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 4\n  },\n  \"origin\": {\n    \"region\": \"Guérande, France\",\n    \"season\": \"Summer months (May-September)\",\n    \"terroir_notes\": \"Harvested from Atlantic coastal salt marshes using traditional hand-raking methods\"\n  },\n  \"notes\": \"Premium finishing salt that adds texture and enhances flavors without overwhelming dishes. Best used at the end of cooking or as a final garnish.\",\n  \"metadata\": {\n    \"source\": \"OGS Core Database\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:fleur-de-sel",
                "expected": "ogs:core:ingredient:fleur-de-sel",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "salt",
                "expected": "salt",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3080,
            "output_tokens": 491,
            "latency_ms": 4137.2,
            "usd": 0.004428,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:fleur-de-sel\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Fleur de sel\",\n  \"description\": \"A delicate, hand-harvested sea salt from the surface of evaporation ponds, known for its mineral complexity and subtle briny character.\",\n  \"category\": \"salt\",\n  \"subcategory\": \"sea salt\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 0,\n      \"sour\": 0,\n      \"salty\": 9.5,\n      \"bitter\": 0,\n      \"umami\": 1.5\n    },\n    \"structural\": {\n      \"body\": 2,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"mineral\",\n        \"item\": \"sea spray\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"iodine\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\n      \"crystalline\",\n      \"delicate\",\n      \"crunchy\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 5\n  },\n  \"origin\": {\n    \"region\": \"Brittany, France (Guérande)\",\n    \"season\": \"Summer months\",\n    \"terroir_notes\": \"Harvested from natural salt marshes with specific tidal and climatic conditions that contribute to mineral content.\"\n  },\n  \"notes\": \"Fleur de sel crystals are finest and most delicate when hand-harvested before settling to the pond bottom. Best used as finishing salt to preserve its delicate structure.\",\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:fleur-de-sel",
                "expected": "ogs:core:ingredient:fleur-de-sel",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "salt",
                "expected": "salt",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3080,
            "output_tokens": 537,
            "latency_ms": 4234.7,
            "usd": 0.004612,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:fleur-de-sel\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"fr\",\n  \"available_languages\": [\"fr\", \"en\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"Fleur de sel\",\n        \"language\": \"fr\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Flower of salt\",\n        \"language\": \"en\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"description\": {\n    \"labels\": [\n      {\n        \"value\": \"Fleur de sel est une fine couche de cristaux de sel qui se forme à la surface des marais salants. C'est un sel marin naturel, non raffiné, récolté à la main et caractérisé par sa texture délicate et sa saveur subtile.\",\n        \"language\": \"fr\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Fleur de sel is a thin layer of delicate salt crystals that forms on the surface of salt marshes. It is a natural, unrefined sea salt hand-harvested and prized for its delicate texture and subtle flavor.\",\n        \"language\": \"en\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"category\": \"salt\",\n  \"subcategory\": \"sea salt\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 0,\n      \"sour\": 0,\n      \"salty\": 9.5,\n      \"bitter\": 0,\n      \"umami\": 0.5\n    },\n    \"structural\": {\n      \"body\": 2,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"sea spray\",\n        \"item\": \"ocean mineral\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"chalk\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\n      \"crystalline\",\n      \"delicate\",\n      \"crunchy\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 4\n  },\n  \"origin\": {\n    \"region\": {\n      \"labels\": [\n        {\n          \"value\": \"Guérande, Bretagne\",\n          \"language\": \"fr\",\n          \"role\": \"preferred\"\n        },\n        {\n          \"value\": \"Guérande, Brittany\",\n          \"language\": \"en\",\n          \"role\": \"alternate\"\n        }\n      ]\n    },\n    \"season\": {\n      \"labels\": [\n        {\n          \"value\": \"Été\",\n          \"language\": \"fr\",\n          \"role\": \"preferred\"\n        },\n        {\n          \"value\": \"Summer\",\n          \"language\": \"en\",\n          \"role\": \"alternate\"\n        }\n      ]\n    },\n    \"terroir_notes\": {\n      \"labels\": [\n        {\n          \"value\": \"Récoltée à la main dans les marais salants côtiers, formée par l'évaporation naturelle de l'eau de mer sous le climat atlantique.\",\n          \"language\": \"fr\",\n          \"role\": \"preferred\"\n        },\n        {\n          \"value\": \"Hand-harvested from coastal salt marshes, formed by natural evaporation of seawater under Atlantic climate conditions.\",\n          \"language\": \"en\",\n          \"role\": \"alternate\"\n        }\n      ]\n    }\n  },\n  \"notes\": {\n    \"labels\": [\n      {\n        \"value\": \"Fleur de sel est prisée pour son utilisation en finition culinaire. Sa structure cristalline délicate se dissout moins rapidement que le sel de table, préservant sa saveur subtile et sa texture croquante.\",\n        \"language\":",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "['fr', 'en'] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:ingredient:fleur-de-sel",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              },
              "category": {
                "predicted": null,
                "expected": "salt",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3080,
            "output_tokens": 1024,
            "latency_ms": 6748.3,
            "usd": 0.00656,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": 0.0133,
        "ci95_hi": 1.32,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3080.0,
        "output_tokens_mean": 684.0,
        "latency_ms_mean": 5040.067,
        "usd_sum": 0.0156
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-002",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "367b58556573440d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing a Sancerre (Sauvignon Blanc, Loire Valley) that validates against the OGS beverage schema. Respond with ONLY the JSON object, no markdown fences.\n\nINPUT:\n{\n  \"beverage_name\": \"Sancerre\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"sancerre-generic\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:sancerre-generic",
          "ogs_version": "0.1.0",
          "type": "wine"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:sancerre-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Sancerre\",\n  \"description\": \"A crisp and minerally Sauvignon Blanc from the Loire Valley of France, known for its herbaceous and tropical fruit aromas with bright acidity.\",\n  \"type\": \"wine\",\n  \"subtype\": \"Sauvignon Blanc\",\n  \"grape_varieties\": [\"Sauvignon Blanc\"],\n  \"region\": \"Loire Valley\",\n  \"appellation\": \"Sancerre\",\n  \"abv\": 12.5,\n  \"residual_sugar\": 1.5,\n  \"serve_temperature_c\": {\n    \"min\": 7,\n    \"max\": 10\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 7.5,\n      \"salty\": 1.0,\n      \"bitter\": 1.0,\n      \"umami\": 0.5\n    },\n    \"structural\": {\n      \"body\": 3.5,\n      \"acidity_perception\": 7.5,\n      \"tannin\": 0.5,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 2.0,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"tropical\",\n        \"item\": \"passion fruit\",\n        \"intensity\": 7.0\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"zest\",\n        \"intensity\": 8.0\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"green\",\n        \"item\": \"grass\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"green\",\n        \"item\": \"bell pepper\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"flint\",\n        \"intensity\": 6.0\n      }\n    ],\n    \"texture\": [\"crisp\", \"clean\", \"mineral\"],\n    \"finish_length\": 6.5,\n    \"complexity\": 6.0\n  },\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:sancerre-generic",
                "expected": "ogs:core:beverage:sancerre-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3088,
            "output_tokens": 664,
            "latency_ms": 4527.7,
            "usd": 0.005126,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:sancerre-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Sancerre\",\n  \"description\": \"A crisp and mineral-driven white wine from the Loire Valley, made from Sauvignon Blanc grapes. Known for its bright acidity, citrus notes, and distinctive terroir expression.\",\n  \"type\": \"wine\",\n  \"subtype\": \"Sauvignon Blanc\",\n  \"grape_varieties\": [\n    \"Sauvignon Blanc\"\n  ],\n  \"region\": \"Loire Valley\",\n  \"appellation\": \"Sancerre\",\n  \"abv\": 12.5,\n  \"residual_sugar\": 1.5,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 12\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 7.5,\n      \"salty\": 1.0,\n      \"bitter\": 0.5,\n      \"umami\": 0.0\n    },\n    \"structural\": {\n      \"body\": 4.0,\n      \"acidity_perception\": 8.0,\n      \"tannin\": 0.0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 2.5,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"grapefruit\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 8.0\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"green\",\n        \"item\": \"grass\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"green\",\n        \"item\": \"bell pepper\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"white flowers\",\n        \"item\": \"honeysuckle\",\n        \"intensity\": 4.0\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"flint\",\n        \"intensity\": 6.5\n      }\n    ],\n    \"texture\": [\n      \"crisp\",\n      \"mineral\",\n      \"bright\"\n    ],\n    \"finish_length\": 6.5,\n    \"complexity\": 6.0\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:sancerre-generic",
                "expected": "ogs:core:beverage:sancerre-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3088,
            "output_tokens": 707,
            "latency_ms": 4433.8,
            "usd": 0.005298,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:sancerre-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Sancerre\",\n  \"type\": \"wine\",\n  \"subtype\": \"Sauvignon Blanc\",\n  \"grape_varieties\": [\"Sauvignon Blanc\"],\n  \"region\": \"Loire Valley\",\n  \"appellation\": \"Sancerre\",\n  \"abv\": 12.5,\n  \"residual_sugar\": 1.5,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 12\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 6.5,\n      \"salty\": 1,\n      \"bitter\": 1.5,\n      \"umami\": 0.5\n    },\n    \"structural\": {\n      \"body\": 4,\n      \"acidity_perception\": 7,\n      \"tannin\": 0.5,\n      \"astringency\": 1,\n      \"alcohol_heat\": 3,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"grapefruit\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone\",\n        \"item\": \"green apple\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"herbs\",\n        \"item\": \"grass\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"vegetables\",\n        \"item\": \"green bell pepper\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"flowers\",\n        \"item\": \"white flowers\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\"crisp\", \"clean\"],\n    \"finish_length\": 6,\n    \"complexity\": 6\n  },\n  \"description\": \"A crisp and mineral-driven Sauvignon Blanc from the Loire Valley with vibrant citrus and herbaceous notes.\",\n  \"primary_language\": \"en\"\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:sancerre-generic",
                "expected": "ogs:core:beverage:sancerre-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3088,
            "output_tokens": 612,
            "latency_ms": 3868.8,
            "usd": 0.004918,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3088.0,
        "output_tokens_mean": 661.0,
        "latency_ms_mean": 4276.767,
        "usd_sum": 0.015342
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-003",
      "category": "structured_output",
      "capability": "schema_valid_pairing",
      "difficulty": "expert",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.pairing",
        "vocab.match_types",
        "vocab.explanation_codes",
        "vocab.grade_scale",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "2740dc70d150e12a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS pairing schema (`schema/pairing.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/pairing.schema.json\",\n  \"title\": \"OGS Pairing\",\n  \"description\": \"A scored, explained pairing between a dish and a beverage.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"dish_ref\",\n    \"beverage_ref\",\n    \"match_types\",\n    \"overall_score\",\n    \"grade\",\n    \"explanation_codes\",\n    \"confidence\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:pairing:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"dish_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"beverage_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"match_types\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/match_type_entry\"\n      }\n    },\n    \"overall_score\": {\n      \"type\": \"integer\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"grade\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"A+\",\n        \"A\",\n        \"B+\",\n        \"B\",\n        \"C+\",\n        \"C\",\n        \"D\",\n        \"F\"\n      ]\n    },\n    \"explanation_codes\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/explanation_code_entry\"\n      }\n    },\n    \"risks\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/risk_entry\"\n      }\n    },\n    \"confidence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 1\n    },\n    \"evidence\": {\n      \"$ref\": \"#/$defs/evidence\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"match_type_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"type\",\n        \"score\"\n      ],\n      \"properties\": {\n        \"type\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"harmony\",\n            \"contrast\",\n            \"bridge\",\n            \"intensity_alignment\",\n            \"structural_balance\",\n            \"refresh\"\n          ]\n        },\n        \"score\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 100\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"primary\": {\n          \"type\": \"boolean\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"explanation_code_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"weight\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 1\n        },\n        \"detail\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"risk_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\",\n        \"severity\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"severity\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"low\",\n            \"medium\",\n            \"high\"\n          ]\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"mitigation\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"evidence\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"expert_panel\",\n            \"instrumental\",\n            \"computed\",\n            \"estimated\"\n          ]\n        },\n        \"sources\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Explanation Codes\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS pairing JSON document for the described match. It MUST validate against the OGS pairing schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"pairing_slug\": \"brie-champagne\",\n  \"dish_ref\": \"ogs:core:dish:brie-on-baguette\",\n  \"beverage_ref\": \"ogs:core:beverage:brut-champagne-nv\",\n  \"intent\": \"A classic refresh pairing driven by effervescence and acidity against creamy brie.\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/pairing.schema.json",
        "required_fields": {
          "id": "ogs:core:pairing:brie-champagne",
          "ogs_version": "0.1.0",
          "dish_ref": "ogs:core:dish:brie-on-baguette",
          "beverage_ref": "ogs:core:beverage:brut-champagne-nv"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:brie-champagne\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Brie on Baguette with Brut Champagne NV\",\n  \"dish_ref\": \"ogs:core:dish:brie-on-baguette\",\n  \"beverage_ref\": \"ogs:core:beverage:brut-champagne-nv\",\n  \"match_types\": [\n    {\n      \"type\": \"refresh\",\n      \"score\": 95,\n      \"primary\": true\n    },\n    {\n      \"type\": \"contrast\",\n      \"score\": 88\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 90\n    }\n  ],\n  \"overall_score\": 91,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"REFRESH_EFFERVESCE\",\n      \"weight\": 0.4,\n      \"detail\": \"Carbonation in Champagne physically cleanses the palate between bites of rich, creamy brie, preventing satiation.\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.35,\n      \"detail\": \"High acidity in brut Champagne cuts through the fat and richness of brie, providing a crisp palate refresh.\"\n    },\n    {\n      \"code\": \"CONTRAST_RICH_CRISP\",\n      \"weight\": 0.15,\n      \"detail\": \"The rich, creamy texture of brie contrasts beautifully with the lean, crisp character of brut Champagne.\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.10,\n      \"detail\": \"The beverage's acidity balances the rich mouthfeel of the dish, creating a harmonious finish.\"\n    }\n  ],\n  \"confidence\": 0.95,\n  \"evidence\": {\n    \"method\": \"expert_panel\",\n    \"sources\": [\n      \"Classic French gastronomic tradition\",\n      \"Professional sommelier consensus\"\n    ],\n    \"notes\": \"This pairing represents one of the most enduring food and wine combinations in Western gastronomy, validated across centuries of French culinary practice.\"\n  },\n  \"notes\": \"A timeless pairing where the effervescence and acidity of Champagne provide both palate refreshment and structural balance against the creamy decadence of brie on bread.\",\n  \"metadata\": {\n    \"source\": \"OGS Core Reference Database\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:brie-champagne",
                "expected": "ogs:core:pairing:brie-champagne",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:brie-on-baguette",
                "expected": "ogs:core:dish:brie-on-baguette",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:brut-champagne-nv",
                "expected": "ogs:core:beverage:brut-champagne-nv",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4199,
            "output_tokens": 660,
            "latency_ms": 5028.9,
            "usd": 0.005999,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:brie-champagne\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Brie on Baguette with Brut Champagne NV\",\n  \"dish_ref\": \"ogs:core:dish:brie-on-baguette\",\n  \"beverage_ref\": \"ogs:core:beverage:brut-champagne-nv\",\n  \"match_types\": [\n    {\n      \"type\": \"refresh\",\n      \"score\": 95,\n      \"primary\": true\n    },\n    {\n      \"type\": \"contrast\",\n      \"score\": 88\n    },\n    {\n      \"type\": \"harmony\",\n      \"score\": 85\n    }\n  ],\n  \"overall_score\": 90,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"REFRESH_EFFERVESCE\",\n      \"weight\": 0.45,\n      \"detail\": \"Champagne's fine carbonation physically cleanses the palate between bites of rich, creamy brie, preventing flavor fatigue.\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.35,\n      \"detail\": \"Brut Champagne's high acidity cuts through the butter fat in brie, providing palate refreshment and brightness.\"\n    },\n    {\n      \"code\": \"CONTRAST_RICH_CRISP\",\n      \"weight\": 0.20,\n      \"detail\": \"The creamy, dense texture of brie contrasts elegantly with the crisp, lean character of dry Champagne.\"\n    }\n  ],\n  \"confidence\": 0.92,\n  \"evidence\": {\n    \"method\": \"expert_panel\",\n    \"sources\": [\n      \"Classic French gastronomic tradition\",\n      \"Sommelier consensus\"\n    ],\n    \"notes\": \"This pairing is historically documented and widely recognized as one of the most successful cheese-wine combinations.\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:brie-champagne",
                "expected": "ogs:core:pairing:brie-champagne",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:brie-on-baguette",
                "expected": "ogs:core:dish:brie-on-baguette",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:brut-champagne-nv",
                "expected": "ogs:core:beverage:brut-champagne-nv",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4199,
            "output_tokens": 504,
            "latency_ms": 3796.9,
            "usd": 0.005375,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:brie-champagne\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Brie on Baguette with Brut Champagne NV\",\n  \"dish_ref\": \"ogs:core:dish:brie-on-baguette\",\n  \"beverage_ref\": \"ogs:core:beverage:brut-champagne-nv\",\n  \"match_types\": [\n    {\n      \"type\": \"refresh\",\n      \"score\": 95,\n      \"primary\": true,\n      \"description\": \"Effervescence and acidity cut through the creamy richness of brie\"\n    },\n    {\n      \"type\": \"contrast\",\n      \"score\": 85,\n      \"description\": \"Crisp, dry champagne contrasts with rich, buttery brie\"\n    },\n    {\n      \"type\": \"harmony\",\n      \"score\": 80,\n      \"description\": \"Regional affinity: both are classic French products with traditional production methods\"\n    }\n  ],\n  \"overall_score\": 92,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"REFRESH_EFFERVESCE\",\n      \"weight\": 0.5,\n      \"detail\": \"Champagne's effervescence physically cleanses the palate between bites of rich brie\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.4,\n      \"detail\": \"High acidity in brut champagne cuts through the fat and creaminess of brie\"\n    },\n    {\n      \"code\": \"CONTRAST_RICH_CRISP\",\n      \"weight\": 0.3,\n      \"detail\": \"The dense, creamy texture of brie is balanced by champagne's crisp, lean character\"\n    },\n    {\n      \"code\": \"HARMONY_REGIONAL\",\n      \"weight\": 0.2,\n      \"detail\": \"Both brie and champagne originate from France, sharing terroir and traditional production methods\"\n    }\n  ],\n  \"confidence\": 0.95,\n  \"evidence\": {\n    \"method\": \"expert_panel\",\n    \"notes\": \"Classic pairing validated across hospitality and wine service traditions\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:brie-champagne",
                "expected": "ogs:core:pairing:brie-champagne",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:brie-on-baguette",
                "expected": "ogs:core:dish:brie-on-baguette",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:brut-champagne-nv",
                "expected": "ogs:core:beverage:brut-champagne-nv",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4199,
            "output_tokens": 571,
            "latency_ms": 3773.1,
            "usd": 0.005643,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 4199.0,
        "output_tokens_mean": 578.333,
        "latency_ms_mean": 4199.633,
        "usd_sum": 0.017017
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-004",
      "category": "structured_output",
      "capability": "sensory_profile_json",
      "difficulty": "medium",
      "cuisine_region": "indian",
      "scorer": "json_match",
      "context_needs": [
        "scales.basic_tastes",
        "scales.structural",
        "vocab.aromatic_families",
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "9bd6860a8c5c69ec",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nProduce a JSON object representing the sensory profile for ripe mango. The object must have keys basic_tastes (with sweet, sour, salty, bitter, umami), structural (with body, acidity_perception, tannin, astringency, alcohol_heat, effervescence), and aromatics (array of entries with family, group, item, intensity). All intensities on 0-10. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"item\": \"ripe Alphonso mango, peak ripeness\"\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"object\",\n  \"required\": [\n    \"basic_tastes\",\n    \"structural\",\n    \"aromatics\"\n  ],\n  \"properties\": {\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"sweet\",\n        \"sour\",\n        \"salty\",\n        \"bitter\",\n        \"umami\"\n      ]\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"body\",\n        \"acidity_perception\",\n        \"tannin\",\n        \"astringency\",\n        \"alcohol_heat\",\n        \"effervescence\"\n      ]\n    },\n    \"aromatics\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"family\",\n          \"group\",\n          \"item\",\n          \"intensity\"\n        ]\n      }\n    }\n  }\n}",
      "expected": {
        "value": {
          "basic_tastes": {
            "sweet": 8.5,
            "sour": 3.0,
            "salty": 0,
            "bitter": 0.5,
            "umami": 0.5
          },
          "structural": {
            "body": 5.5,
            "acidity_perception": 3.0,
            "tannin": 0,
            "astringency": 0.5,
            "alcohol_heat": 0,
            "effervescence": 0
          }
        },
        "required_keys": [
          "basic_tastes.sweet",
          "basic_tastes.sour",
          "structural.body",
          "structural.acidity_perception",
          "structural.tannin"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 8.5,\n    \"sour\": 2.0,\n    \"salty\": 0.0,\n    \"bitter\": 0.0,\n    \"umami\": 0.5\n  },\n  \"structural\": {\n    \"body\": 6.0,\n    \"acidity_perception\": 2.0,\n    \"tannin\": 0.0,\n    \"astringency\": 0.0,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": [\n    {\n      \"family\": \"fruit\",\n      \"group\": \"stone fruit\",\n      \"item\": \"mango\",\n      \"intensity\": 9.0\n    },\n    {\n      \"family\": \"fruit\",\n      \"group\": \"tropical\",\n      \"item\": \"pineapple\",\n      \"intensity\": 6.0\n    },\n    {\n      \"family\": \"fruit\",\n      \"group\": \"citrus\",\n      \"item\": \"peach\",\n      \"intensity\": 5.0\n    },\n    {\n      \"family\": \"floral\",\n      \"group\": \"floral\",\n      \"item\": \"jasmine\",\n      \"intensity\": 3.5\n    },\n    {\n      \"family\": \"confection\",\n      \"group\": \"honey\",\n      \"item\": \"honey\",\n      \"intensity\": 4.0\n    }\n  ]\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.sweet": {
                "predicted": 8.5,
                "expected": 8.5,
                "match": true
              },
              "basic_tastes.sour": {
                "predicted": 2.0,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 6.0,
                "expected": 5.5,
                "match": true
              },
              "structural.acidity_perception": {
                "predicted": 2.0,
                "expected": 3.0,
                "match": true
              },
              "structural.tannin": {
                "predicted": 0.0,
                "expected": 0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 829,
            "output_tokens": 370,
            "latency_ms": 2235.5,
            "usd": 0.002143,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 8.5,\n    \"sour\": 2.0,\n    \"salty\": 0.0,\n    \"bitter\": 0.0,\n    \"umami\": 0.5\n  },\n  \"structural\": {\n    \"body\": 4.0,\n    \"acidity_perception\": 2.0,\n    \"tannin\": 0.0,\n    \"astringency\": 0.0,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": [\n    {\n      \"family\": \"fruit\",\n      \"group\": \"tropical\",\n      \"item\": \"mango\",\n      \"intensity\": 9.5\n    },\n    {\n      \"family\": \"fruit\",\n      \"group\": \"stone\",\n      \"item\": \"peach\",\n      \"intensity\": 5.0\n    },\n    {\n      \"family\": \"floral\",\n      \"group\": \"white_flowers\",\n      \"item\": \"jasmine\",\n      \"intensity\": 3.5\n    },\n    {\n      \"family\": \"fruit\",\n      \"group\": \"citrus\",\n      \"item\": \"orange\",\n      \"intensity\": 4.0\n    },\n    {\n      \"family\": \"confection\",\n      \"group\": \"honey\",\n      \"item\": \"honey\",\n      \"intensity\": 4.5\n    }\n  ]\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.sweet": {
                "predicted": 8.5,
                "expected": 8.5,
                "match": true
              },
              "basic_tastes.sour": {
                "predicted": 2.0,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 4.0,
                "expected": 5.5,
                "match": true
              },
              "structural.acidity_perception": {
                "predicted": 2.0,
                "expected": 3.0,
                "match": true
              },
              "structural.tannin": {
                "predicted": 0.0,
                "expected": 0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 829,
            "output_tokens": 367,
            "latency_ms": 2181.7,
            "usd": 0.002131,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 8.5,\n    \"sour\": 2.0,\n    \"salty\": 0.0,\n    \"bitter\": 0.0,\n    \"umami\": 0.5\n  },\n  \"structural\": {\n    \"body\": 7.0,\n    \"acidity_perception\": 2.0,\n    \"tannin\": 0.0,\n    \"astringency\": 0.0,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": [\n    {\n      \"family\": \"fruit\",\n      \"group\": \"tropical\",\n      \"item\": \"mango\",\n      \"intensity\": 9.5\n    },\n    {\n      \"family\": \"fruit\",\n      \"group\": \"stone fruit\",\n      \"item\": \"peach\",\n      \"intensity\": 6.0\n    },\n    {\n      \"family\": \"floral\",\n      \"group\": \"floral\",\n      \"item\": \"jasmine\",\n      \"intensity\": 4.0\n    },\n    {\n      \"family\": \"fruit\",\n      \"group\": \"citrus\",\n      \"item\": \"orange blossom\",\n      \"intensity\": 3.5\n    },\n    {\n      \"family\": \"confection\",\n      \"group\": \"confection\",\n      \"item\": \"honey\",\n      \"intensity\": 5.0\n    }\n  ]\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.sweet": {
                "predicted": 8.5,
                "expected": 8.5,
                "match": true
              },
              "basic_tastes.sour": {
                "predicted": 2.0,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 7.0,
                "expected": 5.5,
                "match": true
              },
              "structural.acidity_perception": {
                "predicted": 2.0,
                "expected": 3.0,
                "match": true
              },
              "structural.tannin": {
                "predicted": 0.0,
                "expected": 0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 829,
            "output_tokens": 371,
            "latency_ms": 2233.9,
            "usd": 0.002147,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 829.0,
        "output_tokens_mean": 369.333,
        "latency_ms_mean": 2217.033,
        "usd_sum": 0.006421
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-005",
      "category": "structured_output",
      "capability": "schema_valid_dish",
      "difficulty": "expert",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.dish",
        "vocab.component_roles",
        "vocab.cooking_methods",
        "vocab.courses",
        "vocab.service_temperatures",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "9caf9bc474b2464d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS dish schema (`schema/dish.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/dish.schema.json\",\n  \"title\": \"OGS Dish\",\n  \"description\": \"A composed dish with components, cooking methods, and overall sensory profile.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"components\",\n    \"overall_sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"cuisine\": {\n      \"type\": \"string\",\n      \"description\": \"Free-text cuisine name (v0.1) OR a cuisine entity reference matching ogs:<ns>:cuisine:<id> (v0.2+).\"\n    },\n    \"course\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"amuse\",\n        \"appetizer\",\n        \"soup\",\n        \"salad\",\n        \"fish\",\n        \"main\",\n        \"cheese\",\n        \"dessert\",\n        \"petit_four\"\n      ]\n    },\n    \"service_temperature\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"frozen\",\n        \"cold\",\n        \"cool\",\n        \"room\",\n        \"warm\",\n        \"hot\",\n        \"very_hot\"\n      ]\n    },\n    \"components\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/component\"\n      }\n    },\n    \"overall_sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"component\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"role\",\n        \"ingredient_ref\"\n      ],\n      \"properties\": {\n        \"role\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"main_protein\",\n            \"main_vegetable\",\n            \"starch\",\n            \"sauce\",\n            \"fat_component\",\n            \"acid_component\",\n            \"aromatic_garnish\",\n            \"seasoning\",\n            \"textural_element\"\n          ]\n        },\n        \"ingredient_ref\": {\n          \"type\": \"string\",\n          \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\"\n        },\n        \"cooking_method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"raw\",\n            \"seared\",\n            \"grilled\",\n            \"roasted\",\n            \"braised\",\n            \"poached\",\n            \"steamed\",\n            \"fried\",\n            \"confit\",\n            \"smoked\",\n            \"fermented\",\n            \"cured\",\n            \"pickled\",\n            \"dehydrated\",\n            \"sous_vide\",\n            \"blanched\",\n            \"sauteed\",\n            \"baked\",\n            \"charred\",\n            \"emulsified\"\n          ]\n        },\n        \"maillard_index\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"reduction_level\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"proportion\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"dominant\",\n            \"significant\",\n            \"accent\",\n            \"trace\"\n          ]\n        },\n        \"sensory_contribution\": {\n          \"type\": \"object\",\n          \"properties\": {\n            \"basic_tastes\": {\n              \"$ref\": \"#/$defs/basic_tastes\"\n            },\n            \"aromatics\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"$ref\": \"#/$defs/aromatic_entry\"\n              }\n            },\n            \"texture\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"type\": \"string\"\n              }\n            }\n          },\n          \"additionalProperties\": true\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS dish JSON document for 'moules marinieres' (mussels steamed with white wine, shallots, parsley, butter). It must validate against the OGS dish schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"dish_slug\": \"moules-marinieres\",\n  \"cuisine\": \"french\",\n  \"course\": \"main\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/dish.schema.json",
        "required_fields": {
          "id": "ogs:core:dish:moules-marinieres",
          "ogs_version": "0.1.0"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:moules-marinieres\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Moules Marinières\",\n  \"description\": \"Mussels steamed with white wine, shallots, parsley, and butter\",\n  \"cuisine\": \"french\",\n  \"course\": \"main\",\n  \"service_temperature\": \"hot\",\n  \"components\": [\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:mussels\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 6,\n          \"salty\": 4\n        },\n        \"texture\": [\n          \"tender\",\n          \"briny\"\n        ]\n      }\n    },\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:white-wine\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 5,\n          \"salty\": 2\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"citrus\",\n            \"item\": \"lemon\",\n            \"intensity\": 3\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:shallots\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 3,\n          \"salty\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"allium\",\n            \"item\": \"shallot\",\n            \"intensity\": 5\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"fat_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:butter\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"dairy\",\n            \"group\": \"butter\",\n            \"item\": \"butter\",\n            \"intensity\": 4\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:parsley\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"fresh herbs\",\n            \"item\": \"parsley\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 4\n          }\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 4,\n      \"salty\": 5,\n      \"bitter\": 1,\n      \"umami\": 7\n    },\n    \"structural\": {\n      \"body\": 5,\n      \"acidity_perception\": 4\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"allium\",\n        \"item\": \"shallot\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"fresh herbs\",\n        \"item\": \"parsley\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"marine\",\n        \"group\": \"bivalve\",\n        \"item\": \"mussel\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"l",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'main_protein', 'ingredient_ref': 'ogs:core:ingredient:mussels', 'cooking_method': 'steamed', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'umami': 6, 'salty': 4}, 'texture': ['tender', 'briny']}}, {'role': 'sauce', 'ingredient_ref': 'ogs:core:ingredient:white-wine', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sour': 5, 'salty': 2}, 'aromatics': [{'family': 'fruit', 'group': 'citrus', 'item': 'lemon', 'intensity': 3}]}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:shallots', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sweet': 3, 'salty': 1}, 'aromatics': [{'family': 'savory', 'group': 'allium', 'item': 'shallot', 'intensity': 5}]}}, {'role': 'fat_component', 'ingredient_ref': 'ogs:core:ingredient:butter', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'salty': 1}, 'aromatics': [{'family': 'dairy', 'group': 'butter', 'item': 'butter', 'intensity': 4}]}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:parsley', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'aromatics': [{'family': 'herbal', 'group': 'fresh herbs', 'item': 'parsley', 'qualifier': 'fresh', 'intensity': 4}]}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:moules-marinieres",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3751,
            "output_tokens": 1024,
            "latency_ms": 5353.6,
            "usd": 0.007097,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:moules-marinieres\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": {\n    \"en\": \"Moules Marinières\",\n    \"fr\": \"Moules Marinières\"\n  },\n  \"description\": \"Mussels steamed with white wine, shallots, parsley, and butter\",\n  \"cuisine\": \"french\",\n  \"course\": \"main\",\n  \"service_temperature\": \"hot\",\n  \"components\": [\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:mussels\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 5,\n          \"umami\": 6\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"shellfish\",\n            \"item\": \"briny\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"tender\",\n          \"succulent\"\n        ]\n      }\n    },\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:white-wine\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 4,\n          \"salty\": 2,\n          \"umami\": 2\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"grape\",\n            \"item\": \"white wine\",\n            \"intensity\": 4\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatics_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:shallots\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 1,\n          \"umami\": 3\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"shallot\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"soft\"\n        ]\n      }\n    },\n    {\n      \"role\": \"fat_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:butter\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 1,\n          \"umami\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"dairy\",\n            \"group\": \"butter\",\n            \"item\": \"butter\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\n          \"creamy\",\n          \"rich\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:parsley\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"leaf\",\n            \"item\": \"parsley\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 3\n          }\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1,\n      \"sour\": 3,\n      \"salty\": 5,\n      \"bitter\": 0,\n      \"umami\": 6\n    },\n    \"structural\": {\n      \"body\": 6,\n      \"acidity_perception\": 3\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'main_protein', 'ingredient_ref': 'ogs:core:ingredient:mussels', 'cooking_method': 'steamed', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'salty': 5, 'umami': 6}, 'aromatics': [{'family': 'marine', 'group': 'shellfish', 'item': 'briny', 'intensity': 5}], 'texture': ['tender', 'succulent']}}, {'role': 'sauce', 'ingredient_ref': 'ogs:core:ingredient:white-wine', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sour': 4, 'salty': 2, 'umami': 2}, 'aromatics': [{'family': 'fruit', 'group': 'grape', 'item': 'white wine', 'intensity': 4}]}}, {'role': 'aromatics_component', 'ingredient_ref': 'ogs:core:ingredient:shallots', 'cooking_method': 'steamed', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'salty': 1, 'umami': 3}, 'aromatics': [{'family': 'vegetal', 'group': 'allium', 'item': 'shallot', 'qualifier': 'cooked', 'intensity': 5}], 'texture': ['soft']}}, {'role': 'fat_component', 'ingredient_ref': 'ogs:core:ingredient:butter', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'salty': 1, 'umami': 1}, 'aromatics': [{'family': 'dairy', 'group': 'butter', 'item': 'butter', 'intensity': 4}], 'texture': ['creamy', 'rich']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:parsley', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'aromatics': [{'family': 'herbal', 'group': 'leaf', 'item': 'parsley', 'qualifier': 'fresh', 'intensity': 3}]}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:moules-marinieres",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3751,
            "output_tokens": 1024,
            "latency_ms": 5547.6,
            "usd": 0.007097,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:moules-marinieres\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Moules Marinières\",\n  \"description\": \"Mussels steamed with white wine, shallots, parsley, and butter\",\n  \"cuisine\": \"french\",\n  \"course\": \"main\",\n  \"service_temperature\": \"hot\",\n  \"components\": [\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:mussels\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 6,\n          \"salty\": 4\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"shellfish\",\n            \"item\": \"briny\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"tender\", \"succulent\"]\n      }\n    },\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:white-wine\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 4,\n          \"salty\": 2\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"citrus\",\n            \"item\": \"lemon\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 3\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:shallots\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 3,\n          \"salty\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"allium\",\n            \"item\": \"shallot\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\"soft\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:parsley\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"fresh herbs\",\n            \"item\": \"parsley\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 3\n          }\n        ],\n        \"texture\": [\"crisp\"]\n      }\n    },\n    {\n      \"role\": \"fat_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:butter\",\n      \"cooking_method\": \"emulsified\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 2\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"dairy\",\n            \"group\": \"butter\",\n            \"item\": \"butter\",\n            \"intensity\": 3\n          }\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 3,\n      \"salty\": 4,\n      \"bitter\": 0,\n      \"umami\": 6\n    },\n    \"structural\": {\n      \"body\": 5,\n      \"acidity_perception\": 3\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"shellfish\",\n        \"item\": \"briny\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"fresh herbs\",\n        \"item\": \"parsley\",\n        \"qualifier\": \"fresh\",",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'main_protein', 'ingredient_ref': 'ogs:core:ingredient:mussels', 'cooking_method': 'steamed', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'umami': 6, 'salty': 4}, 'aromatics': [{'family': 'marine', 'group': 'shellfish', 'item': 'briny', 'intensity': 5}], 'texture': ['tender', 'succulent']}}, {'role': 'sauce', 'ingredient_ref': 'ogs:core:ingredient:white-wine', 'cooking_method': 'steamed', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sour': 4, 'salty': 2}, 'aromatics': [{'family': 'fruit', 'group': 'citrus', 'item': 'lemon', 'qualifier': 'fresh', 'intensity': 3}]}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:shallots', 'cooking_method': 'steamed', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sweet': 3, 'salty': 1}, 'aromatics': [{'family': 'savory', 'group': 'allium', 'item': 'shallot', 'qualifier': 'cooked', 'intensity': 4}], 'texture': ['soft']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:parsley', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'aromatics': [{'family': 'herbal', 'group': 'fresh herbs', 'item': 'parsley', 'qualifier': 'fresh', 'intensity': 3}], 'texture': ['crisp']}}, {'role': 'fat_component', 'ingredient_ref': 'ogs:core:ingredient:butter', 'cooking_method': 'emulsified', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'salty': 2}, 'aromatics': [{'family': 'dairy', 'group': 'butter', 'item': 'butter', 'intensity': 3}]}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:moules-marinieres",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3751,
            "output_tokens": 1024,
            "latency_ms": 8192.9,
            "usd": 0.007097,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3751.0,
        "output_tokens_mean": 1024.0,
        "latency_ms_mean": 6364.7,
        "usd_sum": 0.021291
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-006",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "medium",
      "cuisine_region": "iranian",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "bb4d0fcee2d1e751",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'saffron' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"saffron\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"saffron-stigma\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:saffron-stigma",
          "ogs_version": "0.1.0",
          "category": "spice"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:saffron-stigma\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Saffron\",\n  \"description\": \"The dried stigmas of Crocus sativus, prized for their distinctive golden-red color, complex aromatic profile, and subtle earthy-floral notes. Used as a premium spice in cuisines worldwide.\",\n  \"category\": \"spice\",\n  \"subcategory\": \"stigma\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 0.5,\n      \"salty\": 0,\n      \"bitter\": 1.5,\n      \"umami\": 2\n    },\n    \"structural\": {\n      \"body\": 3,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"floral\",\n        \"group\": \"delicate florals\",\n        \"item\": \"saffron flower\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"hay\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"honey-like warmth\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"dried herbs\",\n        \"item\": \"grass\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\n      \"powdery when ground\",\n      \"thread-like when whole\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 8\n  },\n  \"origin\": {\n    \"region\": \"Iran (Kashmir, primary producer)\",\n    \"season\": \"Autumn (September-November)\",\n    \"terroir_notes\": \"Mountain cultivation in high-altitude regions contributes to distinctive flavor complexity and color intensity.\"\n  },\n  \"notes\": \"Saffron is the most expensive spice by weight. A small amount imparts significant flavor and golden-red coloration. Best used in risottos, paellas, and infusions. Quality varies greatly by origin and harvest year.\",\n  \"metadata\": {\n    \"created\": \"2024-01-01T00:00:00Z\",\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:saffron-stigma",
                "expected": "ogs:core:ingredient:saffron-stigma",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3064,
            "output_tokens": 679,
            "latency_ms": 5458.0,
            "usd": 0.005167,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:saffron-stigma\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"Saffron\",\n  \"description\": \"The dried stigmas of Crocus sativus, prized for their distinctive golden-red color, delicate floral aroma, and subtle earthy flavor. One of the world's most expensive spices by weight.\",\n  \"category\": \"spice\",\n  \"subcategory\": \"crocus stigma\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 1.5,\n      \"umami\": 3\n    },\n    \"structural\": {\n      \"body\": 2,\n      \"acidity_perception\": 0.5,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"floral\",\n        \"group\": \"crocus\",\n        \"item\": \"saffron flower\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"hay-like\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"earthy\",\n        \"intensity\": 5\n      }\n    ],\n    \"texture\": [\"powder\", \"thread-like\"],\n    \"finish_length\": 6,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": \"Iran (primary), Kashmir, Spain\",\n    \"season\": \"Autumn\",\n    \"terroir_notes\": \"Iranian saffron, particularly from Khorasan, is renowned for its superior quality and distinctive flavor profile.\"\n  },\n  \"notes\": \"Saffron adds distinctive color, aroma, and subtle flavor to dishes. Commonly used in risotto, paella, and rice-based preparations. Use sparingly due to potency and cost.\",\n  \"metadata\": {\n    \"created\": \"2024-01-01T00:00:00Z\",\n    \"source\": \"OGS Core Ingredient Database\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:saffron-stigma",
                "expected": "ogs:core:ingredient:saffron-stigma",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3064,
            "output_tokens": 631,
            "latency_ms": 4784.1,
            "usd": 0.004975,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:saffron-stigma\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"Saffron\",\n  \"description\": \"The dried stigmas of the Crocus sativus flower, prized for their distinctive golden-red color, complex flavor, and aromatic qualities. Known as the world's most expensive spice by weight.\",\n  \"category\": \"spice\",\n  \"subcategory\": \"floral spice\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 1.5,\n      \"umami\": 3\n    },\n    \"structural\": {\n      \"body\": 3,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"floral\",\n        \"group\": \"crocus\",\n        \"item\": \"saffron floral\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"dried herbs\",\n        \"item\": \"hay\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"metallic\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"slightly peppery\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\"powdery when ground\", \"thread-like when whole\"],\n    \"finish_length\": 6,\n    \"complexity\": 8\n  },\n  \"origin\": {\n    \"region\": \"Iran (Khorasan region)\",\n    \"season\": \"Autumn\",\n    \"terroir_notes\": \"High altitude cultivation in volcanic soils produces superior quality stigmas with deeper color and more complex aroma profile.\"\n  },\n  \"notes\": \"Saffron imparts a distinctive golden-yellow color to dishes. Small quantities are sufficient due to its intense flavor and aroma. Best quality saffron exhibits deep red color with minimal yellow portions. The spice's flavor profile includes honeyed, slightly bitter, and metallic notes.\",\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:saffron-stigma",
                "expected": "ogs:core:ingredient:saffron-stigma",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3064,
            "output_tokens": 652,
            "latency_ms": 5254.8,
            "usd": 0.005059,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3064.0,
        "output_tokens_mean": 654.0,
        "latency_ms_mean": 5165.633,
        "usd_sum": 0.015201
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-007",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "0bc2e49ee6c1093e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'miso (red)' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"miso (red)\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"aka-miso\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:aka-miso",
          "ogs_version": "0.1.0",
          "category": "other"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:aka-miso\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Red Miso\",\n  \"description\": \"A deeply fermented soybean paste from Japan with a rich, complex flavor profile. Red miso is aged longer than lighter varieties, resulting in a darker color and more pronounced umami character.\",\n  \"category\": \"legume\",\n  \"subcategory\": \"fermented soybean paste\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 3,\n      \"salty\": 7,\n      \"bitter\": 2,\n      \"umami\": 8.5\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 3,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"fermented\",\n        \"item\": \"soy\",\n        \"qualifier\": \"fermented\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"mushroom\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"nutmeg\",\n        \"intensity\": 2\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"caramel\",\n        \"item\": \"molasses\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\n      \"thick paste\",\n      \"smooth\",\n      \"dense\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 7.5\n  },\n  \"origin\": {\n    \"region\": \"Japan\",\n    \"terroir_notes\": \"Traditional red miso is typically produced in regions with cooler climates that slow fermentation, developing deeper flavors.\"\n  },\n  \"notes\": \"Red miso is typically aged for 1-3 years. It is saltier and more pungent than lighter miso varieties, making it ideal for robust broths, dressings, and marinades.\",\n  \"metadata\": {\n    \"source\": \"OGS gastronomy standard\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:aka-miso",
                "expected": "ogs:core:ingredient:aka-miso",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "legume",
                "expected": "other",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3065,
            "output_tokens": 643,
            "latency_ms": 5204.5,
            "usd": 0.005024,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:aka-miso\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"red miso\",\n  \"description\": \"A fermented soybean paste with a deep red color, rich umami flavor, and complex aged character. Red miso is typically aged longer than lighter varieties, resulting in a more robust and pronounced taste profile.\",\n  \"category\": \"legume\",\n  \"subcategory\": \"fermented soybean paste\",\n  \"primary_language\": \"en\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 3,\n      \"salty\": 7,\n      \"bitter\": 1.5,\n      \"umami\": 8.5\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 3,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"fermented\",\n        \"item\": \"soy sauce notes\",\n        \"qualifier\": \"fermented\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"deep earth\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"aged complexity\",\n        \"intensity\": 5\n      }\n    ],\n    \"texture\": [\n      \"paste\",\n      \"smooth\",\n      \"creamy\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 7.5\n  },\n  \"origin\": {\n    \"region\": \"Japan\",\n    \"terroir_notes\": \"Red miso is traditionally produced throughout Japan, with notable production in regions with suitable climate conditions for long fermentation periods.\"\n  },\n  \"notes\": \"Red miso is aged for 1-3 years or longer, developing its characteristic deep color and complex flavor profile. It is commonly used in soups, dressings, and marinades.\",\n  \"metadata\": {\n    \"license\": \"CC0\"\n  }\n}\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:aka-miso",
                "expected": "ogs:core:ingredient:aka-miso",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "legume",
                "expected": "other",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3065,
            "output_tokens": 578,
            "latency_ms": 4576.3,
            "usd": 0.004764,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:aka-miso\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Red Miso\",\n  \"description\": \"A fermented soybean paste with a deep reddish-brown color, produced through extended fermentation with salt and koji. Known for its rich, complex umami flavors with subtle sweetness and pronounced saltiness.\",\n  \"category\": \"legume\",\n  \"subcategory\": \"fermented soybean paste\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3.5,\n      \"sour\": 2.0,\n      \"salty\": 7.5,\n      \"bitter\": 1.5,\n      \"umami\": 8.5\n    },\n    \"structural\": {\n      \"body\": 7.0,\n      \"acidity_perception\": 2.0,\n      \"tannin\": 0,\n      \"astringency\": 1.0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"fermented\",\n        \"item\": \"miso\",\n        \"intensity\": 9.0\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"fermented\",\n        \"item\": \"earthy undertones\",\n        \"qualifier\": \"fermented\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"beans\",\n        \"item\": \"soybean\",\n        \"intensity\": 7.0\n      }\n    ],\n    \"texture\": [\n      \"paste\",\n      \"smooth\",\n      \"dense\"\n    ],\n    \"finish_length\": 7.0,\n    \"complexity\": 7.5\n  },\n  \"origin\": {\n    \"region\": \"Japan\",\n    \"terroir_notes\": \"Red miso (aka miso) is traditionally produced throughout Japan, with notable production in regions with suitable koji cultivation conditions.\"\n  },\n  \"notes\": \"A staple ingredient in Japanese cuisine used for soups, marinades, dressings, and as a condiment. The longer fermentation period compared to white miso creates deeper, more robust flavors.\",\n  \"metadata\": {\n    \"created\": \"2024-01-01T00:00:00Z\",\n    \"source\": \"OGS Benchmark\"\n  }\n}\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:aka-miso",
                "expected": "ogs:core:ingredient:aka-miso",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "legume",
                "expected": "other",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3065,
            "output_tokens": 622,
            "latency_ms": 4454.7,
            "usd": 0.00494,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5,
      "score_summary": {
        "mean": 0.5,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.5,
        "ci95_hi": 0.5,
        "n": 3
      },
      "score_if_valid": 0.5,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3065.0,
        "output_tokens_mean": 614.333,
        "latency_ms_mean": 4745.167,
        "usd_sum": 0.014728
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-008",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "hard",
      "cuisine_region": "turkish",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "791ac1beebd95233",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'Urfa chili' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"Urfa chili\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"urfa-biber\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:urfa-biber",
          "ogs_version": "0.1.0",
          "category": "spice"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:urfa-biber\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Urfa chili\",\n  \"description\": \"A dark burgundy to nearly black dried chili pepper from the Urfa region of Turkey, known for its smoky, fruity flavor and moderate heat with a distinctive chocolate undertone.\",\n  \"category\": \"spice\",\n  \"subcategory\": \"dried pepper\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 4,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 2,\n      \"umami\": 3\n    },\n    \"structural\": {\n      \"body\": 6,\n      \"acidity_perception\": 1,\n      \"tannin\": 2,\n      \"astringency\": 2,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"spice\",\n        \"group\": \"pepper\",\n        \"item\": \"dried chili\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"smoke\",\n        \"item\": \"smoke\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"dried fruit\",\n        \"item\": \"raisin\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"cocoa\",\n        \"item\": \"chocolate\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\n      \"wrinkled\",\n      \"leathery\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": \"Urfa, Turkey\",\n    \"terroir_notes\": \"The volcanic soils and hot, dry climate of southeastern Turkey contribute to the distinctive smoky and fruity character of these chilies.\"\n  },\n  \"notes\": \"Urfa biber is often preferred in Turkish cuisine for its milder heat compared to other dried chilies and its unique flavor profile combining smoke, fruit, and subtle chocolate notes.\",\n  \"metadata\": {\n    \"source\": \"OGS Core Ingredient Database\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:urfa-biber",
                "expected": "ogs:core:ingredient:urfa-biber",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3067,
            "output_tokens": 625,
            "latency_ms": 4337.9,
            "usd": 0.004954,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:urfa-biber\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Urfa chili\",\n  \"category\": \"spice\",\n  \"subcategory\": \"chili pepper\",\n  \"description\": \"A dark reddish-brown dried chili pepper from the Urfa region of Turkey, known for its smoky, fruity flavor with moderate heat and a slightly sweet undertone.\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 2,\n      \"umami\": 4\n    },\n    \"structural\": {\n      \"body\": 6,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 2,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"spice\",\n        \"group\": \"chili\",\n        \"item\": \"dried chili\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"raisin\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"smoke\",\n        \"item\": \"smoke\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 5\n      }\n    ],\n    \"texture\": [\n      \"coarse powder when ground\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 5\n  },\n  \"origin\": {\n    \"region\": \"Urfa, Turkey\"\n  },\n  \"notes\": \"Also known as Urfa Biber or Isot pepper. Typically smoked-dried with wrinkled dark burgundy to black skin. Heat level is moderate (approximately 30,000-50,000 Scoville units).\"\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:urfa-biber",
                "expected": "ogs:core:ingredient:urfa-biber",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3067,
            "output_tokens": 517,
            "latency_ms": 4253.0,
            "usd": 0.004522,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:urfa-biber\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Urfa chili\",\n  \"category\": \"spice\",\n  \"subcategory\": \"chili pepper\",\n  \"description\": \"A dark reddish-brown dried chili pepper from the Urfa region of Turkey, known for its smoky, complex flavor profile and moderate heat level.\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 2,\n      \"umami\": 4\n    },\n    \"structural\": {\n      \"body\": 6\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"spice\",\n        \"group\": \"chili\",\n        \"item\": \"smoked chili\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"dried fruit\",\n        \"item\": \"raisin\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"smoke\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"chili\",\n        \"item\": \"chili heat\",\n        \"intensity\": 6\n      }\n    ],\n    \"texture\": [\n      \"pliable\",\n      \"wrinkled\"\n    ],\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": \"Urfa, Turkey\"\n  },\n  \"primary_language\": \"en\",\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:urfa-biber",
                "expected": "ogs:core:ingredient:urfa-biber",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3067,
            "output_tokens": 468,
            "latency_ms": 3543.7,
            "usd": 0.004326,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3067.0,
        "output_tokens_mean": 536.667,
        "latency_ms_mean": 4044.867,
        "usd_sum": 0.013802
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-009",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "easy",
      "cuisine_region": "american",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "a6a51cef64e43626",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'Meyer lemon' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"Meyer lemon\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"meyer-lemon\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:meyer-lemon",
          "ogs_version": "0.1.0",
          "category": "fruit"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:meyer-lemon\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Meyer lemon\",\n  \"description\": \"A hybrid citrus fruit that is a cross between a lemon and a mandarin orange, known for its sweeter and less acidic profile compared to regular lemons.\",\n  \"category\": \"fruit\",\n  \"subcategory\": \"citrus\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 4.5,\n      \"sour\": 6.5,\n      \"salty\": 0,\n      \"bitter\": 1,\n      \"umami\": 0.5\n    },\n    \"structural\": {\n      \"body\": 3,\n      \"acidity_perception\": 6,\n      \"tannin\": 0,\n      \"astringency\": 1.5,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"mandarin\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"citrus blossom\",\n        \"item\": \"neroli\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\n      \"juicy\",\n      \"tender\",\n      \"smooth\"\n    ],\n    \"finish_length\": 4,\n    \"complexity\": 6\n  },\n  \"origin\": {\n    \"region\": \"China (ancestral origin); cultivated worldwide\",\n    \"season\": \"Winter to early spring\",\n    \"terroir_notes\": \"Hybrid variety developed in China; popularized in California\"\n  },\n  \"notes\": \"Meyer lemons are sweeter and more aromatic than Eureka or Lisbon lemons, with thinner skin and higher juice content. Excellent for both culinary and beverage applications.\",\n  \"metadata\": {\n    \"source\": \"OGS Gastronomy Standard\",\n    \"license\": \"CC0-1.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:meyer-lemon",
                "expected": "ogs:core:ingredient:meyer-lemon",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fruit",
                "expected": "fruit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3062,
            "output_tokens": 595,
            "latency_ms": 4243.8,
            "usd": 0.00483,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:meyer-lemon\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"Meyer lemon\",\n  \"description\": \"A hybrid citrus fruit that is a cross between a lemon and a mandarin orange, known for its sweet-tart flavor profile and thin, edible skin.\",\n  \"category\": \"fruit\",\n  \"subcategory\": \"citrus\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 6.5,\n      \"sour\": 7.0,\n      \"salty\": 0,\n      \"bitter\": 1.5,\n      \"umami\": 0.5\n    },\n    \"structural\": {\n      \"body\": 4.0,\n      \"acidity_perception\": 7.0,\n      \"tannin\": 0,\n      \"astringency\": 2.0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 8.0\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"mandarin\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"citrus blossom\",\n        \"item\": \"orange blossom\",\n        \"intensity\": 3.0\n      }\n    ],\n    \"texture\": [\n      \"juicy\",\n      \"tender flesh\",\n      \"thin skin\"\n    ],\n    \"finish_length\": 6.0,\n    \"complexity\": 6.5\n  },\n  \"origin\": {\n    \"region\": \"China (original); California (commercial cultivation)\"\n  },\n  \"notes\": \"Meyer lemons are sweeter and less acidic than standard Eureka or Lisbon lemons, with edible thin skin. Best used fresh, in desserts, or preserved for their unique flavor.\"\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:meyer-lemon",
                "expected": "ogs:core:ingredient:meyer-lemon",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fruit",
                "expected": "fruit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3062,
            "output_tokens": 547,
            "latency_ms": 4039.1,
            "usd": 0.004638,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:meyer-lemon\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Meyer lemon\",\n  \"category\": \"fruit\",\n  \"subcategory\": \"citrus\",\n  \"description\": \"A hybrid citrus fruit that is a cross between a lemon and a mandarin orange, characterized by a sweeter and less acidic flavor profile than regular lemons.\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 4.5,\n      \"sour\": 6.5,\n      \"salty\": 0,\n      \"bitter\": 1.5,\n      \"umami\": 0.5\n    },\n    \"structural\": {\n      \"body\": 4,\n      \"acidity_perception\": 6,\n      \"tannin\": 0,\n      \"astringency\": 2,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"peach\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"delicate\",\n        \"item\": \"honeysuckle\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\n      \"juicy\",\n      \"tender\"\n    ],\n    \"finish_length\": 5,\n    \"complexity\": 6\n  },\n  \"origin\": {\n    \"region\": \"China (native origin), cultivated in Mediterranean regions\",\n    \"season\": \"Late winter to spring\"\n  },\n  \"notes\": \"Meyer lemons are sweeter and more fragrant than Eureka or Lisbon lemons, making them popular for beverages, desserts, and culinary applications where a milder lemon flavor is desired.\"\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:meyer-lemon",
                "expected": "ogs:core:ingredient:meyer-lemon",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fruit",
                "expected": "fruit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3062,
            "output_tokens": 520,
            "latency_ms": 3966.4,
            "usd": 0.00453,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3062.0,
        "output_tokens_mean": 554.0,
        "latency_ms_mean": 4083.1,
        "usd_sum": 0.013998
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-010",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "hard",
      "cuisine_region": "mexican",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "6d4d6d3cc23f3283",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'wild huitlacoche' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"wild huitlacoche\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"huitlacoche\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:huitlacoche",
          "ogs_version": "0.1.0",
          "category": "fungus"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:huitlacoche\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"wild huitlacoche\",\n  \"primary_language\": \"en\",\n  \"category\": \"fungus\",\n  \"subcategory\": \"corn fungus\",\n  \"description\": \"A fungal infection of corn that produces dark, irregular kernels with a distinctive earthy and slightly sweet flavor. Prized in Mexican cuisine for its unique umami-rich profile.\",\n  \"origin\": {\n    \"region\": \"Mesoamerica\",\n    \"season\": \"Late summer to early fall\",\n    \"terroir_notes\": \"Grows naturally on corn plants in warm, humid conditions. Traditionally harvested from wild or cultivated maize.\"\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 1,\n      \"salty\": 0.5,\n      \"bitter\": 1,\n      \"umami\": 7\n    },\n    \"structural\": {\n      \"body\": 5,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"soil\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"mushroom\",\n        \"item\": \"mushroom\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"corn\",\n        \"item\": \"sweet corn\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\n      \"soft\",\n      \"creamy\",\n      \"slightly granular\"\n    ],\n    \"finish_length\": 4,\n    \"complexity\": 6\n  },\n  \"notes\": \"Wild huitlacoche offers a more intense earthy and umami profile compared to cultivated varieties. It is traditionally used in Mexican cuisine in quesadillas, tamales, and soups.\",\n  \"metadata\": {\n    \"source\": \"OGS Core\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:huitlacoche",
                "expected": "ogs:core:ingredient:huitlacoche",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fungus",
                "expected": "fungus",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3069,
            "output_tokens": 586,
            "latency_ms": 4559.9,
            "usd": 0.004799,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:huitlacoche\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"wild huitlacoche\",\n  \"description\": \"A delicacy from Mexico, huitlacoche is a fungal infection of corn kernels that develops a dark, truffle-like appearance and earthy, umami-rich flavor. Wild varieties are foraged and prized for their complex taste profile.\",\n  \"category\": \"fungus\",\n  \"subcategory\": \"corn fungus\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 1,\n      \"salty\": 1,\n      \"bitter\": 1,\n      \"umami\": 8\n    },\n    \"structural\": {\n      \"body\": 6,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"mushroom\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"corn\",\n        \"item\": \"corn silk\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"greens\",\n        \"item\": \"fresh corn\",\n        \"intensity\": 5\n      }\n    ],\n    \"texture\": [\n      \"soft\",\n      \"creamy\",\n      \"granular\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": \"Mexico\",\n    \"season\": \"summer to early fall\",\n    \"terroir_notes\": \"Thrives in humid growing conditions; wild specimens found in traditional corn-growing regions\"\n  },\n  \"notes\": \"Traditional ingredient in Mexican cuisine, particularly valued in corn-based dishes. Wild huitlacoche has a more pronounced earthy character than cultivated varieties.\",\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:huitlacoche",
                "expected": "ogs:core:ingredient:huitlacoche",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fungus",
                "expected": "fungus",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3069,
            "output_tokens": 576,
            "latency_ms": 5003.8,
            "usd": 0.004759,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:ingredient:huitlacoche\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"wild huitlacoche\",\n  \"category\": \"fungus\",\n  \"description\": \"A fungal growth on corn caused by Ustilago maydis, prized in Mexican cuisine for its earthy, umami-rich flavor and dark coloring\",\n  \"subcategory\": \"corn smut\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 1,\n      \"umami\": 8\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"soil\",\n        \"item\": \"petrichor\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"mushroom\",\n        \"item\": \"umami\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"corn\",\n        \"item\": \"corn silk\",\n        \"intensity\": 5\n      }\n    ],\n    \"texture\": [\n      \"creamy\",\n      \"tender\",\n      \"slightly gelatinous\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": \"Mexico\",\n    \"terroir_notes\": \"Found on corn plants in warm, humid climates\"\n  },\n  \"notes\": \"Also known as corn smut or Mexican truffle. The fungus typically appears on maize ears and is harvested while still fresh for optimal flavor and texture. Traditional ingredient in Mexican cuisine, particularly in quesadillas and tamales.\",\n  \"primary_language\": \"en\",\n  \"metadata\": {\n    \"source\": \"OGS gastronomy expert evaluation\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:huitlacoche",
                "expected": "ogs:core:ingredient:huitlacoche",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fungus",
                "expected": "fungus",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3069,
            "output_tokens": 554,
            "latency_ms": 4141.7,
            "usd": 0.004671,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3069.0,
        "output_tokens_mean": 572.0,
        "latency_ms_mean": 4568.467,
        "usd_sum": 0.014229
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-011",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "6615664480900680",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'Barolo' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"Barolo\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"barolo-docg\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:barolo-docg",
          "ogs_version": "0.1.0",
          "type": "wine"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:barolo-docg\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Barolo\",\n  \"type\": \"wine\",\n  \"subtype\": \"Nebbiolo\",\n  \"region\": {\n    \"en\": \"Piedmont\",\n    \"it\": \"Piemonte\"\n  },\n  \"appellation\": \"Barolo DOCG\",\n  \"producer\": \"Various\",\n  \"grape_varieties\": [\n    \"Nebbiolo\"\n  ],\n  \"abv\": 14.5,\n  \"residual_sugar\": 1.5,\n  \"serve_temperature_c\": {\n    \"min\": 16,\n    \"max\": 18\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 6.0,\n      \"salty\": 0.5,\n      \"bitter\": 4.0,\n      \"umami\": 2.0\n    },\n    \"structural\": {\n      \"body\": 8.0,\n      \"acidity_perception\": 7.0,\n      \"tannin\": 8.5,\n      \"astringency\": 7.5,\n      \"alcohol_heat\": 5.0,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"dark berries\",\n        \"item\": \"cherry\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"dark berries\",\n        \"item\": \"plum\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"rose\",\n        \"item\": \"rose petals\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"licorice\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"tar\",\n        \"intensity\": 4.0\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"graphite\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"oak\",\n        \"item\": \"oak wood\",\n        \"intensity\": 3.0\n      }\n    ],\n    \"texture\": [\n      \"velvety\",\n      \"structured\"\n    ],\n    \"finish_length\": 8.0,\n    \"complexity\": 8.5\n  },\n  \"primary_language\": \"en\",\n  \"description\": \"Barolo is a prestigious Italian red wine from the Piedmont region, made exclusively from Nebbiolo grapes. Known for its powerful structure, complex aromatics, and exceptional aging potential, Barolo is considered one of Italy's greatest wines.\",\n  \"notes\": \"Barolo DOCG requires a minimum of 38 months of aging, with at least 18 months in oak. The wine develops complexity and elegance over decades.\",\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:barolo-docg",
                "expected": "ogs:core:beverage:barolo-docg",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3082,
            "output_tokens": 853,
            "latency_ms": 5780.3,
            "usd": 0.005878,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:barolo-docg\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Barolo\",\n  \"type\": \"wine\",\n  \"subtype\": \"Nebbiolo\",\n  \"region\": \"Piedmont\",\n  \"appellation\": \"Barolo DOCG\",\n  \"producer\": \"Various\",\n  \"grape_varieties\": [\"Nebbiolo\"],\n  \"abv\": 14.0,\n  \"residual_sugar\": 0.5,\n  \"serve_temperature_c\": {\n    \"min\": 16,\n    \"max\": 18\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.0,\n      \"sour\": 5.5,\n      \"salty\": 0.5,\n      \"bitter\": 4.0,\n      \"umami\": 2.0\n    },\n    \"structural\": {\n      \"body\": 8.5,\n      \"acidity_perception\": 5.5,\n      \"tannin\": 8.0,\n      \"astringency\": 7.5,\n      \"alcohol_heat\": 6.0,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"red berries\",\n        \"item\": \"cherry\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7.0\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruits\",\n        \"item\": \"plum\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"roses\",\n        \"item\": \"rose petal\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"slate\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"licorice\",\n        \"intensity\": 4.0\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"oak\",\n        \"item\": \"oak\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 3.5\n      }\n    ],\n    \"finish_length\": 8.0,\n    \"complexity\": 8.5\n  },\n  \"description\": \"Barolo is a prestigious red wine from the Piedmont region of Italy, produced from the Nebbiolo grape. Known as the 'King of Wines', it is characterized by its full body, high tannins, and complex aromatic profile with notes of dried cherry, plum, roses, and earth.\",\n  \"primary_language\": \"en\"\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:barolo-docg",
                "expected": "ogs:core:beverage:barolo-docg",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3082,
            "output_tokens": 715,
            "latency_ms": 4536.2,
            "usd": 0.005326,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:barolo-docg\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Barolo\",\n  \"type\": \"wine\",\n  \"subtype\": \"dry red wine\",\n  \"grape_varieties\": [\"Nebbiolo\"],\n  \"region\": {\n    \"en\": \"Piedmont\",\n    \"it\": \"Piemonte\"\n  },\n  \"appellation\": \"Barolo DOCG\",\n  \"abv\": 14.0,\n  \"residual_sugar\": 1.5,\n  \"serve_temperature_c\": {\n    \"min\": 16,\n    \"max\": 18\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.0,\n      \"sour\": 4.5,\n      \"salty\": 0.5,\n      \"bitter\": 3.0,\n      \"umami\": 2.0\n    },\n    \"structural\": {\n      \"body\": 8.5,\n      \"acidity_perception\": 7.0,\n      \"tannin\": 8.0,\n      \"astringency\": 7.5,\n      \"alcohol_heat\": 6.5,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"red berries\",\n        \"item\": \"cherry\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7.0\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"plum\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"tar\",\n        \"intensity\": 6.0\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"soil\",\n        \"item\": \"earth\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"rose\",\n        \"item\": \"rose petal\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spice\",\n        \"item\": \"licorice\",\n        \"intensity\": 4.5\n      }\n    ],\n    \"texture\": [\n      \"smooth\",\n      \"silky\",\n      \"velvety\"\n    ],\n    \"finish_length\": 8.5,\n    \"complexity\": 8.0\n  },\n  \"description\": \"A full-bodied dry red wine from Piedmont, Italy, made exclusively from Nebbiolo grapes. Barolo is known for its powerful tannins, complex aromatics, and impressive aging potential.\",\n  \"notes\": \"Traditionally aged for at least 2 years in wood. Often requires decanting. Improves with bottle age for 10-20+ years.\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"it\"],\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:barolo-docg",
                "expected": "ogs:core:beverage:barolo-docg",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3082,
            "output_tokens": 802,
            "latency_ms": 5096.5,
            "usd": 0.005674,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3082.0,
        "output_tokens_mean": 790.0,
        "latency_ms_mean": 5137.667,
        "usd_sum": 0.016878
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-012",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "hard",
      "cuisine_region": "scottish",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "bd330891a3e68b59",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'Islay single malt' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"Islay single malt\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"ardbeg-10-islay\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:ardbeg-10-islay",
          "ogs_version": "0.1.0",
          "type": "spirit"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:ardbeg-10-islay\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Islay single malt\",\n  \"type\": \"spirit\",\n  \"subtype\": \"single malt scotch whisky\",\n  \"region\": \"Islay\",\n  \"producer\": \"Ardbeg\",\n  \"abv\": 46,\n  \"serve_temperature_c\": {\n    \"min\": 15,\n    \"max\": 20\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 1,\n      \"salty\": 3,\n      \"bitter\": 4,\n      \"umami\": 2\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 2,\n      \"tannin\": 3,\n      \"astringency\": 4,\n      \"alcohol_heat\": 6,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"smoke\",\n        \"item\": \"peat\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"zest\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"pepper\",\n        \"item\": \"black pepper\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"oak\",\n        \"item\": \"oak\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 5\n      }\n    ],\n    \"texture\": [\n      \"oily\",\n      \"warming\",\n      \"smoky\"\n    ],\n    \"finish_length\": 7,\n    \"complexity\": 7\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:ardbeg-10-islay",
                "expected": "ogs:core:beverage:ardbeg-10-islay",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "spirit",
                "expected": "spirit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3091,
            "output_tokens": 500,
            "latency_ms": 3234.4,
            "usd": 0.004473,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:ardbeg-10-islay\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Islay single malt\",\n  \"type\": \"spirit\",\n  \"subtype\": \"single malt scotch whisky\",\n  \"region\": \"Islay\",\n  \"abv\": 46,\n  \"serve_temperature_c\": {\n    \"min\": 15,\n    \"max\": 20\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 2,\n      \"salty\": 2,\n      \"bitter\": 6,\n      \"umami\": 1\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 3,\n      \"tannin\": 5,\n      \"astringency\": 4,\n      \"alcohol_heat\": 6,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"smoke\",\n        \"item\": \"peat smoke\",\n        \"intensity\": 9\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"iodine\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"plum\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spices\",\n        \"item\": \"licorice\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\n      \"oily\",\n      \"smoky\"\n    ],\n    \"finish_length\": 8,\n    \"complexity\": 8\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:ardbeg-10-islay",
                "expected": "ogs:core:beverage:ardbeg-10-islay",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "spirit",
                "expected": "spirit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3091,
            "output_tokens": 480,
            "latency_ms": 3119.2,
            "usd": 0.004393,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:ardbeg-10-islay\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Islay single malt\",\n  \"type\": \"spirit\",\n  \"subtype\": \"single malt scotch whisky\",\n  \"region\": \"Islay\",\n  \"abv\": 46,\n  \"serve_temperature_c\": {\n    \"min\": 15,\n    \"max\": 20\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 2,\n      \"salty\": 2,\n      \"bitter\": 4,\n      \"umami\": 1\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 3,\n      \"tannin\": 0,\n      \"astringency\": 2,\n      \"alcohol_heat\": 6,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"peat\",\n        \"item\": \"smoke\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"zest\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"oak\",\n        \"item\": \"oak\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"medicinal\",\n        \"item\": \"iodine\",\n        \"intensity\": 6\n      }\n    ],\n    \"texture\": [\n      \"oily\",\n      \"warming\"\n    ],\n    \"finish_length\": 8,\n    \"complexity\": 7\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:ardbeg-10-islay",
                "expected": "ogs:core:beverage:ardbeg-10-islay",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "spirit",
                "expected": "spirit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3091,
            "output_tokens": 476,
            "latency_ms": 3090.2,
            "usd": 0.004377,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3091.0,
        "output_tokens_mean": 485.333,
        "latency_ms_mean": 3147.933,
        "usd_sum": 0.013243
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-013",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "872d23f39307a374",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'junmai ginjo sake' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"junmai ginjo sake\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"junmai-ginjo-generic\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:junmai-ginjo-generic",
          "ogs_version": "0.1.0",
          "type": "sake"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:junmai-ginjo-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Junmai Ginjo Sake\",\n  \"type\": \"sake\",\n  \"subtype\": \"junmai ginjo\",\n  \"abv\": 15.5,\n  \"serve_temperature_c\": {\n    \"min\": 5,\n    \"max\": 15\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 3,\n      \"salty\": 1,\n      \"bitter\": 1,\n      \"umami\": 2\n    },\n    \"structural\": {\n      \"body\": 3,\n      \"acidity_perception\": 3,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 4,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"peach\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"flower\",\n        \"item\": \"orchid\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"grapefruit\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"herb\",\n        \"item\": \"grass\",\n        \"intensity\": 3\n      }\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 7\n  },\n  \"description\": \"Junmai ginjo sake is a premium sake made from 100% rice with a polishing ratio of 50% or less. Known for its delicate, fruity aromatics and smooth, refined character.\"\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:junmai-ginjo-generic",
                "expected": "ogs:core:beverage:junmai-ginjo-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "sake",
                "expected": "sake",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3088,
            "output_tokens": 489,
            "latency_ms": 3173.7,
            "usd": 0.004426,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:junmai-ginjo-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Junmai Ginjo Sake\",\n  \"type\": \"sake\",\n  \"subtype\": \"junmai ginjo\",\n  \"primary_language\": \"en\",\n  \"description\": \"A premium sake made from rice polished to at least 60% of its original size, using pure koji and water without added alcohol or sugars.\",\n  \"region\": {\n    \"en\": \"Japan\"\n  },\n  \"abv\": 15.6,\n  \"residual_sugar\": 2.0,\n  \"serve_temperature_c\": {\n    \"min\": 5,\n    \"max\": 15\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3.0,\n      \"sour\": 2.5,\n      \"salty\": 1.0,\n      \"bitter\": 1.0,\n      \"umami\": 3.5\n    },\n    \"structural\": {\n      \"body\": 4.0,\n      \"acidity_perception\": 3.0,\n      \"tannin\": 0.0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 4.5,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"pear\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"blossom\",\n        \"item\": \"white flower\",\n        \"intensity\": 4.0\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"grass\",\n        \"item\": \"green grass\",\n        \"intensity\": 2.5\n      }\n    ],\n    \"texture\": [\n      \"smooth\",\n      \"clean\",\n      \"delicate\"\n    ],\n    \"finish_length\": 5.0,\n    \"complexity\": 6.0\n  },\n  \"notes\": \"Junmai ginjo sake represents the pinnacle of rice polishing and traditional brewing, offering refined aromatics and a clean, elegant profile.\"\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:junmai-ginjo-generic",
                "expected": "ogs:core:beverage:junmai-ginjo-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "sake",
                "expected": "sake",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3088,
            "output_tokens": 570,
            "latency_ms": 4099.8,
            "usd": 0.00475,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:junmai-ginjo-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Junmai Ginjo Sake\",\n  \"type\": \"sake\",\n  \"subtype\": \"junmai ginjo\",\n  \"primary_language\": \"en\",\n  \"description\": \"A premium sake made from sake rice polished to at least 50% of its original weight, with no added alcohol. Known for delicate, fruity aromatics and refined flavor profile.\",\n  \"abv\": 15.5,\n  \"residual_sugar\": 0.5,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 15\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3.0,\n      \"sour\": 4.0,\n      \"salty\": 1.0,\n      \"bitter\": 1.5,\n      \"umami\": 2.5\n    },\n    \"structural\": {\n      \"body\": 3.5,\n      \"acidity_perception\": 4.0,\n      \"tannin\": 0.0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 4.5,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"pear\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"white flowers\",\n        \"item\": \"osmanthus\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"grass\",\n        \"item\": \"rice flower\",\n        \"intensity\": 3.5\n      }\n    ],\n    \"texture\": [\n      \"silky\",\n      \"delicate\",\n      \"crisp\"\n    ],\n    \"finish_length\": 6.0,\n    \"complexity\": 7.0\n  },\n  \"notes\": \"Junmai ginjo represents the pinnacle of sake craftsmanship, characterized by extensive rice polishing and pure rice fermentation without added alcohol. The result is an elegant, aromatic beverage with remarkable clarity and finesse.\",\n  \"metadata\": {\n    \"source\": \"OGS Beverage Benchmark\",\n    \"license\": \"CC0-1.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:junmai-ginjo-generic",
                "expected": "ogs:core:beverage:junmai-ginjo-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "sake",
                "expected": "sake",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3088,
            "output_tokens": 660,
            "latency_ms": 4418.6,
            "usd": 0.00511,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3088.0,
        "output_tokens_mean": 573.0,
        "latency_ms_mean": 3897.367,
        "usd_sum": 0.014286
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-014",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "easy",
      "cuisine_region": "german",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "3fe3fe74f8d80260",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'Oktoberfest Märzen lager' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"Oktoberfest Märzen lager\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"oktoberfest-marzen\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:oktoberfest-marzen",
          "ogs_version": "0.1.0",
          "type": "beer"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:oktoberfest-marzen\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Oktoberfest Märzen lager\",\n  \"type\": \"beer\",\n  \"subtype\": \"Märzen\",\n  \"region\": \"Bavaria\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 2,\n      \"salty\": 0,\n      \"bitter\": 4,\n      \"umami\": 1\n    },\n    \"structural\": {\n      \"body\": 6,\n      \"acidity_perception\": 2,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 5,\n      \"effervescence\": 4\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"grain\",\n        \"group\": \"cereal\",\n        \"item\": \"malt\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"traditional\",\n        \"item\": \"clove\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone\",\n        \"item\": \"apricot\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\n      \"smooth\",\n      \"creamy\"\n    ],\n    \"finish_length\": 5,\n    \"complexity\": 6\n  },\n  \"abv\": 5.8,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 12\n  }\n}\n```",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "'grain' is not one of ['fruit', 'floral', 'herbal', 'spice', 'earth', 'wood', 'dairy', 'savory', 'confection', 'vegetal', 'marine']"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:oktoberfest-marzen",
                "expected": "ogs:core:beverage:oktoberfest-marzen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "beer",
                "expected": "beer",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3098,
            "output_tokens": 436,
            "latency_ms": 3360.9,
            "usd": 0.004222,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:oktoberfest-marzen\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Oktoberfest Märzen lager\",\n  \"type\": \"beer\",\n  \"subtype\": \"Märzen\",\n  \"region\": {\n    \"de\": \"Bavaria\",\n    \"en\": \"Bavaria\"\n  },\n  \"abv\": 5.8,\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 4.0,\n      \"sour\": 2.0,\n      \"salty\": 0.0,\n      \"bitter\": 3.5,\n      \"umami\": 1.0\n    },\n    \"structural\": {\n      \"body\": 5.5,\n      \"acidity_perception\": 2.0,\n      \"tannin\": 0.0,\n      \"astringency\": 1.0,\n      \"alcohol_heat\": 4.0,\n      \"effervescence\": 4.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"apricot\",\n        \"intensity\": 4.0\n      },\n      {\n        \"family\": \"grain\",\n        \"group\": \"cereal\",\n        \"item\": \"bread\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spice\",\n        \"item\": \"clove\",\n        \"intensity\": 2.5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"floral herb\",\n        \"item\": \"hay\",\n        \"intensity\": 2.0\n      }\n    ],\n    \"texture\": [\n      \"smooth\",\n      \"creamy\",\n      \"malty\"\n    ],\n    \"finish_length\": 5.0,\n    \"complexity\": 6.0\n  },\n  \"serve_temperature_c\": {\n    \"min\": 7,\n    \"max\": 9\n  },\n  \"description\": \"A traditional German lager brewed for Oktoberfest, characterized by its rich malt complexity, balanced bitterness, and smooth drinkability.\"\n}\n```",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "'grain' is not one of ['fruit', 'floral', 'herbal', 'spice', 'earth', 'wood', 'dairy', 'savory', 'confection', 'vegetal', 'marine']"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:oktoberfest-marzen",
                "expected": "ogs:core:beverage:oktoberfest-marzen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "beer",
                "expected": "beer",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3098,
            "output_tokens": 566,
            "latency_ms": 3727.8,
            "usd": 0.004742,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:beverage:oktoberfest-marzen\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Oktoberfest Märzen lager\",\n  \"type\": \"beer\",\n  \"subtype\": \"Märzen\",\n  \"region\": \"Bavaria\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3.5,\n      \"sour\": 2,\n      \"salty\": 0,\n      \"bitter\": 4,\n      \"umami\": 1\n    },\n    \"structural\": {\n      \"body\": 6,\n      \"acidity_perception\": 2,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 4.5,\n      \"effervescence\": 3\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"stone fruit\",\n        \"item\": \"apricot\",\n        \"intensity\": 3,\n        \"qualifier\": \"dried\"\n      },\n      {\n        \"family\": \"bread\",\n        \"group\": \"grain\",\n        \"item\": \"toast\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warming spice\",\n        \"item\": \"clove\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\n      \"smooth\",\n      \"creamy\"\n    ],\n    \"finish_length\": 5,\n    \"complexity\": 5.5\n  },\n  \"abv\": 6.0,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 10\n  },\n  \"description\": \"A traditional Bavarian lager brewed in the Märzen style, featuring a rich malty character with subtle fruit notes and a clean, crisp finish. Historically brewed in March for the Oktoberfest celebration.\"\n}\n```",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "'bread' is not one of ['fruit', 'floral', 'herbal', 'spice', 'earth', 'wood', 'dairy', 'savory', 'confection', 'vegetal', 'marine']"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:oktoberfest-marzen",
                "expected": "ogs:core:beverage:oktoberfest-marzen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "beer",
                "expected": "beer",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3098,
            "output_tokens": 499,
            "latency_ms": 3366.1,
            "usd": 0.004474,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3098.0,
        "output_tokens_mean": 500.333,
        "latency_ms_mean": 3484.933,
        "usd_sum": 0.013438
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-015",
      "category": "structured_output",
      "capability": "schema_valid_pairing",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.pairing",
        "vocab.match_types",
        "vocab.explanation_codes",
        "vocab.grade_scale",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "f0c01fa20f35eb51",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS pairing schema (`schema/pairing.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/pairing.schema.json\",\n  \"title\": \"OGS Pairing\",\n  \"description\": \"A scored, explained pairing between a dish and a beverage.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"dish_ref\",\n    \"beverage_ref\",\n    \"match_types\",\n    \"overall_score\",\n    \"grade\",\n    \"explanation_codes\",\n    \"confidence\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:pairing:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"dish_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"beverage_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"match_types\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/match_type_entry\"\n      }\n    },\n    \"overall_score\": {\n      \"type\": \"integer\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"grade\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"A+\",\n        \"A\",\n        \"B+\",\n        \"B\",\n        \"C+\",\n        \"C\",\n        \"D\",\n        \"F\"\n      ]\n    },\n    \"explanation_codes\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/explanation_code_entry\"\n      }\n    },\n    \"risks\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/risk_entry\"\n      }\n    },\n    \"confidence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 1\n    },\n    \"evidence\": {\n      \"$ref\": \"#/$defs/evidence\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"match_type_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"type\",\n        \"score\"\n      ],\n      \"properties\": {\n        \"type\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"harmony\",\n            \"contrast\",\n            \"bridge\",\n            \"intensity_alignment\",\n            \"structural_balance\",\n            \"refresh\"\n          ]\n        },\n        \"score\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 100\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"primary\": {\n          \"type\": \"boolean\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"explanation_code_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"weight\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 1\n        },\n        \"detail\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"risk_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\",\n        \"severity\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"severity\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"low\",\n            \"medium\",\n            \"high\"\n          ]\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"mitigation\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"evidence\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"expert_panel\",\n            \"instrumental\",\n            \"computed\",\n            \"estimated\"\n          ]\n        },\n        \"sources\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Explanation Codes\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS pairing JSON document for the described match. It MUST validate against the OGS pairing schema. Respond with ONLY the JSON.\n\nINPUT:\n{\n  \"pairing_slug\": \"crottin-sancerre\",\n  \"dish_ref\": \"ogs:core:dish:crottin-de-chavignol\",\n  \"beverage_ref\": \"ogs:core:beverage:sancerre-blanc\",\n  \"intent\": \"Classic Loire regional pairing: Sancerre's acidity and citrus meet fresh goat cheese's tang.\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/pairing.schema.json",
        "required_fields": {
          "id": "ogs:core:pairing:crottin-sancerre",
          "ogs_version": "0.1.0",
          "dish_ref": "ogs:core:dish:crottin-de-chavignol",
          "beverage_ref": "ogs:core:beverage:sancerre-blanc"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:crottin-sancerre\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Crottin de Chavignol & Sancerre Blanc\",\n  \"dish_ref\": \"ogs:core:dish:crottin-de-chavignol\",\n  \"beverage_ref\": \"ogs:core:beverage:sancerre-blanc\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 95,\n      \"description\": \"Classic Loire Valley regional pairing with shared terroir and complementary flavor profiles\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"refresh\",\n      \"score\": 90,\n      \"description\": \"Sancerre's crisp acidity refreshes the palate between bites of tangy goat cheese\"\n    },\n    {\n      \"type\": \"bridge\",\n      \"score\": 88,\n      \"description\": \"Shared citrus and herbaceous aromatics create an aromatic bridge between cheese and wine\"\n    }\n  ],\n  \"overall_score\": 93,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"HARMONY_REGIONAL\",\n      \"weight\": 0.35,\n      \"detail\": \"Both Crottin de Chavignol and Sancerre Blanc originate from the Loire Valley, sharing terroir and regional compatibility\"\n    },\n    {\n      \"code\": \"HARMONY_FLAVOR_CONVERGE\",\n      \"weight\": 0.30,\n      \"detail\": \"Fresh goat cheese tang converges with Sancerre's citrus and mineral notes for unified taste experience\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.20,\n      \"detail\": \"High acidity in Sancerre cuts through the cheese's richness and refreshes the palate\"\n    },\n    {\n      \"code\": \"AROMA_SHARED_HERBAL\",\n      \"weight\": 0.15,\n      \"detail\": \"Herbaceous aromatics present in both the goat cheese and the Sauvignon Blanc create aromatic continuity\"\n    }\n  ],\n  \"confidence\": 0.95,\n  \"evidence\": {\n    \"method\": \"expert_panel\",\n    \"sources\": [\n      \"Loire Valley gastronomy tradition\",\n      \"Regional pairing documentation\"\n    ],\n    \"notes\": \"This is a time-honored pairing reflecting centuries of Loire Valley culinary tradition\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:crottin-sancerre",
                "expected": "ogs:core:pairing:crottin-sancerre",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:crottin-de-chavignol",
                "expected": "ogs:core:dish:crottin-de-chavignol",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:sancerre-blanc",
                "expected": "ogs:core:beverage:sancerre-blanc",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4203,
            "output_tokens": 669,
            "latency_ms": 5748.7,
            "usd": 0.006038,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:crottin-sancerre\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Crottin de Chavignol & Sancerre Blanc\",\n  \"dish_ref\": \"ogs:core:dish:crottin-de-chavignol\",\n  \"beverage_ref\": \"ogs:core:beverage:sancerre-blanc\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 95,\n      \"primary\": true,\n      \"description\": \"Classic Loire regional pairing with shared terroir and flavor profiles\"\n    },\n    {\n      \"type\": \"contrast\",\n      \"score\": 85,\n      \"description\": \"Wine's crisp acidity contrasts with cheese's creamy richness\"\n    },\n    {\n      \"type\": \"refresh\",\n      \"score\": 88,\n      \"description\": \"High acidity refreshes palate between bites of rich goat cheese\"\n    }\n  ],\n  \"overall_score\": 92,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"HARMONY_REGIONAL\",\n      \"weight\": 0.4,\n      \"detail\": \"Both Crottin de Chavignol and Sancerre originate from the Loire Valley, sharing regional terroir and winemaking/cheesemaking traditions\"\n    },\n    {\n      \"code\": \"HARMONY_FLAVOR_CONVERGE\",\n      \"weight\": 0.3,\n      \"detail\": \"Goat cheese's tangy, fresh character aligns beautifully with Sancerre's citrus and herbaceous notes\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.2,\n      \"detail\": \"Wine's natural acidity cuts through the cheese's fat and richness, creating perfect structural balance\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.1,\n      \"detail\": \"Acidity in Sancerre cleanses the palate between bites\"\n    }\n  ],\n  \"confidence\": 0.95,\n  \"evidence\": {\n    \"method\": \"expert_panel\",\n    \"sources\": [\n      \"Loire Valley culinary tradition\",\n      \"Classic French regional pairings\"\n    ],\n    \"notes\": \"This is an established, time-honored pairing with centuries of evidence supporting its excellence\"\n  },\n  \"metadata\": {\n    \"created\": \"2024-01-01T00:00:00Z\",\n    \"source\": \"OGS Benchmark\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:crottin-sancerre",
                "expected": "ogs:core:pairing:crottin-sancerre",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:crottin-de-chavignol",
                "expected": "ogs:core:dish:crottin-de-chavignol",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:sancerre-blanc",
                "expected": "ogs:core:beverage:sancerre-blanc",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4203,
            "output_tokens": 663,
            "latency_ms": 4930.4,
            "usd": 0.006014,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:crottin-sancerre\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Crottin de Chavignol & Sancerre Blanc\",\n  \"dish_ref\": \"ogs:core:dish:crottin-de-chavignol\",\n  \"beverage_ref\": \"ogs:core:beverage:sancerre-blanc\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 95,\n      \"description\": \"Classic Loire regional pairing with shared terroir and complementary flavor profiles\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 90,\n      \"description\": \"Acidity in Sancerre balances the rich, creamy texture of the cheese\"\n    },\n    {\n      \"type\": \"bridge\",\n      \"score\": 88,\n      \"description\": \"Citrus and herbal notes create aromatic bridges between wine and cheese\"\n    }\n  ],\n  \"overall_score\": 91,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"HARMONY_REGIONAL\",\n      \"weight\": 0.4,\n      \"detail\": \"Both Crottin de Chavignol and Sancerre Blanc originate from the Loire Valley, sharing terroir and cultural heritage\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.35,\n      \"detail\": \"Sancerre's crisp acidity cuts through the creamy richness of the goat cheese, refreshing the palate\"\n    },\n    {\n      \"code\": \"AROMA_SHARED_CITRUS\",\n      \"weight\": 0.25,\n      \"detail\": \"Citrus aromatics present in both the wine's character and the cheese's subtle brightness create continuity\"\n    }\n  ],\n  \"confidence\": 0.92,\n  \"evidence\": {\n    \"method\": \"expert_panel\",\n    \"sources\": [\n      \"Loire Valley sommelier consensus\",\n      \"Traditional French pairing practices\"\n    ],\n    \"notes\": \"This pairing represents a time-honored combination rooted in regional gastronomy\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:crottin-sancerre",
                "expected": "ogs:core:pairing:crottin-sancerre",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:crottin-de-chavignol",
                "expected": "ogs:core:dish:crottin-de-chavignol",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:sancerre-blanc",
                "expected": "ogs:core:beverage:sancerre-blanc",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4203,
            "output_tokens": 573,
            "latency_ms": 4184.1,
            "usd": 0.005654,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 4203.0,
        "output_tokens_mean": 635.0,
        "latency_ms_mean": 4954.4,
        "usd_sum": 0.017706
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-016",
      "category": "structured_output",
      "capability": "schema_valid_dish",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.dish",
        "vocab.component_roles",
        "vocab.cooking_methods",
        "vocab.courses",
        "vocab.service_temperatures",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "284b25ae8f0dc3ef",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS dish schema (`schema/dish.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/dish.schema.json\",\n  \"title\": \"OGS Dish\",\n  \"description\": \"A composed dish with components, cooking methods, and overall sensory profile.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"components\",\n    \"overall_sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"cuisine\": {\n      \"type\": \"string\",\n      \"description\": \"Free-text cuisine name (v0.1) OR a cuisine entity reference matching ogs:<ns>:cuisine:<id> (v0.2+).\"\n    },\n    \"course\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"amuse\",\n        \"appetizer\",\n        \"soup\",\n        \"salad\",\n        \"fish\",\n        \"main\",\n        \"cheese\",\n        \"dessert\",\n        \"petit_four\"\n      ]\n    },\n    \"service_temperature\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"frozen\",\n        \"cold\",\n        \"cool\",\n        \"room\",\n        \"warm\",\n        \"hot\",\n        \"very_hot\"\n      ]\n    },\n    \"components\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/component\"\n      }\n    },\n    \"overall_sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"component\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"role\",\n        \"ingredient_ref\"\n      ],\n      \"properties\": {\n        \"role\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"main_protein\",\n            \"main_vegetable\",\n            \"starch\",\n            \"sauce\",\n            \"fat_component\",\n            \"acid_component\",\n            \"aromatic_garnish\",\n            \"seasoning\",\n            \"textural_element\"\n          ]\n        },\n        \"ingredient_ref\": {\n          \"type\": \"string\",\n          \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\"\n        },\n        \"cooking_method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"raw\",\n            \"seared\",\n            \"grilled\",\n            \"roasted\",\n            \"braised\",\n            \"poached\",\n            \"steamed\",\n            \"fried\",\n            \"confit\",\n            \"smoked\",\n            \"fermented\",\n            \"cured\",\n            \"pickled\",\n            \"dehydrated\",\n            \"sous_vide\",\n            \"blanched\",\n            \"sauteed\",\n            \"baked\",\n            \"charred\",\n            \"emulsified\"\n          ]\n        },\n        \"maillard_index\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"reduction_level\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"proportion\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"dominant\",\n            \"significant\",\n            \"accent\",\n            \"trace\"\n          ]\n        },\n        \"sensory_contribution\": {\n          \"type\": \"object\",\n          \"properties\": {\n            \"basic_tastes\": {\n              \"$ref\": \"#/$defs/basic_tastes\"\n            },\n            \"aromatics\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"$ref\": \"#/$defs/aromatic_entry\"\n              }\n            },\n            \"texture\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"type\": \"string\"\n              }\n            }\n          },\n          \"additionalProperties\": true\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS dish JSON for tonkotsu ramen (pork-bone broth, wheat noodles, chashu, soft egg, nori, scallion). Respond with ONLY the JSON.\n\nINPUT:\n{\n  \"dish_name\": \"tonkotsu ramen\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"tonkotsu-ramen\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/dish.schema.json",
        "required_fields": {
          "id": "ogs:core:dish:tonkotsu-ramen",
          "ogs_version": "0.1.0",
          "course": "main"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:tonkotsu-ramen\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"tonkotsu ramen\",\n  \"cuisine\": \"Japanese\",\n  \"course\": \"main\",\n  \"service_temperature\": \"very_hot\",\n  \"components\": [\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:pork-bone-broth\",\n      \"cooking_method\": \"braised\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 8,\n          \"salty\": 6\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"broth\",\n            \"item\": \"pork bone\",\n            \"intensity\": 9\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"starch\",\n      \"ingredient_ref\": \"ogs:core:ingredient:wheat-noodles\",\n      \"cooking_method\": \"poached\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"texture\": [\n          \"chewy\",\n          \"springy\"\n        ]\n      }\n    },\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:chashu-pork\",\n      \"cooking_method\": \"braised\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 7,\n          \"salty\": 5\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"meat\",\n            \"item\": \"pork\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 8\n          }\n        ],\n        \"texture\": [\n          \"tender\",\n          \"succulent\"\n        ]\n      }\n    },\n    {\n      \"role\": \"textural_element\",\n      \"ingredient_ref\": \"ogs:core:ingredient:soft-boiled-egg\",\n      \"cooking_method\": \"poached\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 5,\n          \"salty\": 3\n        },\n        \"texture\": [\n          \"creamy\",\n          \"silky\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:nori\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 8,\n          \"salty\": 4\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"seaweed\",\n            \"item\": \"nori\",\n            \"intensity\": 7\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:scallion\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"scallion\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"umami\": 8,\n      \"salty\": 6,\n      \"sweet\": 1,\n      \"sour\": 1,\n      \"bitter\": 0\n    },\n    \"structural\": {\n      \"body\": 8,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'sauce', 'ingredient_ref': 'ogs:core:ingredient:pork-bone-broth', 'cooking_method': 'braised', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'umami': 8, 'salty': 6}, 'aromatics': [{'family': 'savory', 'group': 'broth', 'item': 'pork bone', 'intensity': 9}]}}, {'role': 'starch', 'ingredient_ref': 'ogs:core:ingredient:wheat-noodles', 'cooking_method': 'poached', 'proportion': 'dominant', 'sensory_contribution': {'texture': ['chewy', 'springy']}}, {'role': 'main_protein', 'ingredient_ref': 'ogs:core:ingredient:chashu-pork', 'cooking_method': 'braised', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'umami': 7, 'salty': 5}, 'aromatics': [{'family': 'savory', 'group': 'meat', 'item': 'pork', 'qualifier': 'cooked', 'intensity': 8}], 'texture': ['tender', 'succulent']}}, {'role': 'textural_element', 'ingredient_ref': 'ogs:core:ingredient:soft-boiled-egg', 'cooking_method': 'poached', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'umami': 5, 'salty': 3}, 'texture': ['creamy', 'silky']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:nori', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'umami': 8, 'salty': 4}, 'aromatics': [{'family': 'marine', 'group': 'seaweed', 'item': 'nori', 'intensity': 7}], 'texture': ['crisp']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:scallion', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'aromatics': [{'family': 'vegetal', 'group': 'allium', 'item': 'scallion', 'qualifier': 'fresh', 'intensity': 5}], 'texture': ['crisp']}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:tonkotsu-ramen",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              },
              "course": {
                "predicted": null,
                "expected": "main",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3752,
            "output_tokens": 1024,
            "latency_ms": 5359.0,
            "usd": 0.007098,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:tonkotsu-ramen\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"tonkotsu ramen\",\n  \"cuisine\": \"Japanese\",\n  \"course\": \"main\",\n  \"service_temperature\": \"very_hot\",\n  \"components\": [\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:pork-bone-broth\",\n      \"cooking_method\": \"braised\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 8,\n          \"salty\": 6\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"meat\",\n            \"item\": \"pork\",\n            \"intensity\": 7\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"starch\",\n      \"ingredient_ref\": \"ogs:core:ingredient:wheat-noodles\",\n      \"cooking_method\": \"blanched\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 2\n        },\n        \"texture\": [\n          \"chewy\",\n          \"elastic\"\n        ]\n      }\n    },\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:chashu-pork\",\n      \"cooking_method\": \"braised\",\n      \"maillard_index\": 3,\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 7,\n          \"salty\": 5\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"meat\",\n            \"item\": \"pork\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 8\n          }\n        ],\n        \"texture\": [\n          \"tender\",\n          \"succulent\"\n        ]\n      }\n    },\n    {\n      \"role\": \"textural_element\",\n      \"ingredient_ref\": \"ogs:core:ingredient:soft-boiled-egg\",\n      \"cooking_method\": \"poached\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 5,\n          \"salty\": 3\n        },\n        \"texture\": [\n          \"creamy\",\n          \"smooth\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:nori\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 8,\n          \"salty\": 4\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"seaweed\",\n            \"item\": \"nori\",\n            \"intensity\": 6\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:scallion\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"scallion\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1,\n      \"sour\": 0,\n      \"salty\": 5,\n      \"bitter\": 0",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'sauce', 'ingredient_ref': 'ogs:core:ingredient:pork-bone-broth', 'cooking_method': 'braised', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'umami': 8, 'salty': 6}, 'aromatics': [{'family': 'savory', 'group': 'meat', 'item': 'pork', 'intensity': 7}]}}, {'role': 'starch', 'ingredient_ref': 'ogs:core:ingredient:wheat-noodles', 'cooking_method': 'blanched', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'salty': 2}, 'texture': ['chewy', 'elastic']}}, {'role': 'main_protein', 'ingredient_ref': 'ogs:core:ingredient:chashu-pork', 'cooking_method': 'braised', 'maillard_index': 3, 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'umami': 7, 'salty': 5}, 'aromatics': [{'family': 'savory', 'group': 'meat', 'item': 'pork', 'qualifier': 'cooked', 'intensity': 8}], 'texture': ['tender', 'succulent']}}, {'role': 'textural_element', 'ingredient_ref': 'ogs:core:ingredient:soft-boiled-egg', 'cooking_method': 'poached', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'umami': 5, 'salty': 3}, 'texture': ['creamy', 'smooth']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:nori', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'umami': 8, 'salty': 4}, 'aromatics': [{'family': 'marine', 'group': 'seaweed', 'item': 'nori', 'intensity': 6}], 'texture': ['crisp']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:scallion', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'salty': 1}, 'aromatics': [{'family': 'vegetal', 'group': 'allium', 'item': 'scallion', 'qualifier': 'fresh', 'intensity': 4}], 'texture': ['crisp']}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:tonkotsu-ramen",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              },
              "course": {
                "predicted": null,
                "expected": "main",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3752,
            "output_tokens": 1024,
            "latency_ms": 5512.0,
            "usd": 0.007098,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:tonkotsu-ramen\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"tonkotsu ramen\",\n  \"description\": \"Japanese noodle soup with rich pork bone broth, wheat noodles, chashu pork, soft-boiled egg, nori, and scallion garnish\",\n  \"cuisine\": \"Japanese\",\n  \"course\": \"main\",\n  \"service_temperature\": \"very_hot\",\n  \"components\": [\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:pork-bone-broth\",\n      \"cooking_method\": \"braised\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 8,\n          \"salty\": 6\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"meat\",\n            \"item\": \"pork\",\n            \"intensity\": 8\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"starch\",\n      \"ingredient_ref\": \"ogs:core:ingredient:wheat-noodles\",\n      \"cooking_method\": \"blanched\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"texture\": [\n          \"chewy\",\n          \"elastic\"\n        ]\n      }\n    },\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:chashu-pork\",\n      \"cooking_method\": \"braised\",\n      \"maillard_index\": 4,\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 7,\n          \"salty\": 5\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"meat\",\n            \"item\": \"pork\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 7\n          }\n        ],\n        \"texture\": [\n          \"tender\",\n          \"melting\"\n        ]\n      }\n    },\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:soft-boiled-egg\",\n      \"cooking_method\": \"poached\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 5,\n          \"salty\": 3\n        },\n        \"texture\": [\n          \"creamy\",\n          \"runny\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:nori\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"umami\": 7,\n          \"salty\": 4\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"seaweed\",\n            \"item\": \"nori\",\n            \"intensity\": 6\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:scallion\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"scallion\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"fresh\",\n          \"crisp\"\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1,\n      \"sour\": 1,\n      \"salty",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'sauce', 'ingredient_ref': 'ogs:core:ingredient:pork-bone-broth', 'cooking_method': 'braised', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'umami': 8, 'salty': 6}, 'aromatics': [{'family': 'savory', 'group': 'meat', 'item': 'pork', 'intensity': 8}]}}, {'role': 'starch', 'ingredient_ref': 'ogs:core:ingredient:wheat-noodles', 'cooking_method': 'blanched', 'proportion': 'significant', 'sensory_contribution': {'texture': ['chewy', 'elastic']}}, {'role': 'main_protein', 'ingredient_ref': 'ogs:core:ingredient:chashu-pork', 'cooking_method': 'braised', 'maillard_index': 4, 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'umami': 7, 'salty': 5}, 'aromatics': [{'family': 'savory', 'group': 'meat', 'item': 'pork', 'qualifier': 'cooked', 'intensity': 7}], 'texture': ['tender', 'melting']}}, {'role': 'main_protein', 'ingredient_ref': 'ogs:core:ingredient:soft-boiled-egg', 'cooking_method': 'poached', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'umami': 5, 'salty': 3}, 'texture': ['creamy', 'runny']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:nori', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'umami': 7, 'salty': 4}, 'aromatics': [{'family': 'marine', 'group': 'seaweed', 'item': 'nori', 'intensity': 6}], 'texture': ['crisp']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:scallion', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'aromatics': [{'family': 'vegetal', 'group': 'allium', 'item': 'scallion', 'qualifier': 'fresh', 'intensity': 5}], 'texture': ['fresh', 'crisp']}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:tonkotsu-ramen",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              },
              "course": {
                "predicted": null,
                "expected": "main",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3752,
            "output_tokens": 1024,
            "latency_ms": 5790.6,
            "usd": 0.007098,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3752.0,
        "output_tokens_mean": 1024.0,
        "latency_ms_mean": 5553.867,
        "usd_sum": 0.021294
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-017",
      "category": "structured_output",
      "capability": "schema_valid_dish",
      "difficulty": "medium",
      "cuisine_region": "mexican",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.dish",
        "vocab.component_roles",
        "vocab.cooking_methods",
        "vocab.courses",
        "vocab.service_temperatures",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "534d8590759d6dc3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS dish schema (`schema/dish.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/dish.schema.json\",\n  \"title\": \"OGS Dish\",\n  \"description\": \"A composed dish with components, cooking methods, and overall sensory profile.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"components\",\n    \"overall_sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"cuisine\": {\n      \"type\": \"string\",\n      \"description\": \"Free-text cuisine name (v0.1) OR a cuisine entity reference matching ogs:<ns>:cuisine:<id> (v0.2+).\"\n    },\n    \"course\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"amuse\",\n        \"appetizer\",\n        \"soup\",\n        \"salad\",\n        \"fish\",\n        \"main\",\n        \"cheese\",\n        \"dessert\",\n        \"petit_four\"\n      ]\n    },\n    \"service_temperature\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"frozen\",\n        \"cold\",\n        \"cool\",\n        \"room\",\n        \"warm\",\n        \"hot\",\n        \"very_hot\"\n      ]\n    },\n    \"components\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/component\"\n      }\n    },\n    \"overall_sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"component\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"role\",\n        \"ingredient_ref\"\n      ],\n      \"properties\": {\n        \"role\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"main_protein\",\n            \"main_vegetable\",\n            \"starch\",\n            \"sauce\",\n            \"fat_component\",\n            \"acid_component\",\n            \"aromatic_garnish\",\n            \"seasoning\",\n            \"textural_element\"\n          ]\n        },\n        \"ingredient_ref\": {\n          \"type\": \"string\",\n          \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\"\n        },\n        \"cooking_method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"raw\",\n            \"seared\",\n            \"grilled\",\n            \"roasted\",\n            \"braised\",\n            \"poached\",\n            \"steamed\",\n            \"fried\",\n            \"confit\",\n            \"smoked\",\n            \"fermented\",\n            \"cured\",\n            \"pickled\",\n            \"dehydrated\",\n            \"sous_vide\",\n            \"blanched\",\n            \"sauteed\",\n            \"baked\",\n            \"charred\",\n            \"emulsified\"\n          ]\n        },\n        \"maillard_index\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"reduction_level\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"proportion\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"dominant\",\n            \"significant\",\n            \"accent\",\n            \"trace\"\n          ]\n        },\n        \"sensory_contribution\": {\n          \"type\": \"object\",\n          \"properties\": {\n            \"basic_tastes\": {\n              \"$ref\": \"#/$defs/basic_tastes\"\n            },\n            \"aromatics\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"$ref\": \"#/$defs/aromatic_entry\"\n              }\n            },\n            \"texture\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"type\": \"string\"\n              }\n            }\n          },\n          \"additionalProperties\": true\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS dish JSON for pico de gallo (tomato, onion, cilantro, jalapeño, lime). JSON only.\n\nINPUT:\n{\n  \"dish_name\": \"pico de gallo\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"pico-de-gallo\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/dish.schema.json",
        "required_fields": {
          "id": "ogs:core:dish:pico-de-gallo",
          "ogs_version": "0.1.0"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:pico-de-gallo\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"pico de gallo\",\n  \"course\": \"appetizer\",\n  \"service_temperature\": \"cold\",\n  \"components\": [\n    {\n      \"role\": \"main_vegetable\",\n      \"ingredient_ref\": \"ogs:core:ingredient:tomato\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 2,\n          \"sour\": 3,\n          \"salty\": 0,\n          \"umami\": 2\n        },\n        \"texture\": [\"juicy\", \"crisp\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:onion\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"sour\": 0,\n          \"salty\": 0,\n          \"bitter\": 0\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"onion\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\"crisp\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:cilantro\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"fresh herbs\",\n            \"item\": \"cilantro\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"tender\"]\n      }\n    },\n    {\n      \"role\": \"acid_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:lime\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 7,\n          \"salty\": 0,\n          \"sweet\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"citrus\",\n            \"item\": \"lime\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 6\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"seasoning\",\n      \"ingredient_ref\": \"ogs:core:ingredient:jalapeno\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"salty\": 0,\n          \"bitter\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"spice\",\n            \"group\": \"chili pepper\",\n            \"item\": \"jalapeño\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\"crisp\"]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 4,\n      \"salty\": 0,\n      \"bitter\": 0.5,\n      \"umami\": 1\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"allium\",\n        \"item\": \"onion\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"fresh herbs\",\n        \"item\": \"cilantro\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"fruit",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'main_vegetable', 'ingredient_ref': 'ogs:core:ingredient:tomato', 'cooking_method': 'raw', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'sweet': 2, 'sour': 3, 'salty': 0, 'umami': 2}, 'texture': ['juicy', 'crisp']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:onion', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sweet': 1, 'sour': 0, 'salty': 0, 'bitter': 0}, 'aromatics': [{'family': 'vegetal', 'group': 'allium', 'item': 'onion', 'qualifier': 'fresh', 'intensity': 4}], 'texture': ['crisp']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:cilantro', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'aromatics': [{'family': 'herbal', 'group': 'fresh herbs', 'item': 'cilantro', 'qualifier': 'fresh', 'intensity': 5}], 'texture': ['tender']}}, {'role': 'acid_component', 'ingredient_ref': 'ogs:core:ingredient:lime', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sour': 7, 'salty': 0, 'sweet': 1}, 'aromatics': [{'family': 'fruit', 'group': 'citrus', 'item': 'lime', 'qualifier': 'fresh', 'intensity': 6}]}}, {'role': 'seasoning', 'ingredient_ref': 'ogs:core:ingredient:jalapeno', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'sweet': 1, 'salty': 0, 'bitter': 1}, 'aromatics': [{'family': 'spice', 'group': 'chili pepper', 'item': 'jalapeño', 'qualifier': 'fresh', 'intensity': 4}], 'texture': ['crisp']}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:pico-de-gallo",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3735,
            "output_tokens": 1024,
            "latency_ms": 7672.0,
            "usd": 0.007084,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:pico-de-gallo\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"pico de gallo\",\n  \"course\": \"appetizer\",\n  \"service_temperature\": \"cold\",\n  \"components\": [\n    {\n      \"role\": \"main_vegetable\",\n      \"ingredient_ref\": \"ogs:core:ingredient:tomato\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 2,\n          \"sour\": 3,\n          \"salty\": 0,\n          \"bitter\": 0,\n          \"umami\": 3\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"tomato\",\n            \"item\": \"tomato\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 7\n          }\n        ],\n        \"texture\": [\n          \"juicy\",\n          \"firm\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:onion\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 2,\n          \"sour\": 0,\n          \"salty\": 0,\n          \"bitter\": 0,\n          \"umami\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"onion\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 6\n          }\n        ],\n        \"texture\": [\n          \"crisp\",\n          \"crunchy\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:cilantro\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"fresh herbs\",\n            \"item\": \"cilantro\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 6\n          }\n        ],\n        \"texture\": [\n          \"tender\"\n        ]\n      }\n    },\n    {\n      \"role\": \"seasoning\",\n      \"ingredient_ref\": \"ogs:core:ingredient:jalapeno\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"sour\": 0,\n          \"salty\": 0,\n          \"bitter\": 0,\n          \"umami\": 0\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"spice\",\n            \"group\": \"chili pepper\",\n            \"item\": \"jalapeño\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    },\n    {\n      \"role\": \"acid_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:lime\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"sour\": 7,\n          \"salty\": 0,\n          \"bitter\": 0,\n          \"umami\": 0\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"citrus\",\n            \"item\": \"lime\",\n            \"qualifier\": \"fresh\",\n            \"qualifier\": \"zest\",\n            \"intensity\": 4\n          }\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"s",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'main_vegetable', 'ingredient_ref': 'ogs:core:ingredient:tomato', 'cooking_method': 'raw', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'sweet': 2, 'sour': 3, 'salty': 0, 'bitter': 0, 'umami': 3}, 'aromatics': [{'family': 'vegetal', 'group': 'tomato', 'item': 'tomato', 'qualifier': 'fresh', 'intensity': 7}], 'texture': ['juicy', 'firm']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:onion', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sweet': 2, 'sour': 0, 'salty': 0, 'bitter': 0, 'umami': 1}, 'aromatics': [{'family': 'vegetal', 'group': 'allium', 'item': 'onion', 'qualifier': 'fresh', 'intensity': 6}], 'texture': ['crisp', 'crunchy']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:cilantro', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'aromatics': [{'family': 'herbal', 'group': 'fresh herbs', 'item': 'cilantro', 'qualifier': 'fresh', 'intensity': 6}], 'texture': ['tender']}}, {'role': 'seasoning', 'ingredient_ref': 'ogs:core:ingredient:jalapeno', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'sweet': 1, 'sour': 0, 'salty': 0, 'bitter': 0, 'umami': 0}, 'aromatics': [{'family': 'spice', 'group': 'chili pepper', 'item': 'jalapeño', 'qualifier': 'fresh', 'intensity': 5}], 'texture': ['crisp']}}, {'role': 'acid_component', 'ingredient_ref': 'ogs:core:ingredient:lime', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'sweet': 1, 'sour': 7, 'salty': 0, 'bitter': 0, 'umami': 0}, 'aromatics': [{'family': 'fruit', 'group': 'citrus', 'item': 'lime', 'qualifier': 'zest', 'intensity': 4}]}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:pico-de-gallo",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3735,
            "output_tokens": 1024,
            "latency_ms": 5032.8,
            "usd": 0.007084,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:dish:pico-de-gallo\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"pico de gallo\",\n  \"course\": \"appetizer\",\n  \"service_temperature\": \"cold\",\n  \"components\": [\n    {\n      \"role\": \"main_vegetable\",\n      \"ingredient_ref\": \"ogs:core:ingredient:tomato\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 3,\n          \"salty\": 0,\n          \"umami\": 2\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"tomato\",\n            \"item\": \"fresh tomato\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 7\n          }\n        ],\n        \"texture\": [\n          \"juicy\",\n          \"crisp\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:onion\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 2,\n          \"salty\": 0\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"onion\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:cilantro\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"herbs\",\n            \"item\": \"cilantro\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 6\n          }\n        ]\n      }\n    },\n    {\n      \"role\": \"seasoning\",\n      \"ingredient_ref\": \"ogs:core:ingredient:jalapeno\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"bitter\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"spice\",\n            \"group\": \"chile\",\n            \"item\": \"jalapeño\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"crisp\"\n        ]\n      }\n    },\n    {\n      \"role\": \"acid_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:lime\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 8\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"citrus\",\n            \"item\": \"lime\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 6\n          }\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 6,\n      \"salty\": 1,\n      \"bitter\": 0.5,\n      \"umami\": 1\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"tomato\",\n        \"item\": \"fresh tomato\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"herbs\",\n        \"item\": \"cilantro\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5\n      },\n      {",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "[{'role': 'main_vegetable', 'ingredient_ref': 'ogs:core:ingredient:tomato', 'cooking_method': 'raw', 'proportion': 'dominant', 'sensory_contribution': {'basic_tastes': {'sour': 3, 'salty': 0, 'umami': 2}, 'aromatics': [{'family': 'vegetal', 'group': 'tomato', 'item': 'fresh tomato', 'qualifier': 'fresh', 'intensity': 7}], 'texture': ['juicy', 'crisp']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:onion', 'cooking_method': 'raw', 'proportion': 'significant', 'sensory_contribution': {'basic_tastes': {'sweet': 2, 'salty': 0}, 'aromatics': [{'family': 'vegetal', 'group': 'allium', 'item': 'onion', 'qualifier': 'fresh', 'intensity': 5}], 'texture': ['crisp']}}, {'role': 'aromatic_garnish', 'ingredient_ref': 'ogs:core:ingredient:cilantro', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'aromatics': [{'family': 'herbal', 'group': 'herbs', 'item': 'cilantro', 'qualifier': 'fresh', 'intensity': 6}]}}, {'role': 'seasoning', 'ingredient_ref': 'ogs:core:ingredient:jalapeno', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'bitter': 1}, 'aromatics': [{'family': 'spice', 'group': 'chile', 'item': 'jalapeño', 'qualifier': 'fresh', 'intensity': 5}], 'texture': ['crisp']}}, {'role': 'acid_component', 'ingredient_ref': 'ogs:core:ingredient:lime', 'cooking_method': 'raw', 'proportion': 'accent', 'sensory_contribution': {'basic_tastes': {'sour': 8}, 'aromatics': [{'family': 'fruit', 'group': 'citrus', 'item': 'lime', 'qualifier': 'fresh', 'intensity': 6}]}}] is not of type 'object'"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": null,
                "expected": "ogs:core:dish:pico-de-gallo",
                "match": false
              },
              "ogs_version": {
                "predicted": null,
                "expected": "0.1.0",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 3735,
            "output_tokens": 1024,
            "latency_ms": 5155.7,
            "usd": 0.007084,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "max_tokens"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3735.0,
        "output_tokens_mean": 1024.0,
        "latency_ms_mean": 5953.5,
        "usd_sum": 0.021252
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-018",
      "category": "structured_output",
      "capability": "sensory_profile_json",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "json_match",
      "context_needs": [
        "scales.basic_tastes",
        "scales.structural",
        "vocab.aromatic_families",
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "1470a28b8e9ef028",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nProduce a JSON sensory profile for 24-month aged Comté cheese. Keys: basic_tastes, structural, aromatics. All intensities on 0-10. JSON only.\n\nINPUT:\n{\n  \"item\": \"24-month aged Comté\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "value": {
          "basic_tastes": {
            "sweet": 2.0,
            "sour": 2.0,
            "salty": 5.0,
            "bitter": 1.5,
            "umami": 8.0
          },
          "structural": {
            "body": 7.0,
            "acidity_perception": 3.0,
            "tannin": 0,
            "astringency": 0,
            "alcohol_heat": 0,
            "effervescence": 0
          }
        },
        "required_keys": [
          "basic_tastes.umami",
          "basic_tastes.salty",
          "structural.body"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 3.5,\n    \"sour\": 2.0,\n    \"salty\": 6.5,\n    \"bitter\": 1.0,\n    \"umami\": 7.5\n  },\n  \"structural\": {\n    \"body\": 8.0,\n    \"acidity_perception\": 2.5,\n    \"tannin\": 0.0,\n    \"astringency\": 0.5,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": {\n    \"fruit\": {\n      \"family\": \"fruit\",\n      \"qualifier\": \"dried\",\n      \"intensity\": 4.5\n    },\n    \"floral\": {\n      \"family\": \"floral\",\n      \"qualifier\": \"fresh\",\n      \"intensity\": 2.0\n    },\n    \"herbal\": {\n      \"family\": \"herbal\",\n      \"qualifier\": \"dried\",\n      \"intensity\": 3.0\n    },\n    \"earth\": {\n      \"family\": \"earth\",\n      \"qualifier\": \"fresh\",\n      \"intensity\": 5.5\n    },\n    \"dairy\": {\n      \"family\": \"dairy\",\n      \"qualifier\": \"cooked\",\n      \"intensity\": 7.0\n    },\n    \"savory\": {\n      \"family\": \"savory\",\n      \"qualifier\": \"fermented\",\n      \"intensity\": 6.5\n    },\n    \"spice\": {\n      \"family\": \"spice\",\n      \"qualifier\": \"dried\",\n      \"intensity\": 2.5\n    }\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.umami": {
                "predicted": 7.5,
                "expected": 8.0,
                "match": true
              },
              "basic_tastes.salty": {
                "predicted": 6.5,
                "expected": 5.0,
                "match": true
              },
              "structural.body": {
                "predicted": 8.0,
                "expected": 7.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 523,
            "output_tokens": 412,
            "latency_ms": 2761.4,
            "usd": 0.002066,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 3.5,\n    \"sour\": 2.0,\n    \"salty\": 6.5,\n    \"bitter\": 1.0,\n    \"umami\": 7.5\n  },\n  \"structural\": {\n    \"body\": 8.0,\n    \"acidity_perception\": 2.5,\n    \"tannin\": 0.0,\n    \"astringency\": 1.0,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": {\n    \"fruit\": {\n      \"family\": \"fruit\",\n      \"qualifiers\": [\"dried\", \"cooked\"],\n      \"intensity\": 6.5\n    },\n    \"dairy\": {\n      \"family\": \"dairy\",\n      \"qualifiers\": [\"cooked\"],\n      \"intensity\": 7.0\n    },\n    \"nutty\": {\n      \"family\": \"confection\",\n      \"qualifiers\": [\"dried\"],\n      \"intensity\": 7.5\n    },\n    \"earth\": {\n      \"family\": \"earth\",\n      \"qualifiers\": [],\n      \"intensity\": 5.0\n    },\n    \"savory\": {\n      \"family\": \"savory\",\n      \"qualifiers\": [],\n      \"intensity\": 6.5\n    }\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.umami": {
                "predicted": 7.5,
                "expected": 8.0,
                "match": true
              },
              "basic_tastes.salty": {
                "predicted": 6.5,
                "expected": 5.0,
                "match": true
              },
              "structural.body": {
                "predicted": 8.0,
                "expected": 7.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 523,
            "output_tokens": 344,
            "latency_ms": 2167.1,
            "usd": 0.001794,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"item\": \"24-month aged Comté\",\n  \"basic_tastes\": {\n    \"sweet\": 4.5,\n    \"sour\": 2.0,\n    \"salty\": 5.5,\n    \"bitter\": 1.5,\n    \"umami\": 7.0\n  },\n  \"structural\": {\n    \"body\": 8.0,\n    \"acidity_perception\": 2.5,\n    \"tannin\": 0.0,\n    \"astringency\": 1.0,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": [\n    {\n      \"family\": \"dairy\",\n      \"qualifiers\": [\"cooked\"],\n      \"intensity\": 8.0\n    },\n    {\n      \"family\": \"fruit\",\n      \"qualifiers\": [\"dried\"],\n      \"intensity\": 6.5\n    },\n    {\n      \"family\": \"earth\",\n      \"qualifiers\": [],\n      \"intensity\": 5.0\n    },\n    {\n      \"family\": \"savory\",\n      \"qualifiers\": [],\n      \"intensity\": 7.5\n    },\n    {\n      \"family\": \"confection\",\n      \"qualifiers\": [\"candied\"],\n      \"intensity\": 4.5\n    },\n    {\n      \"family\": \"spice\",\n      \"qualifiers\": [],\n      \"intensity\": 3.0\n    }\n  ]\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.umami": {
                "predicted": 7.0,
                "expected": 8.0,
                "match": true
              },
              "basic_tastes.salty": {
                "predicted": 5.5,
                "expected": 5.0,
                "match": true
              },
              "structural.body": {
                "predicted": 8.0,
                "expected": 7.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 523,
            "output_tokens": 369,
            "latency_ms": 2235.1,
            "usd": 0.001894,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 523.0,
        "output_tokens_mean": 375.0,
        "latency_ms_mean": 2387.867,
        "usd_sum": 0.005754
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-019",
      "category": "structured_output",
      "capability": "sensory_profile_json",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "json_match",
      "context_needs": [
        "scales.basic_tastes",
        "scales.structural",
        "vocab.aromatic_families",
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "aa2726470aa7cfb8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nProduce a JSON sensory profile for blistered shishito peppers lightly salted. Keys: basic_tastes, structural, aromatics.\n\nINPUT:\n{\n  \"item\": \"blistered shishito peppers with flaky salt\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "value": {
          "basic_tastes": {
            "sweet": 2.0,
            "sour": 1.0,
            "salty": 5.0,
            "bitter": 3.0,
            "umami": 2.0
          },
          "structural": {
            "body": 3.0,
            "acidity_perception": 2.0,
            "tannin": 0,
            "astringency": 1.0,
            "alcohol_heat": 0,
            "effervescence": 0
          }
        },
        "required_keys": [
          "basic_tastes.salty",
          "basic_tastes.bitter",
          "structural.body"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 3.5,\n    \"sour\": 1.0,\n    \"salty\": 6.0,\n    \"bitter\": 1.5,\n    \"umami\": 5.5\n  },\n  \"structural\": {\n    \"body\": 3.0,\n    \"acidity_perception\": 1.0,\n    \"tannin\": 0.0,\n    \"astringency\": 0.0,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": {\n    \"families\": [\n      {\n        \"family\": \"vegetal\",\n        \"qualifiers\": [\"fresh\", \"cooked\"]\n      },\n      {\n        \"family\": \"fruit\",\n        \"qualifiers\": [\"fresh\"]\n      },\n      {\n        \"family\": \"savory\",\n        \"qualifiers\": []\n      }\n    ]\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.salty": {
                "predicted": 6.0,
                "expected": 5.0,
                "match": true
              },
              "basic_tastes.bitter": {
                "predicted": 1.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 3.0,
                "expected": 3.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 519,
            "output_tokens": 246,
            "latency_ms": 2026.3,
            "usd": 0.001399,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 3.5,\n    \"sour\": 1.0,\n    \"salty\": 6.0,\n    \"bitter\": 1.5,\n    \"umami\": 5.5\n  },\n  \"structural\": {\n    \"body\": 2.5,\n    \"acidity_perception\": 1.0,\n    \"tannin\": 0.0,\n    \"astringency\": 0.0,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": {\n    \"families\": [\n      \"vegetal\",\n      \"savory\",\n      \"fruit\"\n    ],\n    \"qualifiers\": [\n      \"fresh\",\n      \"cooked\"\n    ]\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.salty": {
                "predicted": 6.0,
                "expected": 5.0,
                "match": true
              },
              "basic_tastes.bitter": {
                "predicted": 1.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 2.5,
                "expected": 3.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 519,
            "output_tokens": 211,
            "latency_ms": 1530.7,
            "usd": 0.001259,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"basic_tastes\": {\n    \"sweet\": 3.5,\n    \"sour\": 1.0,\n    \"salty\": 6.0,\n    \"bitter\": 1.5,\n    \"umami\": 5.5\n  },\n  \"structural\": {\n    \"body\": 2.0,\n    \"acidity_perception\": 1.0,\n    \"tannin\": 0.0,\n    \"astringency\": 0.5,\n    \"alcohol_heat\": 0.0,\n    \"effervescence\": 0.0\n  },\n  \"aromatics\": {\n    \"families\": [\n      {\n        \"family\": \"vegetal\",\n        \"qualifiers\": [\"fresh\", \"cooked\"]\n      },\n      {\n        \"family\": \"savory\",\n        \"qualifiers\": []\n      },\n      {\n        \"family\": \"fruit\",\n        \"qualifiers\": [\"fresh\"]\n      }\n    ]\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.salty": {
                "predicted": 6.0,
                "expected": 5.0,
                "match": true
              },
              "basic_tastes.bitter": {
                "predicted": 1.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 2.0,
                "expected": 3.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 519,
            "output_tokens": 246,
            "latency_ms": 1766.2,
            "usd": 0.001399,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 519.0,
        "output_tokens_mean": 234.333,
        "latency_ms_mean": 1774.4,
        "usd_sum": 0.004057
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-020",
      "category": "structured_output",
      "capability": "schema_valid_pairing",
      "difficulty": "expert",
      "cuisine_region": "mexican",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.pairing",
        "vocab.match_types",
        "vocab.explanation_codes",
        "vocab.grade_scale",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "5b115600e8dad71f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS pairing schema (`schema/pairing.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/pairing.schema.json\",\n  \"title\": \"OGS Pairing\",\n  \"description\": \"A scored, explained pairing between a dish and a beverage.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"dish_ref\",\n    \"beverage_ref\",\n    \"match_types\",\n    \"overall_score\",\n    \"grade\",\n    \"explanation_codes\",\n    \"confidence\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:pairing:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"dish_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"beverage_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"match_types\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/match_type_entry\"\n      }\n    },\n    \"overall_score\": {\n      \"type\": \"integer\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"grade\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"A+\",\n        \"A\",\n        \"B+\",\n        \"B\",\n        \"C+\",\n        \"C\",\n        \"D\",\n        \"F\"\n      ]\n    },\n    \"explanation_codes\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/explanation_code_entry\"\n      }\n    },\n    \"risks\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/risk_entry\"\n      }\n    },\n    \"confidence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 1\n    },\n    \"evidence\": {\n      \"$ref\": \"#/$defs/evidence\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"match_type_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"type\",\n        \"score\"\n      ],\n      \"properties\": {\n        \"type\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"harmony\",\n            \"contrast\",\n            \"bridge\",\n            \"intensity_alignment\",\n            \"structural_balance\",\n            \"refresh\"\n          ]\n        },\n        \"score\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 100\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"primary\": {\n          \"type\": \"boolean\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"explanation_code_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"weight\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 1\n        },\n        \"detail\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"risk_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\",\n        \"severity\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"severity\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"low\",\n            \"medium\",\n            \"high\"\n          ]\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"mitigation\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"evidence\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"expert_panel\",\n            \"instrumental\",\n            \"computed\",\n            \"estimated\"\n          ]\n        },\n        \"sources\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Explanation Codes\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS pairing JSON for mole-negro enchiladas with Aragonese Garnacha. JSON only.\n\nINPUT:\n{\n  \"pairing_slug\": \"mole-negro-garnacha\",\n  \"dish_ref\": \"ogs:core:dish:mole-negro-enchiladas\",\n  \"beverage_ref\": \"ogs:core:beverage:garnacha-aragon\",\n  \"intent\": \"Cross-cultural bridge pairing: ripe fruit and body handle chocolate-chile sauce.\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/pairing.schema.json",
        "required_fields": {
          "id": "ogs:core:pairing:mole-negro-garnacha",
          "ogs_version": "0.1.0",
          "dish_ref": "ogs:core:dish:mole-negro-enchiladas",
          "beverage_ref": "ogs:core:beverage:garnacha-aragon"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:mole-negro-garnacha\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Mole Negro Enchiladas with Aragonese Garnacha\",\n  \"dish_ref\": \"ogs:core:dish:mole-negro-enchiladas\",\n  \"beverage_ref\": \"ogs:core:beverage:garnacha-aragon\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 88,\n      \"description\": \"Shared spice and dark fruit aromatics create a unified flavor experience\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"bridge\",\n      \"score\": 85,\n      \"description\": \"Garnacha's ripe red fruit and earthy notes bridge the chocolate and chile complexity\"\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 82,\n      \"description\": \"Wine's body and tannin structure complement the rich mole sauce without overwhelming\"\n    }\n  ],\n  \"overall_score\": 85,\n  \"grade\": \"B+\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"AROMA_SHARED_SPICE\",\n      \"weight\": 0.35,\n      \"detail\": \"Both mole negro and Garnacha feature warm spice notes—chile, cinnamon, and clove—creating aromatic continuity\"\n    },\n    {\n      \"code\": \"AROMA_SHARED_FRUIT\",\n      \"weight\": 0.30,\n      \"detail\": \"Garnacha's ripe red and dark berry aromatics echo the fruity undertones in chocolate-based mole\"\n    },\n    {\n      \"code\": \"STRUCT_BODY_ALIGN\",\n      \"weight\": 0.25,\n      \"detail\": \"Medium-bodied Garnacha matches the weight and richness of mole-smothered enchiladas\"\n    },\n    {\n      \"code\": \"HARMONY_COMPLEXITY_MATCH\",\n      \"weight\": 0.10,\n      \"detail\": \"Both dish and wine display multi-layered complexity without either element dominating\"\n    }\n  ],\n  \"confidence\": 0.78,\n  \"evidence\": {\n    \"method\": \"estimated\",\n    \"notes\": \"Cross-cultural bridge pairing leveraging Garnacha's ripe fruit and body to balance mole negro's chocolate-chile sauce complexity\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS Benchmark Task\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:mole-negro-garnacha",
                "expected": "ogs:core:pairing:mole-negro-garnacha",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:mole-negro-enchiladas",
                "expected": "ogs:core:dish:mole-negro-enchiladas",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:garnacha-aragon",
                "expected": "ogs:core:beverage:garnacha-aragon",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4191,
            "output_tokens": 639,
            "latency_ms": 5510.3,
            "usd": 0.005909,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:mole-negro-garnacha\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Mole Negro Enchiladas with Aragonese Garnacha\",\n  \"dish_ref\": \"ogs:core:dish:mole-negro-enchiladas\",\n  \"beverage_ref\": \"ogs:core:beverage:garnacha-aragon\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 88,\n      \"description\": \"Ripe red fruit notes in Garnacha complement the complex chocolate and chile aromatics of mole negro\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"bridge\",\n      \"score\": 85,\n      \"description\": \"Shared spice and earth notes create an aromatic bridge between the two cuisines\"\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 82,\n      \"description\": \"Medium body and moderate tannins of Garnacha align with the rich, layered mole sauce\"\n    },\n    {\n      \"type\": \"refresh\",\n      \"score\": 80,\n      \"description\": \"Wine acidity cuts through the richness of the enchilada sauce, cleansing the palate\"\n    }\n  ],\n  \"overall_score\": 84,\n  \"grade\": \"B+\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"AROMA_SHARED_SPICE\",\n      \"weight\": 0.25,\n      \"detail\": \"Garnacha's subtle spice notes echo the layered chile spices in mole negro\"\n    },\n    {\n      \"code\": \"AROMA_SHARED_EARTH\",\n      \"weight\": 0.20,\n      \"detail\": \"Earthy undertones in both dish and wine create aromatic continuity\"\n    },\n    {\n      \"code\": \"STRUCT_BODY_ALIGN\",\n      \"weight\": 0.22,\n      \"detail\": \"Medium body of Aragonese Garnacha matches the weight and richness of mole sauce\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.18,\n      \"detail\": \"Acidity in the wine provides brightness against chocolate and chile depth\"\n    },\n    {\n      \"code\": \"HARMONY_COMPLEXITY_MATCH\",\n      \"weight\": 0.15,\n      \"detail\": \"Both elements share sophisticated complexity without one overwhelming the other\"\n    }\n  ],\n  \"confidence\": 0.82,\n  \"evidence\": {\n    \"method\": \"expert_panel\",\n    \"notes\": \"Cross-cultural bridge pairing leveraging Garnacha's ripe fruit body and acidity to complement mole negro's complex chocolate-chile sauce profile\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:mole-negro-garnacha",
                "expected": "ogs:core:pairing:mole-negro-garnacha",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:mole-negro-enchiladas",
                "expected": "ogs:core:dish:mole-negro-enchiladas",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:garnacha-aragon",
                "expected": "ogs:core:beverage:garnacha-aragon",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4191,
            "output_tokens": 708,
            "latency_ms": 5279.5,
            "usd": 0.006185,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n{\n  \"id\": \"ogs:core:pairing:mole-negro-garnacha\",\n  \"ogs_version\": \"0.1.0\",\n  \"name\": \"Mole Negro Enchiladas with Aragonese Garnacha\",\n  \"dish_ref\": \"ogs:core:dish:mole-negro-enchiladas\",\n  \"beverage_ref\": \"ogs:core:beverage:garnacha-aragon\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 88,\n      \"description\": \"The ripe red fruit character of Garnacha harmonizes with the complex chocolate-chile sauce of mole negro, creating a unified flavor experience.\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"bridge\",\n      \"score\": 85,\n      \"description\": \"Shared spice and earth aromatics bridge the Mexican mole tradition with Spanish wine terroir.\"\n    },\n    {\n      \"type\": \"intensity_alignment\",\n      \"score\": 82,\n      \"description\": \"The medium body and moderate alcohol of Aragonese Garnacha align with the intensity of the layered mole sauce without overpowering.\"\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 80,\n      \"description\": \"The wine's natural acidity cuts through the richness of the enchilada sauce while tannins integrate smoothly with the dish's protein and fat content.\"\n    }\n  ],\n  \"overall_score\": 84,\n  \"grade\": \"B\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"AROMA_SHARED_FRUIT\",\n      \"weight\": 0.25,\n      \"detail\": \"Both mole negro (dried fruit notes) and Garnacha (red berries, plum) share ripe fruit aromatics that create continuity.\"\n    },\n    {\n      \"code\": \"AROMA_SHARED_SPICE\",\n      \"weight\": 0.20,\n      \"detail\": \"Spice notes from mole negro's cinnamon, clove, and anise align with subtle spice nuances in the wine.\"\n    },\n    {\n      \"code\": \"STRUCT_BODY_ALIGN\",\n      \"weight\": 0.18,\n      \"detail\": \"Medium-bodied Garnacha matches the full-flavored sauce without becoming thin or heavy.\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.18,\n      \"detail\": \"The wine's acidity refreshes the palate between rich, sauce-laden bites of enchiladas.\"\n    },\n    {\n      \"code\": \"HARMONY_COMPLEXITY_MATCH\",\n      \"weight\": 0.15,\n      \"detail\": \"Both elements present sophisticated, layered complexity that neither element overwhelms.\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.04,\n      \"detail\": \"Garnacha's natural acidity provides palate refreshment against the chocolate-chile richness.\"\n    }\n  ],\n  \"confidence\": 0.82,\n  \"evidence\": {\n    \"method\": \"computed\",\n    \"sources\": [\n      \"Regional pairing principles\",\n      \"Flavor compound analysis\",\n      \"Structural compatibility assessment\"\n    ],\n    \"notes\": \"Cross-cultural pairing leveraging structural harmony between Old World Spanish wine and Mexican culinary tradition.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS Benchmark\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}\n```",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:mole-negro-garnacha",
                "expected": "ogs:core:pairing:mole-negro-garnacha",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:mole-negro-enchiladas",
                "expected": "ogs:core:dish:mole-negro-enchiladas",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:garnacha-aragon",
                "expected": "ogs:core:beverage:garnacha-aragon",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 4191,
            "output_tokens": 883,
            "latency_ms": 7707.3,
            "usd": 0.006885,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 4191.0,
        "output_tokens_mean": 743.333,
        "latency_ms_mean": 6165.7,
        "usd_sum": 0.018979
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-001",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "e52270d817a9dc7f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family (from the controlled vocabulary) that best classifies the aromatic item below. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"lemon zest\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "fruit"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 380,
            "output_tokens": 4,
            "latency_ms": 456.0,
            "usd": 0.00032,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 380,
            "output_tokens": 4,
            "latency_ms": 511.6,
            "usd": 0.00032,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 380,
            "output_tokens": 4,
            "latency_ms": 604.6,
            "usd": 0.00032,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 380.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 524.067,
        "usd_sum": 0.00096
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-002",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "e2096242953ff882",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family that best classifies the item below. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"wet stone\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "earth"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "earth",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "earth",
            "expected": "earth",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 507.2,
            "usd": 0.000312,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "earth",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "earth",
            "expected": "earth",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 501.7,
            "usd": 0.000312,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "earth",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "earth",
            "expected": "earth",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 661.2,
            "usd": 0.000312,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 370.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 556.7,
        "usd_sum": 0.000936
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-003",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "fa5fa32bfe275dfb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family that best classifies the item below. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"cooked beef fat\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "savory"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "savory",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "savory",
            "expected": "savory",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 372,
            "output_tokens": 5,
            "latency_ms": 518.0,
            "usd": 0.000318,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "savory",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "savory",
            "expected": "savory",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 372,
            "output_tokens": 5,
            "latency_ms": 698.5,
            "usd": 0.000318,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "savory",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "savory",
            "expected": "savory",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 372,
            "output_tokens": 5,
            "latency_ms": 674.8,
            "usd": 0.000318,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 372.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 630.433,
        "usd_sum": 0.000954
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-004",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "e0631034b8b8ece4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best describes a protein slowly cooked while fully submerged in fat at a low temperature. Respond with just the method name from the OGS vocabulary.\n\nINPUT:\n{\n  \"description\": \"duck legs slow-cooked submerged in their own rendered fat at 80 C\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "confit"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "confit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "confit",
            "expected": "confit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 469,
            "output_tokens": 5,
            "latency_ms": 516.7,
            "usd": 0.000395,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "confit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "confit",
            "expected": "confit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 469,
            "output_tokens": 5,
            "latency_ms": 517.6,
            "usd": 0.000395,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "confit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "confit",
            "expected": "confit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 469,
            "output_tokens": 5,
            "latency_ms": 536.5,
            "usd": 0.000395,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 469.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 523.6,
        "usd_sum": 0.001185
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-005",
      "category": "vocabulary",
      "capability": "match_type_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "bbbd81bb3c8b8ea4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nSelect the single OGS match type that best describes a pairing whose strength is driven by a specific shared aromatic compound or family between dish and beverage.\n\nINPUT:\n{\n  \"description\": \"dish and wine are linked by a shared earthy, forest-floor aromatic\"\n}\n\nANSWER FORMAT: one_of:harmony,contrast,bridge,intensity_alignment,structural_balance,refresh",
      "expected": {
        "value": "bridge"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "bridge",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "bridge",
            "expected": "bridge",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 1496.5,
            "usd": 0.000304,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "bridge",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "bridge",
            "expected": "bridge",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 984.2,
            "usd": 0.000304,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "harmony",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "harmony",
            "expected": "bridge",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 731.5,
            "usd": 0.000304,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": 0.0133,
        "ci95_hi": 1.32,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 360.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1070.733,
        "usd_sum": 0.000912
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-006",
      "category": "vocabulary",
      "capability": "component_role_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "6c6ae6f82f354be3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits the described element. Use only roles from the OGS controlled vocabulary.\n\nINPUT:\n{\n  \"element\": \"a soy-dashi based liquid ladled over the rice bowl, providing salt, umami and moisture\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "sauce"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauce",
            "expected": "sauce",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 4,
            "latency_ms": 790.6,
            "usd": 0.000335,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauce",
            "expected": "sauce",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 4,
            "latency_ms": 1090.0,
            "usd": 0.000335,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauce",
            "expected": "sauce",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 4,
            "latency_ms": 856.0,
            "usd": 0.000335,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 399.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 912.2,
        "usd_sum": 0.001005
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-007",
      "category": "vocabulary",
      "capability": "qualifier_vocabulary",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "0adac8c06022acf5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nSelect the OGS aromatic qualifier that best describes how the item is expressed.\n\nINPUT:\n{\n  \"item\": \"candied ginger\"\n}\n\nANSWER FORMAT: one_of:fresh,dried,cooked,candied,fermented,zest,overripe",
      "expected": {
        "value": "candied"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "candied",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "candied",
            "expected": "candied",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 343,
            "output_tokens": 5,
            "latency_ms": 503.9,
            "usd": 0.000294,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "candied",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "candied",
            "expected": "candied",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 343,
            "output_tokens": 5,
            "latency_ms": 563.3,
            "usd": 0.000294,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "candied",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "candied",
            "expected": "candied",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 343,
            "output_tokens": 5,
            "latency_ms": 1571.5,
            "usd": 0.000294,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 343.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 879.567,
        "usd_sum": 0.000882
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-008",
      "category": "vocabulary",
      "capability": "grade_scale",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "e58319dc1276cdb2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nUsing the OGS pairing grade scale (ogs-match.md §5), return the letter grade for the given overall score. Respond with just the grade token, e.g. 'A+'.\n\nINPUT:\n{\n  \"overall_score\": 92\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "A"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "A",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "A",
            "expected": "A",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 431,
            "output_tokens": 4,
            "latency_ms": 579.7,
            "usd": 0.000361,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "A",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "A",
            "expected": "A",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 431,
            "output_tokens": 4,
            "latency_ms": 582.6,
            "usd": 0.000361,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "A",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "A",
            "expected": "A",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 431,
            "output_tokens": 4,
            "latency_ms": 477.3,
            "usd": 0.000361,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 431.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 546.533,
        "usd_sum": 0.001083
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-009",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "d51864ab27ecfa64",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family that best characterises smoked paprika. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"smoked paprika (pimentón de la Vera)\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "spice"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "spice",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spice",
            "expected": "spice",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 381,
            "output_tokens": 5,
            "latency_ms": 701.5,
            "usd": 0.000325,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "spice",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spice",
            "expected": "spice",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 381,
            "output_tokens": 5,
            "latency_ms": 454.6,
            "usd": 0.000325,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "spice",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spice",
            "expected": "spice",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 381,
            "output_tokens": 5,
            "latency_ms": 436.4,
            "usd": 0.000325,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 381.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 530.833,
        "usd_sum": 0.000975
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-010",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "1748069ad0ebdde7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single dominant aromatic family of a classic kombu-katsuo dashi.\n\nINPUT:\n{\n  \"item\": \"classic kombu-katsuobushi dashi\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "marine"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "savory",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "savory",
            "expected": "marine",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 375,
            "output_tokens": 5,
            "latency_ms": 443.4,
            "usd": 0.00032,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "savory",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "savory",
            "expected": "marine",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 375,
            "output_tokens": 5,
            "latency_ms": 619.8,
            "usd": 0.00032,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "savory",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "savory",
            "expected": "marine",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 375,
            "output_tokens": 5,
            "latency_ms": 706.3,
            "usd": 0.00032,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 375.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 589.833,
        "usd_sum": 0.00096
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-011",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "easy",
      "cuisine_region": "indian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "85de6add5c1ebfba",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best matches tandoori chicken cooked at 450°C in a clay oven.\n\nINPUT:\n{\n  \"description\": \"tandoori chicken, marinated and cooked in a clay tandoor at ~450°C\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "roasted"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 456,
            "output_tokens": 5,
            "latency_ms": 529.8,
            "usd": 0.000385,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 456,
            "output_tokens": 5,
            "latency_ms": 544.2,
            "usd": 0.000385,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 456,
            "output_tokens": 5,
            "latency_ms": 554.8,
            "usd": 0.000385,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 456.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 542.933,
        "usd_sum": 0.001155
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-012",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "peruvian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "44c9b26dc70c4557",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best describes Peruvian ceviche's treatment of white fish in lime juice.\n\nINPUT:\n{\n  \"description\": \"diced white fish denatured in freshly squeezed lime juice for 10 minutes with red onion and aji limo\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "cured"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "raw",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "raw",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 464,
            "output_tokens": 4,
            "latency_ms": 1119.1,
            "usd": 0.000387,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "raw",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "raw",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 464,
            "output_tokens": 4,
            "latency_ms": 1164.5,
            "usd": 0.000387,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "raw",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "raw",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 464,
            "output_tokens": 4,
            "latency_ms": 603.3,
            "usd": 0.000387,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 464.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 962.3,
        "usd_sum": 0.001161
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-013",
      "category": "vocabulary",
      "capability": "component_role_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "moroccan",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "dc3e62daf8bdf21a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits harissa stirred into a lamb tagine at service.\n\nINPUT:\n{\n  \"description\": \"a spoonful of fiery harissa stirred into a lamb-and-apricot tagine at service\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "seasoning"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "sauce",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "sauce",
            "expected": "seasoning",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 398,
            "output_tokens": 4,
            "latency_ms": 494.4,
            "usd": 0.000334,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "sauce",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "sauce",
            "expected": "seasoning",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 398,
            "output_tokens": 4,
            "latency_ms": 997.3,
            "usd": 0.000334,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "sauce",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "sauce",
            "expected": "seasoning",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 398,
            "output_tokens": 4,
            "latency_ms": 1449.0,
            "usd": 0.000334,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 398.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 980.233,
        "usd_sum": 0.001002
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-014",
      "category": "vocabulary",
      "capability": "qualifier_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "chinese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "3980456a47e9b7c6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nSelect the OGS aromatic qualifier that best describes the state of Sichuan peppercorn as it is commonly used — whole, toasted, and ground over finished mapo tofu.\n\nINPUT:\n{\n  \"item\": \"Sichuan peppercorn (Zanthoxylum), whole, pan-toasted just before grinding over the dish\"\n}\n\nANSWER FORMAT: one_of:fresh,dried,cooked,candied,fermented,zest,overripe",
      "expected": {
        "value": "dried"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cooked",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "cooked",
            "expected": "dried",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 391,
            "output_tokens": 5,
            "latency_ms": 665.1,
            "usd": 0.000333,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "cooked",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "cooked",
            "expected": "dried",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 391,
            "output_tokens": 5,
            "latency_ms": 569.2,
            "usd": 0.000333,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "cooked",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "cooked",
            "expected": "dried",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 391,
            "output_tokens": 5,
            "latency_ms": 493.6,
            "usd": 0.000333,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 391.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 575.967,
        "usd_sum": 0.000999
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-015",
      "category": "vocabulary",
      "capability": "grade_scale",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "ad313a98974d64f7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nUsing the OGS pairing grade scale, return the letter grade for a pairing whose `overall_score` is 78.\n\nINPUT:\n{\n  \"overall_score\": 78\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "C+"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "C+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c+",
            "expected": "c+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 5,
            "latency_ms": 470.3,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "C+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c+",
            "expected": "c+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 5,
            "latency_ms": 551.7,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "C+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c+",
            "expected": "c+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 5,
            "latency_ms": 460.0,
            "usd": 0.00035,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 413.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 494.0,
        "usd_sum": 0.00105
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-016",
      "category": "vocabulary",
      "capability": "match_type_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "british",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "95e41ab82cba029a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nSelect the single OGS match type that best describes a beverage whose carbonation physically scrubs the palate between bites of a rich fried dish.\n\nINPUT:\n{\n  \"description\": \"Champagne cleanses the palate between bites of fish and chips\"\n}\n\nANSWER FORMAT: one_of:harmony,contrast,bridge,intensity_alignment,structural_balance,refresh",
      "expected": {
        "value": "refresh"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "refresh",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "refresh",
            "expected": "refresh",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 359,
            "output_tokens": 4,
            "latency_ms": 1195.5,
            "usd": 0.000303,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "refresh",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "refresh",
            "expected": "refresh",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 359,
            "output_tokens": 4,
            "latency_ms": 651.1,
            "usd": 0.000303,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "refresh",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "refresh",
            "expected": "refresh",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 359,
            "output_tokens": 4,
            "latency_ms": 533.5,
            "usd": 0.000303,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 359.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 793.367,
        "usd_sum": 0.000909
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-017",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "8070fb687fd2f108",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single dominant aromatic family of fresh cilantro leaves.\n\nINPUT:\n{\n  \"item\": \"fresh cilantro / coriander leaves\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "herbal"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 5,
            "latency_ms": 636.2,
            "usd": 0.000315,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 5,
            "latency_ms": 1037.7,
            "usd": 0.000315,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 5,
            "latency_ms": 584.4,
            "usd": 0.000315,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 369.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 752.767,
        "usd_sum": 0.000945
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-018",
      "category": "vocabulary",
      "capability": "component_role_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "levantine",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "9e97b59361dbdf0e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits a drizzle of pomegranate molasses dressed onto a fattoush salad.\n\nINPUT:\n{\n  \"description\": \"pomegranate molasses drizzled over a fattoush salad at service\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "acid_component"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "acid_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "acid_component",
            "expected": "acid_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 400,
            "output_tokens": 6,
            "latency_ms": 522.3,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "acid_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "acid_component",
            "expected": "acid_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 400,
            "output_tokens": 6,
            "latency_ms": 698.7,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "acid_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "acid_component",
            "expected": "acid_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 400,
            "output_tokens": 6,
            "latency_ms": 544.7,
            "usd": 0.000344,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 400.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 588.567,
        "usd_sum": 0.001032
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-019",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "filipino",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "fc9a763682f1225b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best describes bagoong alamang (Philippine fermented salt-cured shrimp paste).\n\nINPUT:\n{\n  \"description\": \"bagoong alamang: salt-cured tiny shrimp left to ferment for 3-6 months\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "fermented"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fermented",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fermented",
            "expected": "fermented",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 462,
            "output_tokens": 5,
            "latency_ms": 753.3,
            "usd": 0.00039,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "fermented",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fermented",
            "expected": "fermented",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 462,
            "output_tokens": 5,
            "latency_ms": 747.2,
            "usd": 0.00039,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "fermented",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fermented",
            "expected": "fermented",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 462,
            "output_tokens": 5,
            "latency_ms": 659.7,
            "usd": 0.00039,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 462.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 720.067,
        "usd_sum": 0.00117
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-020",
      "category": "vocabulary",
      "capability": "grade_scale",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "56ad4f9c58bc1eb7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nUsing the OGS pairing grade scale, return the letter grade for a pairing whose `overall_score` is 64.\n\nINPUT:\n{\n  \"overall_score\": 64\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "D"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "d",
            "expected": "d",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 4,
            "latency_ms": 748.7,
            "usd": 0.000346,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "d",
            "expected": "d",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 4,
            "latency_ms": 568.6,
            "usd": 0.000346,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "d",
            "expected": "d",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 4,
            "latency_ms": 637.6,
            "usd": 0.000346,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 413.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 651.633,
        "usd_sum": 0.001038
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-001",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "a75ad0ee2a385794",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nFrom the OGS explanation-code vocabulary, return the set of codes that best explain WHY the pairing below is excellent. Respond as a JSON array of code IDs. Pick 3 to 6 most relevant positive codes.\n\nINPUT:\n{\n  \"dish\": \"A5 wagyu with black truffle jus (very rich, high umami, earthy aromatics, full body)\",\n  \"beverage\": \"Barolo 2018 (Nebbiolo, high tannin, high acidity, full body, tar/forest-floor)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_FAT",
          "STRUCT_INTENSITY_MATCH",
          "AROMA_SHARED_EARTH",
          "STRUCT_BODY_ALIGN",
          "HARMONY_FLAVOR_CONVERGE"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\n  \"STRUCT_TANNIN_FAT\",\n  \"STRUCT_UMAMI_ENHANCE\",\n  \"AROMA_SHARED_EARTH\",\n  \"STRUCT_BODY_ALIGN\",\n  \"REFRESH_ACID\"\n]\n```",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_FAT",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1441,
            "output_tokens": 72,
            "latency_ms": 1409.2,
            "usd": 0.001441,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\n  \"STRUCT_TANNIN_FAT\",\n  \"STRUCT_UMAMI_ENHANCE\",\n  \"AROMA_SHARED_EARTH\",\n  \"STRUCT_BODY_ALIGN\",\n  \"REFRESH_ACID\"\n]\n```",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_FAT",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1441,
            "output_tokens": 72,
            "latency_ms": 1208.2,
            "usd": 0.001441,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"STRUCT_TANNIN_FAT\", \"STRUCT_UMAMI_ENHANCE\", \"AROMA_SHARED_EARTH\", \"STRUCT_BODY_ALIGN\", \"STRUCT_INTENSITY_MATCH\"]\n```",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.8,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1441,
            "output_tokens": 62,
            "latency_ms": 822.3,
            "usd": 0.001401,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.1155,
        "sem": 0.0667,
        "ci95_lo": 0.536,
        "ci95_hi": 0.7973,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1441.0,
        "output_tokens_mean": 68.667,
        "latency_ms_mean": 1146.567,
        "usd_sum": 0.004283
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-002",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "bdc5d8d74c85e9ab",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain the success of the pairing below. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"fresh oysters on the half shell with lemon\",\n  \"beverage\": \"Chablis Premier Cru (Chardonnay, high acid, mineral, lean body)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_ACIDITY_BALANCE",
          "REFRESH_ACID",
          "REFRESH_MINERAL",
          "CONTRAST_RICH_CRISP",
          "HARMONY_REGIONAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"STRUCT_ACIDITY_BALANCE\", \"AROMA_SHARED_MINERAL\", \"REFRESH_ACID\", \"HARMONY_REGIONAL\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_MINERAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1403,
            "output_tokens": 44,
            "latency_ms": 915.9,
            "usd": 0.001298,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"STRUCT_ACIDITY_BALANCE\", \"AROMA_SHARED_MINERAL\", \"REFRESH_ACID\", \"HARMONY_REGIONAL\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_MINERAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1403,
            "output_tokens": 44,
            "latency_ms": 777.0,
            "usd": 0.001298,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\n  \"STRUCT_ACIDITY_BALANCE\",\n  \"AROMA_SHARED_MINERAL\",\n  \"REFRESH_ACID\",\n  \"HARMONY_REGIONAL\"\n]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_MINERAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1403,
            "output_tokens": 55,
            "latency_ms": 793.3,
            "usd": 0.001342,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1403.0,
        "output_tokens_mean": 47.667,
        "latency_ms_mean": 828.733,
        "usd_sum": 0.003938
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-003",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "3440bcf3978a1c58",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (2 to 4 codes).\n\nINPUT:\n{\n  \"dish\": \"crispy buttermilk fried chicken\",\n  \"beverage\": \"non-vintage Brut Champagne\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "REFRESH_EFFERVESCE",
          "REFRESH_ACID",
          "CONTRAST_RICH_CRISP"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"REFRESH_EFFERVESCE\", \"CONTRAST_RICH_CRISP\", \"STRUCT_ACIDITY_BALANCE\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE"
            ],
            "precision": 0.6667,
            "recall": 0.6667
          },
          "metrics": {
            "input_tokens": 1388,
            "output_tokens": 40,
            "latency_ms": 861.2,
            "usd": 0.00127,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"REFRESH_EFFERVESCE\", \"CONTRAST_RICH_CRISP\", \"STRUCT_ACIDITY_BALANCE\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE"
            ],
            "precision": 0.6667,
            "recall": 0.6667
          },
          "metrics": {
            "input_tokens": 1388,
            "output_tokens": 40,
            "latency_ms": 1232.4,
            "usd": 0.00127,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"REFRESH_EFFERVESCE\", \"CONTRAST_RICH_CRISP\", \"STRUCT_ACIDITY_BALANCE\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE"
            ],
            "precision": 0.6667,
            "recall": 0.6667
          },
          "metrics": {
            "input_tokens": 1388,
            "output_tokens": 40,
            "latency_ms": 742.1,
            "usd": 0.00127,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1388.0,
        "output_tokens_mean": 40.0,
        "latency_ms_mean": 945.233,
        "usd_sum": 0.00381
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-004",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "b3249dafda966f70",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain the success of the pairing below. Respond as a JSON array (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"seared foie gras on brioche (very rich, fatty, slightly sweet-savoury)\",\n  \"beverage\": \"Sauternes (botrytised sweet white, high residual sugar, high acidity)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "STRUCT_ACIDITY_BALANCE",
          "CONTRAST_RICH_CRISP",
          "HARMONY_REGIONAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\n  \"STRUCT_SWEET_OFFSET\",\n  \"STRUCT_ACIDITY_BALANCE\",\n  \"HARMONY_FLAVOR_CONVERGE\",\n  \"AROMA_SHARED_FRUIT\"\n]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1408,
            "output_tokens": 62,
            "latency_ms": 924.4,
            "usd": 0.001374,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"STRUCT_SWEET_OFFSET\", \"STRUCT_ACIDITY_BALANCE\", \"HARMONY_FLAVOR_CONVERGE\", \"AROMA_SHARED_FRUIT\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1408,
            "output_tokens": 51,
            "latency_ms": 869.0,
            "usd": 0.00133,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"STRUCT_SWEET_OFFSET\", \"STRUCT_ACIDITY_BALANCE\", \"HARMONY_FLAVOR_CONVERGE\", \"AROMA_SHARED_FRUIT\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1408,
            "output_tokens": 51,
            "latency_ms": 860.3,
            "usd": 0.00133,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5,
      "score_summary": {
        "mean": 0.5,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.5,
        "ci95_hi": 0.5,
        "n": 3
      },
      "score_if_valid": 0.5,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1408.0,
        "output_tokens_mean": 54.667,
        "latency_ms_mean": 884.567,
        "usd_sum": 0.004034
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-005",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "0cb2f1946943a0e2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain why the pairing works. Respond as a JSON array (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"charcoal-grilled lamb rack with rosemary, black pepper crust, charred edges\",\n  \"beverage\": \"Northern Rhone Syrah (Cote-Rotie) — black pepper, violet, smoked meat, moderate-high tannin\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "AROMA_SHARED_SPICE",
          "AROMA_SHARED_WOOD",
          "AROMA_SHARED_HERBAL",
          "STRUCT_TANNIN_FAT",
          "STRUCT_TANNIN_PROTEIN"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_SPICE\", \"AROMA_SHARED_WOOD\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_INTENSITY_MATCH\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1418,
            "output_tokens": 53,
            "latency_ms": 1773.2,
            "usd": 0.001346,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"AROMA_SHARED_SPICE\", \"AROMA_SHARED_WOOD\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_INTENSITY_MATCH\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1418,
            "output_tokens": 53,
            "latency_ms": 1084.7,
            "usd": 0.001346,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"AROMA_SHARED_SPICE\", \"AROMA_SHARED_WOOD\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_INTENSITY_MATCH\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1418,
            "output_tokens": 53,
            "latency_ms": 820.9,
            "usd": 0.001346,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1418.0,
        "output_tokens_mean": 53.0,
        "latency_ms_mean": 1226.267,
        "usd_sum": 0.004038
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-006",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "9a46b0d141442ce3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Pizza Margherita (tomato, mozzarella, basil)\",\n  \"beverage\": \"Chianti Classico (Sangiovese, high acid, medium tannin)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_ACIDITY_BALANCE",
          "HARMONY_REGIONAL",
          "STRUCT_INTENSITY_MATCH",
          "AROMA_SHARED_HERBAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_HERBAL\", \"STRUCT_ACIDITY_BALANCE\", \"HARMONY_REGIONAL\", \"STRUCT_TANNIN_FAT\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1406,
            "output_tokens": 50,
            "latency_ms": 1087.5,
            "usd": 0.001325,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\n  \"AROMA_SHARED_HERBAL\",\n  \"STRUCT_ACIDITY_BALANCE\",\n  \"STRUCT_TANNIN_FAT\",\n  \"HARMONY_REGIONAL\"\n]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1406,
            "output_tokens": 61,
            "latency_ms": 1686.5,
            "usd": 0.001369,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"AROMA_SHARED_HERBAL\", \"STRUCT_ACIDITY_BALANCE\", \"HARMONY_REGIONAL\", \"STRUCT_TANNIN_FAT\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1406,
            "output_tokens": 50,
            "latency_ms": 793.7,
            "usd": 0.001325,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.75,
      "score_summary": {
        "mean": 0.75,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.75,
        "ci95_hi": 0.75,
        "n": 3
      },
      "score_if_valid": 0.75,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1406.0,
        "output_tokens_mean": 53.667,
        "latency_ms_mean": 1189.233,
        "usd_sum": 0.004019
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-007",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "chinese",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "497adc64f3303eb4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"mapo tofu with Sichuan peppercorn\",\n  \"beverage\": \"off-dry German Spätlese Riesling\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "REFRESH_ACID",
          "STRUCT_INTENSITY_MATCH",
          "CONTRAST_SWEET_SALTY"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\n  \"CONTRAST_SWEET_ACID\",\n  \"STRUCT_SWEET_OFFSET\",\n  \"REFRESH_ACID\",\n  \"AROMA_SHARED_SPICE\"\n]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "CONTRAST_SWEET_ACID",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1397,
            "output_tokens": 58,
            "latency_ms": 956.7,
            "usd": 0.00135,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"CONTRAST_SWEET_SALTY\", \"CONTRAST_SWEET_ACID\", \"STRUCT_SWEET_OFFSET\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_ACID",
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1397,
            "output_tokens": 46,
            "latency_ms": 893.3,
            "usd": 0.001302,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"CONTRAST_SWEET_SALTY\", \"CONTRAST_SWEET_ACID\", \"STRUCT_SWEET_OFFSET\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_ACID",
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1397,
            "output_tokens": 46,
            "latency_ms": 1021.9,
            "usd": 0.001302,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.1443,
        "sem": 0.0833,
        "ci95_lo": 0.5033,
        "ci95_hi": 0.83,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1397.0,
        "output_tokens_mean": 50.0,
        "latency_ms_mean": 957.3,
        "usd_sum": 0.003954
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-008",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "2120f96a0444a50e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"herb-crusted rack of lamb, medium-rare\",\n  \"beverage\": \"Northern Rhône Syrah\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_PROTEIN",
          "AROMA_SHARED_HERBAL",
          "STRUCT_BODY_ALIGN",
          "HARMONY_FLAVOR_CONVERGE"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_SPICE\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_BODY_ALIGN\", \"AROMA_SHARED_EARTH\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "AROMA_SHARED_SPICE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1389,
            "output_tokens": 53,
            "latency_ms": 1037.0,
            "usd": 0.001323,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\n  \"AROMA_SHARED_SPICE\",\n  \"AROMA_SHARED_HERBAL\",\n  \"STRUCT_TANNIN_PROTEIN\",\n  \"STRUCT_BODY_ALIGN\",\n  \"HARMONY_REGIONAL\"\n]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "AROMA_SHARED_SPICE",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1389,
            "output_tokens": 74,
            "latency_ms": 1092.1,
            "usd": 0.001407,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"AROMA_SHARED_SPICE\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_BODY_ALIGN\", \"HARMONY_REGIONAL\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1389,
            "output_tokens": 49,
            "latency_ms": 1050.3,
            "usd": 0.001307,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5556,
      "score_summary": {
        "mean": 0.5556,
        "stdev": 0.0962,
        "sem": 0.0556,
        "ci95_lo": 0.4467,
        "ci95_hi": 0.6645,
        "n": 3
      },
      "score_if_valid": 0.5556,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1389.0,
        "output_tokens_mean": 58.667,
        "latency_ms_mean": 1059.8,
        "usd_sum": 0.004037
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-009",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "thai",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "c077c498ceed5f73",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Thai green curry with chicken and Thai basil\",\n  \"beverage\": \"off-dry New Zealand Sauvignon Blanc\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "AROMA_SHARED_HERBAL",
          "REFRESH_ACID",
          "STRUCT_INTENSITY_MATCH"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\n  \"CONTRAST_SWEET_ACID\",\n  \"AROMA_SHARED_HERBAL\",\n  \"REFRESH_ACID\",\n  \"STRUCT_INTENSITY_MATCH\"\n]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "CONTRAST_SWEET_ACID",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1391,
            "output_tokens": 58,
            "latency_ms": 2661.7,
            "usd": 0.001345,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"CONTRAST_SWEET_ACID\", \"REFRESH_ACID\", \"AROMA_SHARED_HERBAL\", \"STRUCT_INTENSITY_MATCH\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "CONTRAST_SWEET_ACID",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1391,
            "output_tokens": 47,
            "latency_ms": 823.9,
            "usd": 0.001301,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"CONTRAST_SWEET_SALTY\", \"STRUCT_ACIDITY_BALANCE\", \"AROMA_SHARED_HERBAL\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1391,
            "output_tokens": 49,
            "latency_ms": 1014.5,
            "usd": 0.001309,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.1443,
        "sem": 0.0833,
        "ci95_lo": 0.5033,
        "ci95_hi": 0.83,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1391.0,
        "output_tokens_mean": 51.333,
        "latency_ms_mean": 1500.033,
        "usd_sum": 0.003955
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-010",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "956c6ce654c2e297",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"moules marinières\",\n  \"beverage\": \"Muscadet sur lie\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_ACIDITY_BALANCE",
          "REFRESH_ACID",
          "AROMA_SHARED_HERBAL",
          "HARMONY_REGIONAL",
          "REFRESH_MINERAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_HERBAL\", \"STRUCT_ACIDITY_BALANCE\", \"HARMONY_REGIONAL\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 1.0,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1381,
            "output_tokens": 45,
            "latency_ms": 1333.1,
            "usd": 0.001285,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"AROMA_SHARED_HERBAL\", \"STRUCT_ACIDITY_BALANCE\", \"REFRESH_ACID\", \"HARMONY_REGIONAL\"]\n```",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 1.0,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1381,
            "output_tokens": 45,
            "latency_ms": 814.5,
            "usd": 0.001285,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\n  \"HARMONY_REGIONAL\",\n  \"STRUCT_ACIDITY_BALANCE\",\n  \"AROMA_SHARED_HERBAL\",\n  \"REFRESH_ACID\"\n]\n```",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 1.0,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1381,
            "output_tokens": 56,
            "latency_ms": 1303.5,
            "usd": 0.001329,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.8889,
      "score_summary": {
        "mean": 0.8889,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8889,
        "ci95_hi": 0.8889,
        "n": 3
      },
      "score_if_valid": 0.8889,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1381.0,
        "output_tokens_mean": 48.667,
        "latency_ms_mean": 1150.367,
        "usd_sum": 0.003899
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-011",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "mexican",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "1dfeb782059a4814",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"tacos al pastor with pineapple and chili\",\n  \"beverage\": \"dry sparkling rosé (Cava rosado)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "REFRESH_EFFERVESCE",
          "AROMA_SHARED_FRUIT",
          "CONTRAST_RICH_CRISP"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_FRUIT\", \"CONTRAST_SWEET_SALTY\", \"REFRESH_EFFERVESCE\", \"STRUCT_ACIDITY_BALANCE\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_SWEET_SALTY",
              "REFRESH_EFFERVESCE",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1394,
            "output_tokens": 51,
            "latency_ms": 1795.7,
            "usd": 0.001319,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\n  \"AROMA_SHARED_FRUIT\",\n  \"CONTRAST_SWEET_ACID\",\n  \"REFRESH_EFFERVESCE\",\n  \"STRUCT_ACIDITY_BALANCE\"\n]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_SWEET_ACID",
              "REFRESH_EFFERVESCE",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1394,
            "output_tokens": 61,
            "latency_ms": 1045.7,
            "usd": 0.001359,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"AROMA_SHARED_FRUIT\", \"CONTRAST_SWEET_SALTY\", \"REFRESH_EFFERVESCE\", \"STRUCT_ACIDITY_BALANCE\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_SWEET_SALTY",
              "REFRESH_EFFERVESCE",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1394,
            "output_tokens": 51,
            "latency_ms": 1281.2,
            "usd": 0.001319,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5,
      "score_summary": {
        "mean": 0.5,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.5,
        "ci95_hi": 0.5,
        "n": 3
      },
      "score_if_valid": 0.5,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1394.0,
        "output_tokens_mean": 54.333,
        "latency_ms_mean": 1374.2,
        "usd_sum": 0.003997
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-012",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "28df36c8ae155697",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"duck confit, crispy skin\",\n  \"beverage\": \"mature Burgundy Pinot Noir (10 years)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_FAT",
          "AROMA_SHARED_EARTH",
          "HARMONY_REGIONAL",
          "HARMONY_COMPLEXITY_MATCH"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"STRUCT_TANNIN_FAT\", \"STRUCT_INTENSITY_MATCH\", \"AROMA_SHARED_EARTH\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_COMPLEXITY_MATCH",
              "HARMONY_REGIONAL",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1391,
            "output_tokens": 48,
            "latency_ms": 835.8,
            "usd": 0.001305,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"STRUCT_TANNIN_FAT\", \"STRUCT_INTENSITY_MATCH\", \"HARMONY_REGIONAL\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_COMPLEXITY_MATCH",
              "HARMONY_REGIONAL",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1391,
            "output_tokens": 44,
            "latency_ms": 978.3,
            "usd": 0.001289,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"STRUCT_TANNIN_FAT\", \"STRUCT_ACIDITY_BALANCE\", \"AROMA_SHARED_EARTH\", \"HARMONY_REGIONAL\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_COMPLEXITY_MATCH",
              "HARMONY_REGIONAL",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1391,
            "output_tokens": 49,
            "latency_ms": 1355.4,
            "usd": 0.001309,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5833,
      "score_summary": {
        "mean": 0.5833,
        "stdev": 0.1443,
        "sem": 0.0833,
        "ci95_lo": 0.42,
        "ci95_hi": 0.7467,
        "n": 3
      },
      "score_if_valid": 0.5833,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1391.0,
        "output_tokens_mean": 47.0,
        "latency_ms_mean": 1056.5,
        "usd_sum": 0.003903
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-013",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "spanish",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "18a20e7e30844c41",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"jamón ibérico de bellota, sliced thin\",\n  \"beverage\": \"chilled Fino sherry (Manzanilla)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_UMAMI_ENHANCE",
          "HARMONY_REGIONAL",
          "AROMA_SHARED_SAVORY",
          "REFRESH_MINERAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_EARTH\", \"STRUCT_INTENSITY_MATCH\", \"REFRESH_ACID\", \"HARMONY_REGIONAL\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_SAVORY",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1395,
            "output_tokens": 43,
            "latency_ms": 943.8,
            "usd": 0.001288,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\n  \"HARMONY_REGIONAL\",\n  \"STRUCT_INTENSITY_MATCH\",\n  \"STRUCT_BODY_ALIGN\",\n  \"REFRESH_ACID\",\n  \"AROMA_SHARED_SPICE\"\n]\n```",
          "compliant": true,
          "score": 0.2222,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_SAVORY",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.2,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1395,
            "output_tokens": 67,
            "latency_ms": 976.2,
            "usd": 0.001384,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"AROMA_SHARED_HERBAL\", \"STRUCT_ACIDITY_BALANCE\", \"REFRESH_ACID\", \"HARMONY_REGIONAL\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_SAVORY",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1395,
            "output_tokens": 45,
            "latency_ms": 741.8,
            "usd": 0.001296,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.2407,
      "score_summary": {
        "mean": 0.2407,
        "stdev": 0.0161,
        "sem": 0.0093,
        "ci95_lo": 0.2226,
        "ci95_hi": 0.2589,
        "n": 3
      },
      "score_if_valid": 0.2407,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1395.0,
        "output_tokens_mean": 51.667,
        "latency_ms_mean": 887.267,
        "usd_sum": 0.003968
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-014",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "korean",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "f9be4ef2853d67b3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"bulgogi: soy-pear-garlic marinated grilled beef\",\n  \"beverage\": \"medium-bodied unoaked Merlot\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_PROTEIN",
          "STRUCT_SWEET_OFFSET",
          "AROMA_SHARED_FRUIT",
          "STRUCT_BODY_ALIGN"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_SPICE\", \"STRUCT_INTENSITY_MATCH\", \"STRUCT_BODY_ALIGN\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1398,
            "output_tokens": 47,
            "latency_ms": 1036.6,
            "usd": 0.001306,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"STRUCT_INTENSITY_MATCH\", \"STRUCT_BODY_ALIGN\", \"AROMA_SHARED_SPICE\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1398,
            "output_tokens": 47,
            "latency_ms": 1478.4,
            "usd": 0.001306,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"AROMA_SHARED_SPICE\", \"STRUCT_INTENSITY_MATCH\", \"STRUCT_TANNIN_PROTEIN\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1398,
            "output_tokens": 49,
            "latency_ms": 993.5,
            "usd": 0.001314,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.25,
      "score_summary": {
        "mean": 0.25,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.25,
        "ci95_hi": 0.25,
        "n": 3
      },
      "score_if_valid": 0.25,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1398.0,
        "output_tokens_mean": 47.667,
        "latency_ms_mean": 1169.5,
        "usd_sum": 0.003926
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-015",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "easy",
      "cuisine_region": "british",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "5474f731e5d9751c",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Stilton blue cheese\",\n  \"beverage\": \"Vintage Port\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "CONTRAST_SWEET_SALTY",
          "STRUCT_INTENSITY_MATCH",
          "HARMONY_COMPLEXITY_MATCH",
          "STRUCT_SWEET_OFFSET"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"HARMONY_FLAVOR_CONVERGE\", \"STRUCT_TANNIN_FAT\", \"STRUCT_SWEET_OFFSET\", \"AROMA_SHARED_FRUIT\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_COMPLEXITY_MATCH",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1379,
            "output_tokens": 52,
            "latency_ms": 1138.0,
            "usd": 0.001311,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"HARMONY_FLAVOR_CONVERGE\", \"STRUCT_SWEET_OFFSET\", \"STRUCT_UMAMI_ENHANCE\", \"AROMA_SHARED_FRUIT\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_COMPLEXITY_MATCH",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1379,
            "output_tokens": 51,
            "latency_ms": 1924.2,
            "usd": 0.001307,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"HARMONY_FLAVOR_CONVERGE\", \"STRUCT_SWEET_OFFSET\", \"STRUCT_TANNIN_PROTEIN\", \"AROMA_SHARED_FRUIT\"]\n```",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_COMPLEXITY_MATCH",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1379,
            "output_tokens": 52,
            "latency_ms": 828.3,
            "usd": 0.001311,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.25,
      "score_summary": {
        "mean": 0.25,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.25,
        "ci95_hi": 0.25,
        "n": 3
      },
      "score_if_valid": 0.25,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1379.0,
        "output_tokens_mean": 51.667,
        "latency_ms_mean": 1296.833,
        "usd_sum": 0.003929
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-016",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "fc7f227aebed6756",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"shrimp tempura, light fry\",\n  \"beverage\": \"chilled junmai ginjo sake\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_UMAMI_ENHANCE",
          "REFRESH_MINERAL",
          "AROMA_SHARED_FLORAL",
          "STRUCT_INTENSITY_MATCH",
          "HARMONY_REGIONAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"REFRESH_ACID\", \"STRUCT_INTENSITY_MATCH\", \"HARMONY_REGIONAL\", \"AROMA_SHARED_FLORAL\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FLORAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FLORAL",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1387,
            "output_tokens": 44,
            "latency_ms": 812.4,
            "usd": 0.001286,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\n  \"CONTRAST_RICH_CRISP\",\n  \"STRUCT_INTENSITY_MATCH\",\n  \"REFRESH_EFFERVESCE\",\n  \"AROMA_SHARED_HERBAL\"\n]\n```",
          "compliant": true,
          "score": 0.2222,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FLORAL",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.25,
            "recall": 0.2
          },
          "metrics": {
            "input_tokens": 1387,
            "output_tokens": 62,
            "latency_ms": 805.9,
            "usd": 0.001358,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\n  \"CONTRAST_RICH_CRISP\",\n  \"STRUCT_INTENSITY_MATCH\",\n  \"REFRESH_ACID\",\n  \"HARMONY_REGIONAL\"\n]\n```",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FLORAL",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.5,
            "recall": 0.4
          },
          "metrics": {
            "input_tokens": 1387,
            "output_tokens": 54,
            "latency_ms": 967.2,
            "usd": 0.001326,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.4444,
      "score_summary": {
        "mean": 0.4444,
        "stdev": 0.2223,
        "sem": 0.1283,
        "ci95_lo": 0.1929,
        "ci95_hi": 0.6959,
        "n": 3
      },
      "score_if_valid": 0.4444,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1387.0,
        "output_tokens_mean": 53.333,
        "latency_ms_mean": 861.833,
        "usd_sum": 0.00397
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-017",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "greek",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "024198497e327bb1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"horiatiki salad with feta, tomato, olive, oregano\",\n  \"beverage\": \"Santorini Assyrtiko\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "REFRESH_ACID",
          "REFRESH_MINERAL",
          "HARMONY_REGIONAL",
          "AROMA_SHARED_HERBAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"HARMONY_REGIONAL\", \"AROMA_SHARED_HERBAL\", \"STRUCT_ACIDITY_BALANCE\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1395,
            "output_tokens": 45,
            "latency_ms": 1030.2,
            "usd": 0.001296,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"HARMONY_REGIONAL\", \"AROMA_SHARED_HERBAL\", \"STRUCT_ACIDITY_BALANCE\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1395,
            "output_tokens": 45,
            "latency_ms": 927.4,
            "usd": 0.001296,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\n  \"HARMONY_REGIONAL\",\n  \"AROMA_SHARED_HERBAL\",\n  \"STRUCT_ACIDITY_BALANCE\",\n  \"REFRESH_ACID\"\n]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1395,
            "output_tokens": 56,
            "latency_ms": 814.7,
            "usd": 0.00134,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.75,
      "score_summary": {
        "mean": 0.75,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.75,
        "ci95_hi": 0.75,
        "n": 3
      },
      "score_if_valid": 0.75,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1395.0,
        "output_tokens_mean": 48.667,
        "latency_ms_mean": 924.1,
        "usd_sum": 0.003932
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-018",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "chinese",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "283bbb9e703e6e30",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Peking duck with pancake, scallion, hoisin\",\n  \"beverage\": \"10-year-old off-dry Mosel Riesling\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "AROMA_SHARED_FRUIT",
          "REFRESH_ACID",
          "HARMONY_COMPLEXITY_MATCH"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"AROMA_SHARED_FRUIT\", \"CONTRAST_SWEET_SALTY\", \"STRUCT_ACIDITY_BALANCE\", \"STRUCT_SWEET_OFFSET\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_SWEET_SALTY",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_COMPLEXITY_MATCH",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1401,
            "output_tokens": 51,
            "latency_ms": 737.3,
            "usd": 0.001325,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\n  \"AROMA_SHARED_FRUIT\",\n  \"CONTRAST_SWEET_SALTY\",\n  \"STRUCT_ACIDITY_BALANCE\",\n  \"STRUCT_SWEET_OFFSET\"\n]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_SWEET_SALTY",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_COMPLEXITY_MATCH",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1401,
            "output_tokens": 62,
            "latency_ms": 1304.9,
            "usd": 0.001369,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\n  \"AROMA_SHARED_FRUIT\",\n  \"CONTRAST_SWEET_SALTY\",\n  \"STRUCT_SWEETNESS_OFFSET\",\n  \"REFRESH_ACID\"\n]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_SWEETNESS_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_COMPLEXITY_MATCH",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1401,
            "output_tokens": 59,
            "latency_ms": 752.2,
            "usd": 0.001357,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.5,
      "score_summary": {
        "mean": 0.5,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.5,
        "ci95_hi": 0.5,
        "n": 3
      },
      "score_if_valid": 0.5,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1401.0,
        "output_tokens_mean": 57.333,
        "latency_ms_mean": 931.467,
        "usd_sum": 0.004051
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-019",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "9ea6cd79384f9ee2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Sunday ragù napoletano (long-cooked beef and pork)\",\n  \"beverage\": \"Campanian Aglianico, 5 years\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_PROTEIN",
          "HARMONY_REGIONAL",
          "STRUCT_BODY_ALIGN",
          "AROMA_SHARED_WOOD"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"HARMONY_REGIONAL\", \"STRUCT_TANNIN_FAT\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_INTENSITY_MATCH\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_WOOD",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1399,
            "output_tokens": 49,
            "latency_ms": 2167.1,
            "usd": 0.001315,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"HARMONY_REGIONAL\", \"STRUCT_INTENSITY_MATCH\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_BODY_ALIGN\", \"AROMA_SHARED_EARTH\"]\n```",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_WOOD",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1399,
            "output_tokens": 58,
            "latency_ms": 810.1,
            "usd": 0.001351,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```\n[\"HARMONY_REGIONAL\", \"STRUCT_TANNIN_PROTEIN\", \"STRUCT_BODY_ALIGN\", \"AROMA_SHARED_EARTH\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_WOOD",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1399,
            "output_tokens": 47,
            "latency_ms": 1069.0,
            "usd": 0.001307,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6389,
      "score_summary": {
        "mean": 0.6389,
        "stdev": 0.1273,
        "sem": 0.0735,
        "ci95_lo": 0.4948,
        "ci95_hi": 0.783,
        "n": 3
      },
      "score_if_valid": 0.6389,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1399.0,
        "output_tokens_mean": 51.333,
        "latency_ms_mean": 1348.733,
        "usd_sum": 0.003973
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-020",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "43005f44c870684b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"foie gras torchon with brioche\",\n  \"beverage\": \"Sauternes\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "CONTRAST_SWEET_SALTY",
          "HARMONY_REGIONAL",
          "STRUCT_BODY_ALIGN"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "```json\n[\"STRUCT_SWEET_OFFSET\", \"STRUCT_TANNIN_FAT\", \"HARMONY_REGIONAL\", \"CONTRAST_SWEET_SALTY\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_REGIONAL",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1385,
            "output_tokens": 48,
            "latency_ms": 809.9,
            "usd": 0.0013,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 1,
          "raw_response": "```json\n[\"STRUCT_SWEET_OFFSET\", \"STRUCT_BODY_ALIGN\", \"HARMONY_REGIONAL\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1385,
            "output_tokens": 42,
            "latency_ms": 1807.7,
            "usd": 0.001276,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        },
        {
          "sample": 2,
          "raw_response": "```json\n[\"STRUCT_SWEET_OFFSET\", \"STRUCT_TANNIN_FAT\", \"HARMONY_REGIONAL\", \"REFRESH_ACID\"]\n```",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.5,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1385,
            "output_tokens": 44,
            "latency_ms": 894.4,
            "usd": 0.001284,
            "model_version": "claude-haiku-4-5-20251001",
            "finish_reason": "end_turn"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.1443,
        "sem": 0.0833,
        "ci95_lo": 0.5033,
        "ci95_hi": 0.83,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1385.0,
        "output_tokens_mean": 44.667,
        "latency_ms_mean": 1170.667,
        "usd_sum": 0.00386
      }
    }
  ],
  "category_scores": {
    "composition": {
      "mean_score": 0.8427,
      "ci95_lo": 0.6828,
      "ci95_hi": 1.0025,
      "stdev": 0.3647,
      "task_count": 20,
      "weight": 0.12,
      "format_rate": 1.0,
      "score_if_valid": 0.8427
    },
    "dietary": {
      "mean_score": 0.7184,
      "ci95_lo": 0.4922,
      "ci95_hi": 0.9445,
      "stdev": 0.3997,
      "task_count": 12,
      "weight": 0.06,
      "format_rate": 1.0,
      "score_if_valid": 0.7184
    },
    "explanation": {
      "mean_score": 0.5759,
      "ci95_lo": 0.4994,
      "ci95_hi": 0.6525,
      "stdev": 0.1747,
      "task_count": 20,
      "weight": 0.1,
      "format_rate": 1.0,
      "score_if_valid": 0.5759
    },
    "menu_engineering": {
      "mean_score": 1.0,
      "ci95_lo": 1.0,
      "ci95_hi": 1.0,
      "stdev": 0.0,
      "task_count": 8,
      "weight": 0.03,
      "format_rate": 1.0,
      "score_if_valid": 1.0
    },
    "nutrition": {
      "mean_score": 0.2506,
      "ci95_lo": -0.0031,
      "ci95_hi": 0.5043,
      "stdev": 0.4093,
      "task_count": 10,
      "weight": 0.03,
      "format_rate": 0.9667,
      "score_if_valid": 0.2663
    },
    "pairing": {
      "mean_score": 1.0,
      "ci95_lo": 1.0,
      "ci95_hi": 1.0,
      "stdev": 0.0,
      "task_count": 22,
      "weight": 0.15,
      "format_rate": 1.0,
      "score_if_valid": 1.0
    },
    "regional": {
      "mean_score": 0.9583,
      "ci95_lo": 0.8767,
      "ci95_hi": 1.04,
      "stdev": 0.2041,
      "task_count": 24,
      "weight": 0.05,
      "format_rate": 1.0,
      "score_if_valid": 0.9583
    },
    "risk": {
      "mean_score": 0.8167,
      "ci95_lo": 0.6493,
      "ci95_hi": 0.9841,
      "stdev": 0.382,
      "task_count": 20,
      "weight": 0.08,
      "format_rate": 1.0,
      "score_if_valid": 0.8167
    },
    "safety": {
      "mean_score": 0.9762,
      "ci95_lo": 0.9295,
      "ci95_hi": 1.0229,
      "stdev": 0.0825,
      "task_count": 12,
      "weight": 0.06,
      "format_rate": 1.0,
      "score_if_valid": 0.9762
    },
    "sensory": {
      "mean_score": 0.5837,
      "ci95_lo": 0.476,
      "ci95_hi": 0.6914,
      "stdev": 0.2518,
      "task_count": 21,
      "weight": 0.12,
      "format_rate": 1.0,
      "score_if_valid": 0.5837
    },
    "structured_output": {
      "mean_score": 0.7583,
      "ci95_lo": 0.5787,
      "ci95_hi": 0.938,
      "stdev": 0.4099,
      "task_count": 20,
      "weight": 0.12,
      "format_rate": 1.0,
      "score_if_valid": 0.7583
    },
    "vocabulary": {
      "mean_score": 0.8116,
      "ci95_lo": 0.6534,
      "ci95_hi": 0.9698,
      "stdev": 0.387,
      "task_count": 23,
      "weight": 0.08,
      "format_rate": 1.0,
      "score_if_valid": 0.8116
    }
  },
  "aggregate": {
    "score": 78.71,
    "ci95_lo": 74.63,
    "ci95_hi": 82.79,
    "scale": "0-100",
    "partial": false,
    "missing_categories": []
  },
  "metrics": {
    "latency_ms_mean": 1321.811,
    "latency_ms_median": 941.233,
    "input_tokens_mean": 876.0,
    "output_tokens_mean": 82.0,
    "usd_total": 0.657499,
    "tasks_with_usd": 213
  },
  "weights": {
    "vocabulary": 0.08,
    "sensory": 0.12,
    "composition": 0.12,
    "pairing": 0.15,
    "explanation": 0.1,
    "risk": 0.08,
    "structured_output": 0.12,
    "regional": 0.05,
    "dietary": 0.06,
    "safety": 0.06,
    "nutrition": 0.03,
    "menu_engineering": 0.03
  },
  "judge": {
    "judge_model": "gpt-5.4",
    "judge_prompt_hash": "756db6a487714bca",
    "judge_validated": false,
    "excluded_task_ids": [
      "ogs:core:benchmark-task:menu-009"
    ],
    "note": "Rubric-judge tasks are scored and retained in `tasks`, but excluded from `aggregate` until the judge clears the human-validation protocol (r > 0.8 against a 50-task sample)."
  }
}