{
  "benchmark_version": "0.2.0",
  "ogs_version": "0.1.0",
  "run_id": "49d08146-ec4c-4b46-bd54-1df08af5ad2b",
  "run_at": "2026-04-18T05:59:32.428541+00:00",
  "model": {
    "id": "gpt-5.4",
    "adapter": "openai",
    "options": {}
  },
  "config": {
    "samples": 3,
    "closed_book": false,
    "seed": null,
    "skip_disputed": true
  },
  "environment": {
    "python": "3.14.3",
    "platform": "macOS-26.4.1-arm64-arm-64bit-Mach-O",
    "sdk_versions": {
      "openai": "2.32.0",
      "jsonschema": "4.26.0"
    }
  },
  "reproducibility": {
    "task_set_sha": "345fbe2cedeac504",
    "task_count": 213,
    "prompt_hashes": {
      "ogs:core:benchmark-task:comp-001": "5dc05e883ee75cf7",
      "ogs:core:benchmark-task:comp-002": "d5fef8891eee5391",
      "ogs:core:benchmark-task:comp-003": "7b7db2c313e1d55b",
      "ogs:core:benchmark-task:comp-004": "ca8109a73af762fa",
      "ogs:core:benchmark-task:comp-005": "2885728399552fb5",
      "ogs:core:benchmark-task:comp-006": "bb37bf992b2f6c39",
      "ogs:core:benchmark-task:comp-007": "59ae41132f38bbfd",
      "ogs:core:benchmark-task:comp-008": "5121ce4bf46aa4d5",
      "ogs:core:benchmark-task:comp-009": "2a8a86bb58372ed7",
      "ogs:core:benchmark-task:comp-010": "d420afcaea603a0e",
      "ogs:core:benchmark-task:comp-011": "1d18b06eac9f3ddc",
      "ogs:core:benchmark-task:comp-012": "fa91f74e86a4cffb",
      "ogs:core:benchmark-task:comp-013": "f8257062be99b5d3",
      "ogs:core:benchmark-task:comp-014": "114b0796c10cf141",
      "ogs:core:benchmark-task:comp-015": "07976f0ee97244e5",
      "ogs:core:benchmark-task:comp-016": "609a0643399a8df0",
      "ogs:core:benchmark-task:comp-017": "09a6e47eda264ded",
      "ogs:core:benchmark-task:comp-018": "32d4a921f4fbdd98",
      "ogs:core:benchmark-task:comp-019": "eaa27377a264e2e2",
      "ogs:core:benchmark-task:comp-020": "535bf59391539a15",
      "ogs:core:benchmark-task:diet-001": "18b5c9c7c958a521",
      "ogs:core:benchmark-task:diet-002": "04840e46f80ccb5b",
      "ogs:core:benchmark-task:diet-003": "013d6295fee472b5",
      "ogs:core:benchmark-task:diet-004": "85fc94958c4357ef",
      "ogs:core:benchmark-task:diet-005": "f40f4c25a2ee4b71",
      "ogs:core:benchmark-task:diet-006": "7f45e388301a68a5",
      "ogs:core:benchmark-task:diet-007": "eb092a2c162960f6",
      "ogs:core:benchmark-task:diet-008": "480be99b3f35a51d",
      "ogs:core:benchmark-task:diet-009": "936752891554dba3",
      "ogs:core:benchmark-task:diet-010": "35c29846f76046d7",
      "ogs:core:benchmark-task:diet-011": "bdd7970997c11172",
      "ogs:core:benchmark-task:diet-012": "42c897d741064981",
      "ogs:core:benchmark-task:menu-001": "138c09ed611a2d43",
      "ogs:core:benchmark-task:menu-002": "b5df321fd829de76",
      "ogs:core:benchmark-task:menu-003": "9795b72bdda437f1",
      "ogs:core:benchmark-task:menu-004": "fc0da6dcbbc842fb",
      "ogs:core:benchmark-task:menu-005": "f33710f88f89121e",
      "ogs:core:benchmark-task:menu-006": "7bb019ca9ca4b1be",
      "ogs:core:benchmark-task:menu-007": "68b7fb2ec1e432f4",
      "ogs:core:benchmark-task:menu-008": "dc5bf7ed3a0583c2",
      "ogs:core:benchmark-task:menu-009": "8883cacb0aca5024",
      "ogs:core:benchmark-task:ml-ar-001": "94e19d81d80368a1",
      "ogs:core:benchmark-task:ml-ar-002": "6b39ec376be689ad",
      "ogs:core:benchmark-task:ml-es-001": "5fe5c7f1f85e3675",
      "ogs:core:benchmark-task:ml-es-002": "9f03b90711bec40f",
      "ogs:core:benchmark-task:ml-fr-001": "8d252ae3fb170e15",
      "ogs:core:benchmark-task:ml-fr-002": "96a07aeab44536ab",
      "ogs:core:benchmark-task:ml-it-001": "c3aae88ebfcbe08b",
      "ogs:core:benchmark-task:ml-it-002": "30919f613b1fbcd6",
      "ogs:core:benchmark-task:ml-ja-001": "ff1f43723314705c",
      "ogs:core:benchmark-task:ml-ja-002": "f68d4882814e0ecf",
      "ogs:core:benchmark-task:nut-001": "94fadc6f5c8c5968",
      "ogs:core:benchmark-task:nut-002": "4b141daaadd9af57",
      "ogs:core:benchmark-task:nut-003": "518ee70ef096f9ec",
      "ogs:core:benchmark-task:nut-004": "42947503c82b308d",
      "ogs:core:benchmark-task:nut-005": "375ca5d44d346632",
      "ogs:core:benchmark-task:nut-006": "1ba6175ed99ad486",
      "ogs:core:benchmark-task:nut-007": "b9e5625ed0828c94",
      "ogs:core:benchmark-task:nut-008": "95ef839d58e96bbc",
      "ogs:core:benchmark-task:nut-009": "641416c8b33c6dbc",
      "ogs:core:benchmark-task:nut-010": "fd7edc20921469ff",
      "ogs:core:benchmark-task:pair-001": "3fa6a67d5bfe7f1a",
      "ogs:core:benchmark-task:pair-002": "4b85899cd6206acb",
      "ogs:core:benchmark-task:pair-003": "41ed3ee840823502",
      "ogs:core:benchmark-task:pair-004": "82a4b41d0e3d5668",
      "ogs:core:benchmark-task:pair-005": "441705237fe65e54",
      "ogs:core:benchmark-task:pair-006": "b195ee1dfe7968ab",
      "ogs:core:benchmark-task:pair-007": "81bf14f26f4fa761",
      "ogs:core:benchmark-task:pair-008": "51ad14586a063779",
      "ogs:core:benchmark-task:pair-009": "d7ade64ce48bfbe4",
      "ogs:core:benchmark-task:pair-010": "637717fdf3577204",
      "ogs:core:benchmark-task:pair-011": "2b655f8b1ea64df8",
      "ogs:core:benchmark-task:pair-012": "3fe344c62017694c",
      "ogs:core:benchmark-task:pair-013": "4d30c6e439c9c85f",
      "ogs:core:benchmark-task:pair-014": "848f37f91777acfb",
      "ogs:core:benchmark-task:pair-015": "f5ecf9dcd59f0d0a",
      "ogs:core:benchmark-task:pair-016": "3dd7c0170877ac9b",
      "ogs:core:benchmark-task:pair-017": "8f60336584f33baf",
      "ogs:core:benchmark-task:pair-018": "9209873bc05779b7",
      "ogs:core:benchmark-task:pair-019": "da735293a9d3102b",
      "ogs:core:benchmark-task:pair-020": "bac251ebb1c064ef",
      "ogs:core:benchmark-task:region-001": "1df867dd3a55af5b",
      "ogs:core:benchmark-task:region-002": "24a95f8a1c0a0a45",
      "ogs:core:benchmark-task:region-003": "15d36148f8b853e6",
      "ogs:core:benchmark-task:region-004": "e332af036a2bd591",
      "ogs:core:benchmark-task:region-005": "4aa2c63ffdeeacfb",
      "ogs:core:benchmark-task:region-006": "979e7579b34d9797",
      "ogs:core:benchmark-task:region-007": "23a69016b1082553",
      "ogs:core:benchmark-task:region-008": "0a12eb057eb17654",
      "ogs:core:benchmark-task:region-009": "767d178c3cce3333",
      "ogs:core:benchmark-task:region-010": "1001169bc92aaacd",
      "ogs:core:benchmark-task:region-011": "571e5074e66e6a11",
      "ogs:core:benchmark-task:region-012": "d9792be4d5ca7335",
      "ogs:core:benchmark-task:region-013": "c831a091be5025f9",
      "ogs:core:benchmark-task:region-014": "7d761cbfb86342af",
      "ogs:core:benchmark-task:region-015": "2f35d24bb9d58118",
      "ogs:core:benchmark-task:region-016": "efb489c56faf463a",
      "ogs:core:benchmark-task:region-017": "53530e4634162afc",
      "ogs:core:benchmark-task:region-018": "28a636bdc82633de",
      "ogs:core:benchmark-task:region-019": "2c5869ce479f65cc",
      "ogs:core:benchmark-task:region-020": "aa3761bdbf2d490e",
      "ogs:core:benchmark-task:risk-001": "08845fcf7f8beb09",
      "ogs:core:benchmark-task:risk-002": "7a188b424515258f",
      "ogs:core:benchmark-task:risk-003": "a60fa6dd5c1ff452",
      "ogs:core:benchmark-task:risk-004": "16286de7fea14ae1",
      "ogs:core:benchmark-task:risk-005": "cd44f8ac361108e8",
      "ogs:core:benchmark-task:risk-006": "28cb271326a709c7",
      "ogs:core:benchmark-task:risk-007": "93c46eac23eca249",
      "ogs:core:benchmark-task:risk-008": "2409e606629eebf9",
      "ogs:core:benchmark-task:risk-009": "a00498fd7fce7cb7",
      "ogs:core:benchmark-task:risk-010": "b28af5fc40407ce0",
      "ogs:core:benchmark-task:risk-011": "fba6f85afa81595f",
      "ogs:core:benchmark-task:risk-012": "2f3a4360a39afdc9",
      "ogs:core:benchmark-task:risk-013": "390e90cae060ada8",
      "ogs:core:benchmark-task:risk-014": "d83b8b2ecabe9907",
      "ogs:core:benchmark-task:risk-015": "1e83097d483e0639",
      "ogs:core:benchmark-task:risk-016": "b2e02272ca87b2d4",
      "ogs:core:benchmark-task:risk-017": "12372c8de13add55",
      "ogs:core:benchmark-task:risk-018": "7a97c002792dfad9",
      "ogs:core:benchmark-task:risk-019": "0e3d6c3f9d86af0b",
      "ogs:core:benchmark-task:risk-020": "2dd649d25c606ee7",
      "ogs:core:benchmark-task:safe-001": "8e37b185cfaed2aa",
      "ogs:core:benchmark-task:safe-002": "d38d6470a18f7e97",
      "ogs:core:benchmark-task:safe-003": "97f1d8e6fc44b9e2",
      "ogs:core:benchmark-task:safe-004": "9e5770b6c7036769",
      "ogs:core:benchmark-task:safe-005": "cb2ff73059921b59",
      "ogs:core:benchmark-task:safe-006": "b3c9f3a848d65390",
      "ogs:core:benchmark-task:safe-007": "f120f4a38737e981",
      "ogs:core:benchmark-task:safe-008": "9b8f975bd042c7a0",
      "ogs:core:benchmark-task:safe-009": "9f4b49c1c20e4376",
      "ogs:core:benchmark-task:safe-010": "eea6814938f18565",
      "ogs:core:benchmark-task:safe-011": "edc2981cec11c39e",
      "ogs:core:benchmark-task:safe-012": "2ba4113ffe26b2fc",
      "ogs:core:benchmark-task:sense-001": "913118356da8150b",
      "ogs:core:benchmark-task:sense-002": "2690028d89b60292",
      "ogs:core:benchmark-task:sense-003": "b09502eb4fd5d61f",
      "ogs:core:benchmark-task:sense-004": "94f009713056d52f",
      "ogs:core:benchmark-task:sense-005": "c950ec9851ad52bf",
      "ogs:core:benchmark-task:sense-006": "d68ef8562addd55d",
      "ogs:core:benchmark-task:sense-007": "658c14cce434cc0a",
      "ogs:core:benchmark-task:sense-008": "873c6c2f9608e43f",
      "ogs:core:benchmark-task:sense-009": "523e04f39252cb4b",
      "ogs:core:benchmark-task:sense-010": "1813cae53ba17a84",
      "ogs:core:benchmark-task:sense-011": "1886b2e009716d1d",
      "ogs:core:benchmark-task:sense-012": "8d40a83f1b501685",
      "ogs:core:benchmark-task:sense-013": "691f59088f3e98f3",
      "ogs:core:benchmark-task:sense-014": "c4a83a79265c0e9f",
      "ogs:core:benchmark-task:sense-015": "66d64ca2254c09ee",
      "ogs:core:benchmark-task:sense-016": "70318c6fd3b033d2",
      "ogs:core:benchmark-task:sense-017": "c3cedde89152ce76",
      "ogs:core:benchmark-task:sense-018": "7d31de11173a7fd2",
      "ogs:core:benchmark-task:sense-019": "2e7df899a0148ebd",
      "ogs:core:benchmark-task:sense-020": "ca5b9ab85fe71acb",
      "ogs:core:benchmark-task:struct-001": "4523d68b7da194f7",
      "ogs:core:benchmark-task:struct-002": "367b58556573440d",
      "ogs:core:benchmark-task:struct-003": "2740dc70d150e12a",
      "ogs:core:benchmark-task:struct-004": "9bd6860a8c5c69ec",
      "ogs:core:benchmark-task:struct-005": "9caf9bc474b2464d",
      "ogs:core:benchmark-task:struct-006": "bb4d0fcee2d1e751",
      "ogs:core:benchmark-task:struct-007": "0bc2e49ee6c1093e",
      "ogs:core:benchmark-task:struct-008": "791ac1beebd95233",
      "ogs:core:benchmark-task:struct-009": "a6a51cef64e43626",
      "ogs:core:benchmark-task:struct-010": "6d4d6d3cc23f3283",
      "ogs:core:benchmark-task:struct-011": "6615664480900680",
      "ogs:core:benchmark-task:struct-012": "bd330891a3e68b59",
      "ogs:core:benchmark-task:struct-013": "872d23f39307a374",
      "ogs:core:benchmark-task:struct-014": "3fe3fe74f8d80260",
      "ogs:core:benchmark-task:struct-015": "f0c01fa20f35eb51",
      "ogs:core:benchmark-task:struct-016": "284b25ae8f0dc3ef",
      "ogs:core:benchmark-task:struct-017": "534d8590759d6dc3",
      "ogs:core:benchmark-task:struct-018": "1470a28b8e9ef028",
      "ogs:core:benchmark-task:struct-019": "aa2726470aa7cfb8",
      "ogs:core:benchmark-task:struct-020": "5b115600e8dad71f",
      "ogs:core:benchmark-task:vocab-001": "e52270d817a9dc7f",
      "ogs:core:benchmark-task:vocab-002": "e2096242953ff882",
      "ogs:core:benchmark-task:vocab-003": "fa5fa32bfe275dfb",
      "ogs:core:benchmark-task:vocab-004": "e0631034b8b8ece4",
      "ogs:core:benchmark-task:vocab-005": "bbbd81bb3c8b8ea4",
      "ogs:core:benchmark-task:vocab-006": "6c6ae6f82f354be3",
      "ogs:core:benchmark-task:vocab-007": "0adac8c06022acf5",
      "ogs:core:benchmark-task:vocab-008": "e58319dc1276cdb2",
      "ogs:core:benchmark-task:vocab-009": "d51864ab27ecfa64",
      "ogs:core:benchmark-task:vocab-010": "1748069ad0ebdde7",
      "ogs:core:benchmark-task:vocab-011": "85de6add5c1ebfba",
      "ogs:core:benchmark-task:vocab-012": "44c9b26dc70c4557",
      "ogs:core:benchmark-task:vocab-013": "dc3e62daf8bdf21a",
      "ogs:core:benchmark-task:vocab-014": "3980456a47e9b7c6",
      "ogs:core:benchmark-task:vocab-015": "ad313a98974d64f7",
      "ogs:core:benchmark-task:vocab-016": "95e41ab82cba029a",
      "ogs:core:benchmark-task:vocab-017": "8070fb687fd2f108",
      "ogs:core:benchmark-task:vocab-018": "9e97b59361dbdf0e",
      "ogs:core:benchmark-task:vocab-019": "fc9a763682f1225b",
      "ogs:core:benchmark-task:vocab-020": "56ad4f9c58bc1eb7",
      "ogs:core:benchmark-task:xplain-001": "a75ad0ee2a385794",
      "ogs:core:benchmark-task:xplain-002": "bdc5d8d74c85e9ab",
      "ogs:core:benchmark-task:xplain-003": "3440bcf3978a1c58",
      "ogs:core:benchmark-task:xplain-004": "b3249dafda966f70",
      "ogs:core:benchmark-task:xplain-005": "0cb2f1946943a0e2",
      "ogs:core:benchmark-task:xplain-006": "9a46b0d141442ce3",
      "ogs:core:benchmark-task:xplain-007": "497adc64f3303eb4",
      "ogs:core:benchmark-task:xplain-008": "2120f96a0444a50e",
      "ogs:core:benchmark-task:xplain-009": "c077c498ceed5f73",
      "ogs:core:benchmark-task:xplain-010": "956c6ce654c2e297",
      "ogs:core:benchmark-task:xplain-011": "1dfeb782059a4814",
      "ogs:core:benchmark-task:xplain-012": "28df36c8ae155697",
      "ogs:core:benchmark-task:xplain-013": "18a20e7e30844c41",
      "ogs:core:benchmark-task:xplain-014": "f9be4ef2853d67b3",
      "ogs:core:benchmark-task:xplain-015": "5474f731e5d9751c",
      "ogs:core:benchmark-task:xplain-016": "fc7f227aebed6756",
      "ogs:core:benchmark-task:xplain-017": "024198497e327bb1",
      "ogs:core:benchmark-task:xplain-018": "283bbb9e703e6e30",
      "ogs:core:benchmark-task:xplain-019": "9ea6cd79384f9ee2",
      "ogs:core:benchmark-task:xplain-020": "43005f44c870684b"
    }
  },
  "tasks": [
    {
      "task_id": "ogs:core:benchmark-task:comp-001",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "5dc05e883ee75cf7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0 = no browning, 5 = fully charred), return the integer value best describing the described cook. Respond with just the integer.\n\nINPUT:\n{\n  \"description\": \"a duck breast seared skin-side down until the skin is deep mahogany brown and crisp, but not black\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 3,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "3",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 804.0,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "3",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 623.8,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "3",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 628.6,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 408.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 685.467,
        "usd_sum": 0.00165
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-002",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "d5fef8891eee5391",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0 = no browning, 5 = fully charred), return the integer value for the described cook. Respond with just the integer.\n\nINPUT:\n{\n  \"description\": \"a fresh egg poached in 82 C water for 4 minutes\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 4,
            "latency_ms": 807.7,
            "usd": 0.000539,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 4,
            "latency_ms": 666.9,
            "usd": 0.000539,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 399,
            "output_tokens": 4,
            "latency_ms": 770.0,
            "usd": 0.000539,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 399.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 748.2,
        "usd_sum": 0.001617
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-003",
      "category": "composition",
      "capability": "reduction_level",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.reduction_level"
      ],
      "prompt_template_hash": "7b7db2c313e1d55b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Reduction level (0-5 integer)\n\n- 0: no reduction (raw stock, unreduced liquid)\n- 1: light simmer reduction (~10%)\n- 2: modest reduction (~30%, nappe forming)\n- 3: substantial reduction (~50%, demi-glace territory)\n- 4: heavy reduction (~75%, glace-like, syrupy)\n- 5: near-complete reduction (glace de viande, paste)\n\n---\n\nUsing the OGS reduction level (0 = no reduction, 5 = glace-level), return the integer best describing the sauce. Respond with just the integer.\n\nINPUT:\n{\n  \"description\": \"a veal demi-glace reduced from 4 L of stock down to about 600 mL, thickly coating a spoon with a glossy sheen\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 4,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 792.5,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 1769.4,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 738.1,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 397.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1100.0,
        "usd_sum": 0.001608
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-004",
      "category": "composition",
      "capability": "cooking_method_recognition",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "ca8109a73af762fa",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nA piece of black cod is packed in salt-sugar-sake-lees paste and kept refrigerated for 48 hours, then rinsed and grilled over binchotan. What is the single OGS cooking method that best labels the BEFORE-grilling stage? Respond with just the method name.\n\nINPUT:\n{\n  \"description\": \"black cod packed in salt-sugar-sake-lees paste for 48h before grilling\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "cured"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 431,
            "output_tokens": 5,
            "latency_ms": 593.1,
            "usd": 0.000589,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 431,
            "output_tokens": 5,
            "latency_ms": 761.8,
            "usd": 0.000589,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 431,
            "output_tokens": 5,
            "latency_ms": 943.3,
            "usd": 0.000589,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 431.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 766.067,
        "usd_sum": 0.001767
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-005",
      "category": "composition",
      "capability": "component_role_assignment",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "2885728399552fb5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nFor the described dish element, return the single OGS component role that best applies. Respond with just the role name.\n\nINPUT:\n{\n  \"element\": \"a small handful of thin, deep-fried shallot rings scattered on top of a braised beef rendang primarily for crunchy contrast\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "textural_element"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "textural_element",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "textural_element",
            "expected": "textural_element",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 356,
            "output_tokens": 6,
            "latency_ms": 730.8,
            "usd": 0.000505,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "textural_element",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "textural_element",
            "expected": "textural_element",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 356,
            "output_tokens": 6,
            "latency_ms": 642.6,
            "usd": 0.000505,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "textural_element",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "textural_element",
            "expected": "textural_element",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 356,
            "output_tokens": 6,
            "latency_ms": 1184.2,
            "usd": 0.000505,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 356.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 852.533,
        "usd_sum": 0.001515
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-006",
      "category": "composition",
      "capability": "component_set_identification",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "bb37bf992b2f6c39",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nList the OGS component ROLES that appear in a classic Roman spaghetti alla carbonara (guanciale, eggs, pecorino romano, spaghetti, black pepper). Respond with a JSON array of role strings from the OGS vocabulary, without duplicates, in any order.\n\nINPUT:\n{\n  \"dish\": \"classic Roman spaghetti alla carbonara\",\n  \"components_present\": [\n    \"spaghetti\",\n    \"guanciale\",\n    \"egg yolks\",\n    \"pecorino romano\",\n    \"black pepper\"\n  ]\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"array\",\n  \"items\": {\n    \"type\": \"string\",\n    \"enum\": [\n      \"main_protein\",\n      \"main_vegetable\",\n      \"starch\",\n      \"sauce\",\n      \"fat_component\",\n      \"acid_component\",\n      \"aromatic_garnish\",\n      \"seasoning\",\n      \"textural_element\"\n    ]\n  },\n  \"uniqueItems\": true\n}",
      "expected": {
        "values": [
          "starch",
          "main_protein",
          "fat_component",
          "sauce",
          "seasoning"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"starch\",\"main_protein\",\"sauce\",\"seasoning\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "expected": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "precision": 1.0,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 467,
            "output_tokens": 17,
            "latency_ms": 745.8,
            "usd": 0.000754,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"starch\",\"main_protein\",\"sauce\",\"seasoning\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "expected": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "precision": 1.0,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 467,
            "output_tokens": 17,
            "latency_ms": 733.3,
            "usd": 0.000754,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"starch\",\"main_protein\",\"sauce\",\"seasoning\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "expected": [
              "fat_component",
              "main_protein",
              "sauce",
              "seasoning",
              "starch"
            ],
            "precision": 1.0,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 467,
            "output_tokens": 17,
            "latency_ms": 912.7,
            "usd": 0.000754,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.8889,
      "score_summary": {
        "mean": 0.8889,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8889,
        "ci95_hi": 0.8889,
        "n": 3
      },
      "score_if_valid": 0.8889,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 467.0,
        "output_tokens_mean": 17.0,
        "latency_ms_mean": 797.267,
        "usd_sum": 0.002262
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-007",
      "category": "composition",
      "capability": "service_temperature",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.service_temperatures"
      ],
      "prompt_template_hash": "59ae41132f38bbfd",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n---\n\nReturn the single OGS service_temperature value that matches the traditional service of the dish below. Respond with just the value.\n\nINPUT:\n{\n  \"dish\": \"classic French vichyssoise (leek and potato soup)\"\n}\n\nANSWER FORMAT: one_of:frozen,cold,cool,room,warm,hot,very_hot",
      "expected": {
        "value": "cold"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 312,
            "output_tokens": 4,
            "latency_ms": 688.6,
            "usd": 0.00043,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 312,
            "output_tokens": 4,
            "latency_ms": 751.7,
            "usd": 0.00043,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 312,
            "output_tokens": 4,
            "latency_ms": 826.6,
            "usd": 0.00043,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 312.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 755.633,
        "usd_sum": 0.00129
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-008",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "5121ce4bf46aa4d5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0 = no browning, 5 = fully charred), return the integer for a steamed dim sum dumpling wrapper.\n\nINPUT:\n{\n  \"description\": \"translucent wheat-starch wrapper on a har gow, bamboo-steamed\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 721.4,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 1218.9,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 956.8,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 397.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 965.7,
        "usd_sum": 0.001608
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-009",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "2a8a86bb58372ed7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index (0-5 integer), return the value for a chicken breast poached in stock at 82°C with no prior searing.\n\nINPUT:\n{\n  \"description\": \"chicken breast poached in 82°C stock, never seared\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 0,
        "tolerance": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 700.2,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 1066.5,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 0.0,
            "expected": 0.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 1282.5,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 397.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1016.4,
        "usd_sum": 0.001608
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-010",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "d420afcaea603a0e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index, return the integer for a scallop seared in hot butter until a deep golden crust forms on one side.\n\nINPUT:\n{\n  \"description\": \"diver scallop seared in foaming butter, one side deep golden, interior translucent\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 3,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 686.3,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 890.3,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 397,
            "output_tokens": 4,
            "latency_ms": 1009.3,
            "usd": 0.000536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 397.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 861.967,
        "usd_sum": 0.001608
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-011",
      "category": "composition",
      "capability": "reduction_level",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.reduction_level"
      ],
      "prompt_template_hash": "1d18b06eac9f3ddc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Reduction level (0-5 integer)\n\n- 0: no reduction (raw stock, unreduced liquid)\n- 1: light simmer reduction (~10%)\n- 2: modest reduction (~30%, nappe forming)\n- 3: substantial reduction (~50%, demi-glace territory)\n- 4: heavy reduction (~75%, glace-like, syrupy)\n- 5: near-complete reduction (glace de viande, paste)\n\n---\n\nUsing the OGS reduction level (0-5 integer), return the value for a pan sauce reduced until it coats the back of a spoon (nappe).\n\nINPUT:\n{\n  \"description\": \"red wine pan sauce reduced until it coats the back of a spoon without running\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 3,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 382,
            "output_tokens": 4,
            "latency_ms": 738.1,
            "usd": 0.000518,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 382,
            "output_tokens": 4,
            "latency_ms": 626.1,
            "usd": 0.000518,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "2",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 2.0,
            "expected": 3.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 382,
            "output_tokens": 4,
            "latency_ms": 732.8,
            "usd": 0.000518,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 382.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 699.0,
        "usd_sum": 0.001554
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-012",
      "category": "composition",
      "capability": "reduction_level",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.reduction_level"
      ],
      "prompt_template_hash": "fa91f74e86a4cffb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Reduction level (0-5 integer)\n\n- 0: no reduction (raw stock, unreduced liquid)\n- 1: light simmer reduction (~10%)\n- 2: modest reduction (~30%, nappe forming)\n- 3: substantial reduction (~50%, demi-glace territory)\n- 4: heavy reduction (~75%, glace-like, syrupy)\n- 5: near-complete reduction (glace de viande, paste)\n\n---\n\nUsing the OGS reduction level, return the integer for glace de viande (gelatinous concentrate that sets firmly at room temperature).\n\nINPUT:\n{\n  \"description\": \"glace de viande, reduced from 10 L brown stock to ~500 mL\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 5,
        "tolerance": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 5.0,
            "expected": 5.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 4,
            "latency_ms": 702.5,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 5.0,
            "expected": 5.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 4,
            "latency_ms": 757.5,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 5.0,
            "expected": 5.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 4,
            "latency_ms": 1091.2,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 378.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 850.4,
        "usd_sum": 0.001539
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-013",
      "category": "composition",
      "capability": "cooking_method_recognition",
      "difficulty": "medium",
      "cuisine_region": "korean",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "f8257062be99b5d3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nReturn the single OGS cooking method that best fits the main cooking step for bossam pork belly (~90 min at low simmer in aromatic water).\n\nINPUT:\n{\n  \"description\": \"pork belly gently simmered for 90 minutes in water with ginger, doenjang, and pepper until fork-tender\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "poached"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "poached",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "poached",
            "expected": "poached",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 5,
            "latency_ms": 904.4,
            "usd": 0.00056,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "poached",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "poached",
            "expected": "poached",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 5,
            "latency_ms": 870.6,
            "usd": 0.00056,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "poached",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "poached",
            "expected": "poached",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 5,
            "latency_ms": 920.3,
            "usd": 0.00056,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 408.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 898.433,
        "usd_sum": 0.00168
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-014",
      "category": "composition",
      "capability": "component_role_assignment",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "114b0796c10cf141",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits a sprinkle of gremolata (parsley, garlic, lemon zest) finishing an osso buco at service.\n\nINPUT:\n{\n  \"description\": \"gremolata finishing an osso buco milanese at service\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "aromatic_garnish"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "aromatic_garnish",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "aromatic_garnish",
            "expected": "aromatic_garnish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 8,
            "latency_ms": 663.4,
            "usd": 0.000523,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "aromatic_garnish",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "aromatic_garnish",
            "expected": "aromatic_garnish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 8,
            "latency_ms": 941.7,
            "usd": 0.000523,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "aromatic_garnish",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "aromatic_garnish",
            "expected": "aromatic_garnish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 8,
            "latency_ms": 703.4,
            "usd": 0.000523,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 354.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 769.5,
        "usd_sum": 0.001569
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-015",
      "category": "composition",
      "capability": "component_set_identification",
      "difficulty": "hard",
      "cuisine_region": "thai",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "07976f0ee97244e5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nList the OGS component roles that appear in a classic Pad Thai (rice noodles, shrimp, tofu, egg, tamarind-palm sugar-fish sauce, bean sprouts, crushed peanuts, chives). Return a JSON array of role names.\n\nINPUT:\n{\n  \"dish\": \"classic Pad Thai\"\n}\n\nANSWER FORMAT: set_of_roles",
      "expected": {
        "values": [
          "starch",
          "main_protein",
          "acid_component",
          "seasoning",
          "aromatic_garnish",
          "textural_element",
          "fat_component"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"main_protein\",\"starch\",\"sauce\",\"main_vegetable\",\"textural_element\",\"aromatic_garnish\"]",
          "compliant": true,
          "score": 0.6154,
          "detail": {
            "predicted": [
              "aromatic_garnish",
              "main_protein",
              "main_vegetable",
              "sauce",
              "starch",
              "textural_element"
            ],
            "expected": [
              "acid_component",
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "precision": 0.6667,
            "recall": 0.5714
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 29,
            "latency_ms": 1214.6,
            "usd": 0.0007,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"main_protein\",\"starch\",\"sauce\",\"main_vegetable\",\"textural_element\",\"aromatic_garnish\"]",
          "compliant": true,
          "score": 0.6154,
          "detail": {
            "predicted": [
              "aromatic_garnish",
              "main_protein",
              "main_vegetable",
              "sauce",
              "starch",
              "textural_element"
            ],
            "expected": [
              "acid_component",
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "precision": 0.6667,
            "recall": 0.5714
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 29,
            "latency_ms": 1188.7,
            "usd": 0.0007,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"main_protein\",\"starch\",\"sauce\",\"main_vegetable\",\"textural_element\",\"aromatic_garnish\"]",
          "compliant": true,
          "score": 0.6154,
          "detail": {
            "predicted": [
              "aromatic_garnish",
              "main_protein",
              "main_vegetable",
              "sauce",
              "starch",
              "textural_element"
            ],
            "expected": [
              "acid_component",
              "aromatic_garnish",
              "fat_component",
              "main_protein",
              "seasoning",
              "starch",
              "textural_element"
            ],
            "precision": 0.6667,
            "recall": 0.5714
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 29,
            "latency_ms": 932.5,
            "usd": 0.0007,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6154,
      "score_summary": {
        "mean": 0.6154,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6154,
        "ci95_hi": 0.6154,
        "n": 3
      },
      "score_if_valid": 0.6154,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 328.0,
        "output_tokens_mean": 29.0,
        "latency_ms_mean": 1111.933,
        "usd_sum": 0.0021
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-016",
      "category": "composition",
      "capability": "service_temperature",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.service_temperatures"
      ],
      "prompt_template_hash": "609a0643399a8df0",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n---\n\nReturn the single OGS service_temperature value for a Spanish ajo blanco (chilled almond-garlic soup).\n\nINPUT:\n{\n  \"dish\": \"ajo blanco, chilled Andalusian almond and garlic soup\"\n}\n\nANSWER FORMAT: one_of:frozen,cold,cool,room,warm,hot,very_hot",
      "expected": {
        "value": "cold"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 308,
            "output_tokens": 4,
            "latency_ms": 818.4,
            "usd": 0.000425,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 308,
            "output_tokens": 4,
            "latency_ms": 889.1,
            "usd": 0.000425,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 308,
            "output_tokens": 4,
            "latency_ms": 923.1,
            "usd": 0.000425,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 308.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 876.867,
        "usd_sum": 0.001275
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-017",
      "category": "composition",
      "capability": "cooking_method_recognition",
      "difficulty": "medium",
      "cuisine_region": "chinese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "09a6e47eda264ded",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nReturn the single OGS cooking method that best describes the signature crisp skin of Peking duck (air-dried then hung-roasted in a wood-fired oven).\n\nINPUT:\n{\n  \"description\": \"Peking duck hung-roasted in a jujube-wood oven at ~230°C until the skin is lacquered and crisp\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "roasted"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 5,
            "latency_ms": 865.0,
            "usd": 0.000566,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 5,
            "latency_ms": 737.3,
            "usd": 0.000566,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 413,
            "output_tokens": 5,
            "latency_ms": 839.7,
            "usd": 0.000566,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 413.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 814.0,
        "usd_sum": 0.001698
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-018",
      "category": "composition",
      "capability": "component_role_assignment",
      "difficulty": "medium",
      "cuisine_region": "levantine",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "32d4a921f4fbdd98",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the OGS component role that best fits tahini as an ingredient in a classic hummus recipe.\n\nINPUT:\n{\n  \"description\": \"25% tahini by weight blended into chickpea puree to make hummus\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "fat_component"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fat_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fat_component",
            "expected": "fat_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 342,
            "output_tokens": 5,
            "latency_ms": 643.4,
            "usd": 0.000478,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "fat_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fat_component",
            "expected": "fat_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 342,
            "output_tokens": 5,
            "latency_ms": 752.8,
            "usd": 0.000478,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "fat_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fat_component",
            "expected": "fat_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 342,
            "output_tokens": 5,
            "latency_ms": 5586.5,
            "usd": 0.000478,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 342.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 2327.567,
        "usd_sum": 0.001434
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-019",
      "category": "composition",
      "capability": "service_temperature",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.service_temperatures"
      ],
      "prompt_template_hash": "eaa27377a264e2e2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n---\n\nReturn the OGS service_temperature for a stirred-over-ice Negroni served in a rocks glass with a large cube.\n\nINPUT:\n{\n  \"dish\": \"stirred Negroni served over a single large ice cube\"\n}\n\nANSWER FORMAT: one_of:frozen,cold,cool,room,warm,hot,very_hot",
      "expected": {
        "value": "cold"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 313,
            "output_tokens": 4,
            "latency_ms": 628.1,
            "usd": 0.000431,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 313,
            "output_tokens": 4,
            "latency_ms": 802.0,
            "usd": 0.000431,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "cold",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cold",
            "expected": "cold",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 313,
            "output_tokens": 4,
            "latency_ms": 906.4,
            "usd": 0.000431,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 313.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 778.833,
        "usd_sum": 0.001293
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:comp-020",
      "category": "composition",
      "capability": "maillard_index",
      "difficulty": "hard",
      "cuisine_region": "american",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.maillard_index"
      ],
      "prompt_template_hash": "535bf59391539a15",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Maillard index (0-5 integer)\n\n- 0: no browning (raw, poached, steamed)\n- 1: pale gold (soft-baked dough, lightly sauteed)\n- 2: golden brown (roasted chicken skin)\n- 3: deep mahogany (well-seared steak crust, duck breast skin)\n- 4: very dark brown with bitter edge (heavy crust, blackened fish)\n- 5: fully charred (intentionally blackened, carbonised)\n\n---\n\nUsing the OGS Maillard index, return the integer for blackened redfish cooked in a smoking-hot cast iron skillet with Cajun spice rub.\n\nINPUT:\n{\n  \"description\": \"redfish fillet crusted with spice and cooked in a 300°C dry cast-iron pan until the surface is blackened and slightly bitter\"\n}\n\nANSWER FORMAT: integer_0_5",
      "expected": {
        "value": 4,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 978.3,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 815.9,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "4",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 4.0,
            "expected": 4.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 408,
            "output_tokens": 4,
            "latency_ms": 606.4,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 408.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 800.2,
        "usd_sum": 0.00165
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-001",
      "category": "dietary",
      "capability": "vegan_compatibility",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "18b5c9c7c958a521",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nIs a traditional Bolognese tagliatelle al ragù (tagliatelle with ragù made from beef, pancetta, soffritto, tomato, red wine, whole milk, and butter) compatible with a strict vegan diet? Answer with a single word: yes or no.\n\nINPUT:\n{\n  \"dish\": \"tagliatelle al ragù\",\n  \"diet\": \"vegan\"\n}\n\nANSWER FORMAT: exact:yes|no",
      "expected": {
        "value": "no"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 910,
            "output_tokens": 4,
            "latency_ms": 737.5,
            "usd": 0.001178,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 910,
            "output_tokens": 4,
            "latency_ms": 742.0,
            "usd": 0.001178,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 910,
            "output_tokens": 4,
            "latency_ms": 666.7,
            "usd": 0.001178,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 910.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 715.4,
        "usd_sum": 0.003534
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-002",
      "category": "dietary",
      "capability": "gluten_free_violation",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "04840e46f80ccb5b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nA kitchen claims this stir-fry is gluten-free: rice noodles, tofu, shiitake, broccoli, garlic, ginger, regular Japanese soy sauce, sesame oil, and scallions. Return the single ingredient that breaks the claim.\n\nINPUT:\n{\n  \"dish\": \"stir fry\",\n  \"ingredients\": [\n    \"rice noodles\",\n    \"tofu\",\n    \"shiitake\",\n    \"broccoli\",\n    \"garlic\",\n    \"ginger\",\n    \"regular Japanese soy sauce\",\n    \"sesame oil\",\n    \"scallions\"\n  ]\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "regular Japanese soy sauce",
        "aliases": [
          "soy sauce",
          "regular soy sauce"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "regular Japanese soy sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "regular japanese soy sauce",
            "expected": "regular japanese soy sauce",
            "aliases": [
              "soy sauce",
              "regular soy sauce"
            ]
          },
          "metrics": {
            "input_tokens": 938,
            "output_tokens": 7,
            "latency_ms": 989.7,
            "usd": 0.001242,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "regular Japanese soy sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "regular japanese soy sauce",
            "expected": "regular japanese soy sauce",
            "aliases": [
              "soy sauce",
              "regular soy sauce"
            ]
          },
          "metrics": {
            "input_tokens": 938,
            "output_tokens": 7,
            "latency_ms": 923.5,
            "usd": 0.001242,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "regular Japanese soy sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "regular japanese soy sauce",
            "expected": "regular japanese soy sauce",
            "aliases": [
              "soy sauce",
              "regular soy sauce"
            ]
          },
          "metrics": {
            "input_tokens": 938,
            "output_tokens": 7,
            "latency_ms": 766.3,
            "usd": 0.001242,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 938.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 893.167,
        "usd_sum": 0.003726
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-003",
      "category": "dietary",
      "capability": "halal_compliance",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "013d6295fee472b5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nIs a classic panna cotta (cream, sugar, vanilla, leaf gelatine) halal? Answer with yes or no.\n\nINPUT:\n{\n  \"dish\": \"panna cotta\",\n  \"ingredients\": [\n    \"cream\",\n    \"sugar\",\n    \"vanilla\",\n    \"leaf gelatine\"\n  ]\n}\n\nANSWER FORMAT: exact:yes|no",
      "expected": {
        "value": "no"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 890,
            "output_tokens": 4,
            "latency_ms": 716.3,
            "usd": 0.001153,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 890,
            "output_tokens": 4,
            "latency_ms": 591.1,
            "usd": 0.001153,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "no",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "no",
            "expected": "no",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 890,
            "output_tokens": 4,
            "latency_ms": 687.1,
            "usd": 0.001153,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 890.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 664.833,
        "usd_sum": 0.003459
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-004",
      "category": "dietary",
      "capability": "kosher_meat_dairy",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "85fc94958c4357ef",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nChicken parmigiana is breaded chicken breast topped with tomato sauce and melted mozzarella, served with a side of cream-rich garlic mashed potatoes. From a kosher perspective, what is the core violation? Respond with one short phrase.\n\nINPUT:\n{\n  \"dish\": \"chicken parmigiana with creamy mash\"\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "mixing meat and dairy",
        "aliases": [
          "meat and dairy mixed",
          "basar b'chalav",
          "poultry with dairy",
          "meat with dairy"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "mixing meat and dairy",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mixing meat and dairy",
            "expected": "mixing meat and dairy",
            "aliases": [
              "meat and dairy mixed",
              "basar b'chalav",
              "poultry with dairy",
              "meat with dairy"
            ]
          },
          "metrics": {
            "input_tokens": 884,
            "output_tokens": 8,
            "latency_ms": 817.8,
            "usd": 0.001185,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "mixing meat and dairy",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mixing meat and dairy",
            "expected": "mixing meat and dairy",
            "aliases": [
              "meat and dairy mixed",
              "basar b'chalav",
              "poultry with dairy",
              "meat with dairy"
            ]
          },
          "metrics": {
            "input_tokens": 884,
            "output_tokens": 8,
            "latency_ms": 758.3,
            "usd": 0.001185,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "mixing meat and dairy",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mixing meat and dairy",
            "expected": "mixing meat and dairy",
            "aliases": [
              "meat and dairy mixed",
              "basar b'chalav",
              "poultry with dairy",
              "meat with dairy"
            ]
          },
          "metrics": {
            "input_tokens": 884,
            "output_tokens": 8,
            "latency_ms": 835.3,
            "usd": 0.001185,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 884.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 803.8,
        "usd_sum": 0.003555
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-005",
      "category": "dietary",
      "capability": "allergen_detection",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "f40f4c25a2ee4b71",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nFor a dish of pesto genovese made from basil, garlic, pine nuts, parmesan, pecorino, and olive oil, list the EU Annex II allergens that must be declared. Return a comma-separated list using the OGS allergen IDs.\n\nINPUT:\n{\n  \"dish\": \"pesto genovese\",\n  \"ingredients\": [\n    \"basil\",\n    \"garlic\",\n    \"pine nuts\",\n    \"parmesan\",\n    \"pecorino\",\n    \"olive oil\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated OGS allergen IDs",
      "expected": {
        "values": [
          "tree_nuts",
          "milk"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "milk,tree_nuts",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "milk",
              "tree_nuts"
            ],
            "expected": [
              "milk",
              "tree_nuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 933,
            "output_tokens": 8,
            "latency_ms": 850.0,
            "usd": 0.001246,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "milk,tree_nuts",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "milk",
              "tree_nuts"
            ],
            "expected": [
              "milk",
              "tree_nuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 933,
            "output_tokens": 8,
            "latency_ms": 807.1,
            "usd": 0.001246,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "milk,tree_nuts",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "milk",
              "tree_nuts"
            ],
            "expected": [
              "milk",
              "tree_nuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 933,
            "output_tokens": 8,
            "latency_ms": 853.8,
            "usd": 0.001246,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 933.0,
        "output_tokens_mean": 8.0,
        "latency_ms_mean": 836.967,
        "usd_sum": 0.003738
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-006",
      "category": "dietary",
      "capability": "allergen_detection",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "7f45e388301a68a5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nReturn the EU Annex II allergens that must be declared for a standard pad thai (rice noodles, shrimp, egg, tamarind, fish sauce, palm sugar, roasted peanuts, bean sprouts, garlic chives, lime, chili).\n\nINPUT:\n{\n  \"dish\": \"pad thai\"\n}\n\nANSWER FORMAT: set:comma-separated OGS allergen IDs",
      "expected": {
        "values": [
          "crustaceans",
          "eggs",
          "fish",
          "peanuts"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "crustaceans,eggs,fish,peanuts",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "expected": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 887,
            "output_tokens": 15,
            "latency_ms": 846.8,
            "usd": 0.001259,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "crustaceans,eggs,fish,peanuts",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "expected": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 887,
            "output_tokens": 15,
            "latency_ms": 893.4,
            "usd": 0.001259,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "crustaceans,eggs,fish,peanuts",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "expected": [
              "crustaceans",
              "eggs",
              "fish",
              "peanuts"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 887,
            "output_tokens": 15,
            "latency_ms": 854.5,
            "usd": 0.001259,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 887.0,
        "output_tokens_mean": 15.0,
        "latency_ms_mean": 864.9,
        "usd_sum": 0.003777
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-007",
      "category": "dietary",
      "capability": "vegan_substitution",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "eb092a2c162960f6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nName a single plant-based ingredient commonly used to emulsify and thicken a vegan carbonara sauce in place of eggs, while also delivering umami. Respond with one short ingredient name.\n\nINPUT:\n{\n  \"target_role\": \"egg emulsifier and umami in vegan carbonara\"\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "silken tofu",
        "aliases": [
          "silken tofu blended with nutritional yeast",
          "soft tofu"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "miso",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "miso",
            "expected": "silken tofu",
            "aliases": [
              "silken tofu blended with nutritional yeast",
              "soft tofu"
            ]
          },
          "metrics": {
            "input_tokens": 879,
            "output_tokens": 5,
            "latency_ms": 678.6,
            "usd": 0.001149,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "miso",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "miso",
            "expected": "silken tofu",
            "aliases": [
              "silken tofu blended with nutritional yeast",
              "soft tofu"
            ]
          },
          "metrics": {
            "input_tokens": 879,
            "output_tokens": 5,
            "latency_ms": 841.4,
            "usd": 0.001149,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "miso",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "miso",
            "expected": "silken tofu",
            "aliases": [
              "silken tofu blended with nutritional yeast",
              "soft tofu"
            ]
          },
          "metrics": {
            "input_tokens": 879,
            "output_tokens": 5,
            "latency_ms": 746.2,
            "usd": 0.001149,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 879.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 755.4,
        "usd_sum": 0.003447
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-008",
      "category": "dietary",
      "capability": "gluten_free_substitution",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "480be99b3f35a51d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nSuggest a single widely-used gluten-free flour to make a béchamel roux that maintains classic texture. Respond with one word.\n\nINPUT:\n{\n  \"target_role\": \"béchamel roux, gluten-free\"\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "rice flour",
        "aliases": [
          "cornstarch",
          "corn starch",
          "cornflour",
          "tapioca starch",
          "tapioca flour"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "rice",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "rice",
            "expected": "rice flour",
            "aliases": [
              "cornstarch",
              "corn starch",
              "cornflour",
              "tapioca starch",
              "tapioca flour"
            ]
          },
          "metrics": {
            "input_tokens": 866,
            "output_tokens": 4,
            "latency_ms": 764.3,
            "usd": 0.001123,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "rice",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "rice",
            "expected": "rice flour",
            "aliases": [
              "cornstarch",
              "corn starch",
              "cornflour",
              "tapioca starch",
              "tapioca flour"
            ]
          },
          "metrics": {
            "input_tokens": 866,
            "output_tokens": 4,
            "latency_ms": 703.4,
            "usd": 0.001123,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "rice",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "rice",
            "expected": "rice flour",
            "aliases": [
              "cornstarch",
              "corn starch",
              "cornflour",
              "tapioca starch",
              "tapioca flour"
            ]
          },
          "metrics": {
            "input_tokens": 866,
            "output_tokens": 4,
            "latency_ms": 721.0,
            "usd": 0.001123,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 866.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 729.567,
        "usd_sum": 0.003369
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-009",
      "category": "dietary",
      "capability": "vegan_compatibility",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "936752891554dba3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nFrom this list, return every item that is NOT vegan as a comma-separated list: honey, agave nectar, wine clarified with isinglass, wine clarified with bentonite, carmine, nutritional yeast, shellac, maple syrup, worcestershire sauce (traditional), vegetable glycerin.\n\nINPUT:\n{\n  \"items\": [\n    \"honey\",\n    \"agave nectar\",\n    \"wine clarified with isinglass\",\n    \"wine clarified with bentonite\",\n    \"carmine\",\n    \"nutritional yeast\",\n    \"shellac\",\n    \"maple syrup\",\n    \"worcestershire sauce (traditional)\",\n    \"vegetable glycerin\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated item names",
      "expected": {
        "values": [
          "honey",
          "wine clarified with isinglass",
          "carmine",
          "shellac",
          "worcestershire sauce (traditional)"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "honey,wine clarified with isinglass,carmine,shellac,worcestershire sauce (traditional)",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "expected": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 963,
            "output_tokens": 25,
            "latency_ms": 868.5,
            "usd": 0.001454,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "honey, wine clarified with isinglass, carmine, shellac, worcestershire sauce (traditional)",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "expected": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 963,
            "output_tokens": 26,
            "latency_ms": 885.8,
            "usd": 0.001464,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "honey,wine clarified with isinglass,carmine,shellac,worcestershire sauce (traditional)",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "expected": [
              "carmine",
              "honey",
              "shellac",
              "wine clarified with isinglass",
              "worcestershire sauce (traditional)"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 963,
            "output_tokens": 25,
            "latency_ms": 817.8,
            "usd": 0.001454,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 963.0,
        "output_tokens_mean": 25.333,
        "latency_ms_mean": 857.367,
        "usd_sum": 0.004372
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-010",
      "category": "dietary",
      "capability": "cross_contact_risk",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "35c29846f76046d7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nA kitchen fries breaded calamari, battered cod, and french fries in a single deep fryer. Return the EU allergens that a customer ordering 'fries only' must be warned about as a comma-separated OGS allergen ID list.\n\nINPUT:\n{\n  \"scenario\": \"single shared fryer\",\n  \"items\": [\n    \"breaded calamari\",\n    \"battered cod\",\n    \"french fries\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated OGS allergen IDs",
      "expected": {
        "values": [
          "molluscs",
          "fish",
          "cereals_containing_gluten"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cereals_containing_gluten,fish,molluscs",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "expected": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 915,
            "output_tokens": 16,
            "latency_ms": 841.4,
            "usd": 0.001304,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "cereals_containing_gluten,fish,molluscs",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "expected": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 915,
            "output_tokens": 16,
            "latency_ms": 932.3,
            "usd": 0.001304,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "cereals_containing_gluten,fish,molluscs",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "expected": [
              "cereals_containing_gluten",
              "fish",
              "molluscs"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 915,
            "output_tokens": 16,
            "latency_ms": 717.2,
            "usd": 0.001304,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 915.0,
        "output_tokens_mean": 16.0,
        "latency_ms_mean": 830.3,
        "usd_sum": 0.003912
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-011",
      "category": "dietary",
      "capability": "low_fodmap",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "bdd7970997c11172",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nReturn the low-FODMAP violations in this dish as a comma-separated list of ingredient names: garlic-infused olive oil, white onion, lactose-free milk, wheat pasta, lean chicken, fresh chives, honey, aged parmesan.\n\nINPUT:\n{\n  \"ingredients\": [\n    \"garlic-infused olive oil\",\n    \"white onion\",\n    \"lactose-free milk\",\n    \"wheat pasta\",\n    \"lean chicken\",\n    \"fresh chives\",\n    \"honey\",\n    \"aged parmesan\"\n  ]\n}\n\nANSWER FORMAT: set:comma-separated ingredient names",
      "expected": {
        "values": [
          "white onion",
          "wheat pasta",
          "honey"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "white onion,wheat pasta,honey",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "expected": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 932,
            "output_tokens": 10,
            "latency_ms": 898.7,
            "usd": 0.001265,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "white onion,wheat pasta,honey",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "expected": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 932,
            "output_tokens": 10,
            "latency_ms": 705.0,
            "usd": 0.001265,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "white onion,wheat pasta,honey",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "expected": [
              "honey",
              "wheat pasta",
              "white onion"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 932,
            "output_tokens": 10,
            "latency_ms": 697.2,
            "usd": 0.001265,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 932.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 766.967,
        "usd_sum": 0.003795
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:diet-012",
      "category": "dietary",
      "capability": "pescatarian_vs_vegetarian",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "set_f1",
      "context_needs": [
        "rules.dietary",
        "rules.allergens_eu",
        "vocab.dietary_labels"
      ],
      "prompt_template_hash": "42c897d741064981",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n### Dietary labels\n\nvegan, vegetarian, pescatarian, halal, kosher, gluten_free, dairy_free, nut_free, shellfish_free, egg_free, soy_free, low_fodmap\n\n---\n\nA dish contains miso, dashi (made with katsuobushi), tofu, and wakame. Mark which labels it qualifies for out of: vegan, vegetarian, pescatarian. Return a comma-separated list.\n\nINPUT:\n{\n  \"dish\": \"classic awase-dashi miso soup with tofu and wakame\"\n}\n\nANSWER FORMAT: set:comma-separated labels",
      "expected": {
        "values": [
          "pescatarian"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "pescatarian",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "pescatarian"
            ],
            "expected": [
              "pescatarian"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 892,
            "output_tokens": 7,
            "latency_ms": 663.7,
            "usd": 0.001185,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "pescatarian",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "pescatarian"
            ],
            "expected": [
              "pescatarian"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 892,
            "output_tokens": 7,
            "latency_ms": 797.8,
            "usd": 0.001185,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "pescatarian",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": [
              "pescatarian"
            ],
            "expected": [
              "pescatarian"
            ],
            "precision": 1.0,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 892,
            "output_tokens": 7,
            "latency_ms": 870.9,
            "usd": 0.001185,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 892.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 777.467,
        "usd_sum": 0.003555
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-001",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "138c09ed611a2d43",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has above-average popularity and above-average contribution margin on the menu. Using the OGS menu-engineering matrix, which quadrant is it? Respond with one word.\n\nINPUT:\n{\n  \"popularity\": \"above\",\n  \"margin\": \"above\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "star"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "star",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "star",
            "expected": "star",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 498,
            "output_tokens": 4,
            "latency_ms": 708.1,
            "usd": 0.000663,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "star",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "star",
            "expected": "star",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 498,
            "output_tokens": 4,
            "latency_ms": 693.3,
            "usd": 0.000663,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "star",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "star",
            "expected": "star",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 498,
            "output_tokens": 4,
            "latency_ms": 696.2,
            "usd": 0.000663,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 498.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 699.2,
        "usd_sum": 0.001989
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-002",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "b5df321fd829de76",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish sells strongly but the contribution margin is below the menu average. Classify it.\n\nINPUT:\n{\n  \"popularity\": \"above\",\n  \"margin\": \"below\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "plowhorse"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "plowhorse",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "plowhorse",
            "expected": "plowhorse",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 482,
            "output_tokens": 6,
            "latency_ms": 651.5,
            "usd": 0.000663,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "plowhorse",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "plowhorse",
            "expected": "plowhorse",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 482,
            "output_tokens": 6,
            "latency_ms": 716.4,
            "usd": 0.000663,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "plowhorse",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "plowhorse",
            "expected": "plowhorse",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 482,
            "output_tokens": 6,
            "latency_ms": 758.8,
            "usd": 0.000663,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 482.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 708.9,
        "usd_sum": 0.001989
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-003",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "9795b72bdda437f1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has strong contribution margin but is rarely ordered. Classify it.\n\nINPUT:\n{\n  \"popularity\": \"below\",\n  \"margin\": \"above\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "puzzle"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "puzzle",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "puzzle",
            "expected": "puzzle",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 479,
            "output_tokens": 5,
            "latency_ms": 541.3,
            "usd": 0.000649,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "puzzle",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "puzzle",
            "expected": "puzzle",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 479,
            "output_tokens": 5,
            "latency_ms": 573.8,
            "usd": 0.000649,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "puzzle",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "puzzle",
            "expected": "puzzle",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 479,
            "output_tokens": 5,
            "latency_ms": 712.8,
            "usd": 0.000649,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 479.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 609.3,
        "usd_sum": 0.001947
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-004",
      "category": "menu_engineering",
      "capability": "matrix_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "fc0da6dcbbc842fb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish is rarely ordered and has a contribution margin below the menu average. Classify it.\n\nINPUT:\n{\n  \"popularity\": \"below\",\n  \"margin\": \"below\"\n}\n\nANSWER FORMAT: exact:star|plowhorse|puzzle|dog",
      "expected": {
        "value": "dog"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "dog",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "dog",
            "expected": "dog",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 483,
            "output_tokens": 4,
            "latency_ms": 572.6,
            "usd": 0.000644,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "dog",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "dog",
            "expected": "dog",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 483,
            "output_tokens": 4,
            "latency_ms": 623.8,
            "usd": 0.000644,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "dog",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "dog",
            "expected": "dog",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 483,
            "output_tokens": 4,
            "latency_ms": 1806.4,
            "usd": 0.000644,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 483.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1000.933,
        "usd_sum": 0.001932
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-005",
      "category": "menu_engineering",
      "capability": "food_cost",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "f33710f88f89121e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has a food cost of £4.20 and a menu price of £14.00. Return the food-cost percentage as a single integer.\n\nINPUT:\n{\n  \"food_cost\": 4.2,\n  \"menu_price\": 14.0\n}\n\nANSWER FORMAT: numeric:integer percent",
      "expected": {
        "value": 30,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "30",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 30.0,
            "expected": 30.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 492,
            "output_tokens": 4,
            "latency_ms": 746.1,
            "usd": 0.000655,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "30",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 30.0,
            "expected": 30.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 492,
            "output_tokens": 4,
            "latency_ms": 645.1,
            "usd": 0.000655,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "30",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 30.0,
            "expected": 30.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 492,
            "output_tokens": 4,
            "latency_ms": 657.3,
            "usd": 0.000655,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 492.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 682.833,
        "usd_sum": 0.001965
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-006",
      "category": "menu_engineering",
      "capability": "contribution_margin",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "7bb019ca9ca4b1be",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA dish has a food cost of $6.50 and a menu price of $22.00. Return the contribution margin in dollars as a single number with two decimals.\n\nINPUT:\n{\n  \"food_cost\": 6.5,\n  \"menu_price\": 22.0\n}\n\nANSWER FORMAT: numeric:decimal USD",
      "expected": {
        "value": 15.5,
        "tolerance": 0.05
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "15.50",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 15.5,
            "expected": 15.5,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 496,
            "output_tokens": 6,
            "latency_ms": 666.9,
            "usd": 0.00068,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "15.50",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 15.5,
            "expected": 15.5,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 496,
            "output_tokens": 6,
            "latency_ms": 700.2,
            "usd": 0.00068,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "15.50",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 15.5,
            "expected": 15.5,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 496,
            "output_tokens": 6,
            "latency_ms": 805.0,
            "usd": 0.00068,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 496.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 724.033,
        "usd_sum": 0.00204
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-007",
      "category": "menu_engineering",
      "capability": "menu_rebalance",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "68b7fb2ec1e432f4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nFive main courses have this (popularity %, margin $) profile: A (35, 18), B (30, 9), C (5, 6), D (20, 16), E (10, 7). Average popularity is 20%; average margin is $11.20. Which single dish is the dog? Respond with the letter only.\n\nINPUT:\n{\n  \"dishes\": [\n    {\n      \"id\": \"A\",\n      \"pop\": 35,\n      \"margin\": 18\n    },\n    {\n      \"id\": \"B\",\n      \"pop\": 30,\n      \"margin\": 9\n    },\n    {\n      \"id\": \"C\",\n      \"pop\": 5,\n      \"margin\": 6\n    },\n    {\n      \"id\": \"D\",\n      \"pop\": 20,\n      \"margin\": 16\n    },\n    {\n      \"id\": \"E\",\n      \"pop\": 10,\n      \"margin\": 7\n    }\n  ],\n  \"avg_pop\": 20,\n  \"avg_margin\": 11.2\n}\n\nANSWER FORMAT: exact:A|B|C|D|E",
      "expected": {
        "value": "C"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c",
            "expected": "c",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 674,
            "output_tokens": 4,
            "latency_ms": 602.6,
            "usd": 0.000883,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c",
            "expected": "c",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 674,
            "output_tokens": 4,
            "latency_ms": 653.9,
            "usd": 0.000883,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c",
            "expected": "c",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 674,
            "output_tokens": 4,
            "latency_ms": 634.0,
            "usd": 0.000883,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 674.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 630.167,
        "usd_sum": 0.002649
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-008",
      "category": "menu_engineering",
      "capability": "target_price",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles"
      ],
      "prompt_template_hash": "dc5bf7ed3a0583c2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nA chef wants a 28% food cost on a dish whose ingredient cost is £3.92. Return the target menu price in pounds, two decimals.\n\nINPUT:\n{\n  \"food_cost\": 3.92,\n  \"target_food_cost_pct\": 28\n}\n\nANSWER FORMAT: numeric:decimal GBP",
      "expected": {
        "value": 14.0,
        "tolerance": 0.05
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "14.00",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 14.0,
            "expected": 14.0,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 493,
            "output_tokens": 6,
            "latency_ms": 754.7,
            "usd": 0.000676,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "14.00",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 14.0,
            "expected": 14.0,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 493,
            "output_tokens": 6,
            "latency_ms": 857.3,
            "usd": 0.000676,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "14.00",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 14.0,
            "expected": 14.0,
            "tolerance": 0.05
          },
          "metrics": {
            "input_tokens": 493,
            "output_tokens": 6,
            "latency_ms": 838.4,
            "usd": 0.000676,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 493.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 816.8,
        "usd_sum": 0.002028
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:menu-009",
      "category": "menu_engineering",
      "capability": "menu_critique",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "rubric_judge",
      "context_needs": [
        "rules.menu_engineering",
        "vocab.courses",
        "vocab.component_roles",
        "vocab.match_types",
        "rules.dietary",
        "rules.allergens_eu"
      ],
      "prompt_template_hash": "8883cacb0aca5024",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Menu engineering reference\n\nClassical Boston Matrix quadrants (popularity vs. contribution margin, each split at the menu average):\n- **Star**: high popularity, high margin → feature prominently, do   not change.\n- **Plowhorse**: high popularity, low margin → redesign the dish to   cut food cost or reprice gently.\n- **Puzzle**: low popularity, high margin → reposition on the menu,   rename, or add a compelling description.\n- **Dog**: low popularity, low margin → remove or replace.\n\nContribution margin (CM) = menu price − food cost. Food-cost percentage = food cost / menu price. Industry target for casual dining: 28–35% food cost.\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Dietary labels and rules\n\nCanonical dietary label IDs (lowercase snake_case):\n`vegan`, `vegetarian`, `pescatarian`, `halal`, `kosher`, `gluten_free`, `dairy_free`, `nut_free`, `shellfish_free`, `egg_free`, `soy_free`, `low_fodmap`\n\nHigh-level rules (non-exhaustive):\n- `vegan`: no animal-derived ingredients, including honey, dairy, eggs, gelatine, carmine, isinglass, shellac, animal rennet, or fish sauce.\n- `vegetarian`: no meat, poultry, fish, shellfish, or animal-derived rennet/gelatine. Dairy and eggs permitted.\n- `pescatarian`: no meat or poultry. Fish and shellfish permitted.\n- `halal`: no pork or pork derivatives, no alcohol, meat must be zabihah-slaughtered. Gelatine must be from halal source.\n- `kosher`: no pork or shellfish, no mixing of meat and dairy, meat must be shechita-slaughtered. Separate utensils for meat/dairy.\n- `gluten_free`: no wheat, barley, rye, or derivatives (including malt, soy sauce with wheat, seitan).\n- `dairy_free`: no milk, butter, cream, cheese, whey, casein, ghee, yogurt, or lactose.\n- `nut_free`: no tree nuts (almond, cashew, walnut, hazelnut, pecan, pistachio, macadamia, brazil, pine nut). Peanuts are legumes but are usually treated as a separate allergen.\n- `shellfish_free`: no crustaceans (shrimp, crab, lobster, crayfish) or molluscs (mussel, clam, oyster, scallop, squid, octopus).\n- `low_fodmap`: restrict high-FODMAP ingredients (onion, garlic, wheat, beans, lactose, honey, most stone fruit).\n\nA recipe is compatible with a label if and only if **all** its ingredients comply with the label's rules.\n\n### EU 14 declarable allergens (Regulation 1169/2011 Annex II)\n\n- `cereals_containing_gluten`\n- `crustaceans`\n- `eggs`\n- `fish`\n- `peanuts`\n- `soybeans`\n- `milk`\n- `tree_nuts`\n- `celery`\n- `mustard`\n- `sesame`\n- `sulphur_dioxide`\n- `lupin`\n- `molluscs`\n\nUse exactly these IDs when declaring allergens in OGS documents.\n\n---\n\nCritique the following five-course autumn tasting menu. Identify at least two concrete weaknesses (flow, seasonality, allergen risk, cost-to-price ratio, or structural monotony), suggest one specific, constructive change, and justify each point in 3 short paragraphs. Avoid generic praise.\n\nINPUT:\n{\n  \"menu\": [\n    {\n      \"course\": \"amuse\",\n      \"dish\": \"pumpkin velouté, sage oil\"\n    },\n    {\n      \"course\": \"appetizer\",\n      \"dish\": \"butternut squash risotto, brown butter, amaretti crumble\"\n    },\n    {\n      \"course\": \"fish\",\n      \"dish\": \"pan-seared scallops, pumpkin purée, crispy pancetta\"\n    },\n    {\n      \"course\": \"main\",\n      \"dish\": \"roast pork loin, sweet potato purée, cider jus\"\n    },\n    {\n      \"course\": \"dessert\",\n      \"dish\": \"pumpkin cheesecake, spiced pecan brittle\"\n    }\n  ],\n  \"target_food_cost_pct\": 30,\n  \"dietary_constraints\": \"must offer a viable vegetarian and a viable nut-free path\"\n}\n\nANSWER FORMAT: free text, 3 short paragraphs",
      "expected": {
        "rubric": [
          {
            "id": "diagnosis",
            "weight": 0.4,
            "description": "Identifies concrete, defensible weaknesses (flavor monotony from repeated pumpkin/squash, nut allergen in amaretti and pecans, heavy starches in consecutive courses).",
            "anchors": {
              "0.0": "No specific weakness identified.",
              "0.5": "Identifies one concrete weakness but misses another major one.",
              "1.0": "Identifies at least two of: pumpkin/squash repetition across 4 of 5 courses; nut allergen across amaretti and pecans; consecutive purée/starch heaviness."
            }
          },
          {
            "id": "remediation",
            "weight": 0.3,
            "description": "Offers a specific, actionable, seasonally-plausible change that addresses the identified weakness.",
            "anchors": {
              "0.0": "No change proposed or change is vague.",
              "0.5": "Change is specific but does not address the main weakness.",
              "1.0": "Concrete substitute with named ingredient/technique (e.g. swap scallop course to cured mackerel with pickled apple)."
            }
          },
          {
            "id": "justification",
            "weight": 0.3,
            "description": "Justification is grounded in OGS concepts (match types, structural balance, service flow) rather than marketing language.",
            "anchors": {
              "0.0": "Pure adjective salad.",
              "0.5": "Some reasoning, but appeals to authority not mechanism.",
              "1.0": "Explicitly cites a mechanism: intensity alignment, contrast, allergen cross-contact, food-cost math."
            }
          }
        ],
        "reference_answer": "Key weaknesses: (1) pumpkin/squash appears in 4 of 5 courses, collapsing the menu's flavour arc into a single beta-carotene/sweet-starch note; (2) the amaretti crumble (bitter almond) and the pecan brittle together block every nut-allergic guest from the tasting; (3) velouté, risotto, two purées and a cheesecake stack four viscous/starchy textures in a row with no acid or raw element. A concrete fix is to replace the scallop course with cured mackerel over pickled Granny Smith apple and wasabi-lime gel: it re-introduces sea/fruit/acid, breaks the texture loop, removes one pork-cross-contact dish, and simultaneously opens a pescatarian-friendly option. Justification: on OGS terms this swap adds a `contrast` match (acid/fat), raises `acidity_perception` mid-meal to refresh the palate before the roast, and removes one of the two tree-nut exposure points."
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "The menu’s biggest weakness is structural monotony: four of five courses lean on orange autumn purées or creamy squash/pumpkin formats, so the progression feels repetitive rather than cumulative. Pumpkin velouté, squash risotto, pumpkin purée with scallops, sweet potato purée, then pumpkin cheesecake creates the same soft texture, warm spice register, and visual palette over and over. A specific fix is to replace the fish garnish with something sharper and more textural—say scallops with shaved fennel, apple, and a browned-butter vinaigrette, or even a brassica-based element—so the middle of the menu resets the palate instead of echoing the earlier courses.\n\nThere is also a serious dietary design problem given the stated requirement for a viable vegetarian and a viable nut-free path. The current sequence gives no obvious vegetarian substitute for scallops, pork, or pancetta, and the dessert and risotto introduce tree-nut risk through pecans and amaretti; if amaretti contains almond, that also compromises a nut-free guest before dessert. One constructive change would be to rewrite the appetizer now as a clearly nut-free vegetarian dish—e.g. butternut squash risotto with crispy sage and pumpkin seeds instead of amaretti crumble—and then build matching vegetarian swaps for fish and main from there, because right now the menu reads as if special-diet guests are an afterthought rather than fully accommodated.\n\nCost balance may also be awkward against a 30% target food-cost percentage because scallops are expensive, yet they are paired with relatively low-cost pumpkin purée and followed by pork loin, a much cheaper main, which can make the perceived luxury peak too early and compress value in the back half. Meanwhile, pumpkin, squash, and sweet potato are inexpensive ingredients used repeatedly, so unless pricing is very disciplined, guests may feel they are paying tasting-menu rates for commodity produce presented in similar forms. If only one dish is changed, I would make the fish course a less volatile-cost option or reduce its luxury dependence—either a smaller scallop portion with a more dynamic garnish, or swap to a seasonal fish—because that would improve margin control, help the menu flow, and reduce the sense that one costly plate is carrying an otherwise repetitive autumn theme.",
          "compliant": true,
          "score": 0.97,
          "detail": {
            "per_criterion": [
              {
                "id": "diagnosis",
                "weight": 0.4,
                "score": 1.0,
                "note": "It clearly identifies multiple concrete weaknesses: repeated pumpkin/squash monotony, nut-allergen exposure, lack of a viable vegetarian path, and heavy puree/starch sequencing."
              },
              {
                "id": "remediation",
                "weight": 0.3,
                "score": 1.0,
                "note": "It proposes specific, actionable substitutions such as scallops with shaved fennel and apple and replacing amaretti with crispy sage and pumpkin seeds, both of which directly address the cited issues."
              },
              {
                "id": "justification",
                "weight": 0.3,
                "score": 0.9,
                "note": "The reasoning is mostly mechanism-based, citing palate reset, texture contrast, allergen risk, and food-cost pressure, though it is slightly less explicit on service-flow terminology than the strongest possible answer."
              }
            ],
            "summary": "The candidate follows the requested free-text format in three short paragraphs and avoids generic praise. It diagnoses several defensible menu-design problems and offers concrete fixes with sound culinary and operational justification.",
            "judge_model": "gpt-5.4",
            "judge_prompt_hash": "756db6a487714bca",
            "judge_validated": false
          },
          "metrics": {
            "input_tokens": 1251,
            "output_tokens": 459,
            "latency_ms": 10382.3,
            "usd": 0.006154,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "The menu’s biggest weakness is structural monotony: pumpkin/squash appears in the amuse, appetizer, fish garnish, and dessert, while purées and soft textures dominate four of five courses. That creates palate fatigue and makes the progression feel narrower than a tasting menu should, especially in autumn when mushrooms, brassicas, quince, apple, beet, chestnut, game birds, and bitter leaves are all available. The fish course is also awkwardly redundant because scallops on pumpkin purée after squash risotto reads like a variation on the same sweet-orange, creamy idea rather than a new act.\n\nA second problem is the dietary brief. As written, there is no clear viable vegetarian path because the fish course includes scallops and pancetta and the main is pork; swapping only one plate would still leave a hole in a five-course tasting structure. The nut-free path is also risky because amaretti commonly contains almond and the dessert explicitly uses pecan brittle, creating two tree-nut points in a short menu and increasing cross-contact pressure in pastry. Cost-wise, scallops are the likeliest margin problem against a 30% target food cost unless the menu price is quite high, while risotto and pork loin are comparatively forgiving; that imbalance can work, but only if the premium course feels distinct enough to earn it.\n\nA specific fix: replace the fish course with “roasted hen-of-the-woods, celeriac, apple, sage jus,” and offer it as the default middle course for the vegetarian path while keeping scallops only as an omnivore upgrade if desired. This breaks the pumpkin repetition, introduces firmer roasted texture and earthy-bitter autumn character, and gives the menu a cleaner flow from soup to rice to mushroom to pork to dessert. Then make the appetizer crumble pumpkin-seed based instead of amaretti and change the dessert garnish from pecan brittle to oat streusel or pepita praline if nut-free service is required; those are concrete swaps that preserve an autumn identity while making the menu more operationally inclusive.",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_criterion": [
              {
                "id": "diagnosis",
                "weight": 0.4,
                "score": 1.0,
                "note": "It clearly identifies multiple concrete weaknesses: repeated pumpkin/squash across four courses, soft purée/starch monotony, nut-allergen exposure, lack of a viable vegetarian path, and scallops as a likely food-cost pressure point."
              },
              {
                "id": "remediation",
                "weight": 0.3,
                "score": 1.0,
                "note": "It proposes specific, actionable substitutions with named ingredients and formats, especially replacing the fish course with roasted hen-of-the-woods, celeriac, apple, and sage jus, plus precise nut-free garnish swaps."
              },
              {
                "id": "justification",
                "weight": 0.3,
                "score": 1.0,
                "note": "The reasoning is mechanism-based, citing palate fatigue, menu flow, texture contrast, cross-contact pressure, and margin implications relative to a 30% food-cost target."
              }
            ],
            "summary": "The candidate fully meets the brief with three short paragraphs, avoids generic praise, and diagnoses several defensible menu problems. Its proposed changes are concrete and seasonally plausible, and the justifications are grounded in structural balance, dietary viability, and cost logic rather than vague adjectives.",
            "judge_model": "gpt-5.4",
            "judge_prompt_hash": "756db6a487714bca",
            "judge_validated": false
          },
          "metrics": {
            "input_tokens": 1251,
            "output_tokens": 413,
            "latency_ms": 9676.1,
            "usd": 0.005694,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "The menu leans too hard on one autumn note: pumpkin/squash appears in four of five courses, and three plates are built around soft purées or creamy textures. That creates structural monotony and weakens the progression; by the time the cheesecake arrives, the guest has already had velouté, risotto, pumpkin purée, and sweet potato purée with very little crisp, bitter, raw, or acidic relief. Seasonality is present, but it is expressed narrowly rather than as a broader autumn palette.\n\nThere is also a clear dietary-design problem. As written, there is no convincing vegetarian path through the menu because the fish course uses scallops and pancetta, the main is pork, and the appetizer contains amaretti, which may introduce nuts depending on formulation; dessert explicitly contains pecans, so the nut-free path is compromised as well. Allergen risk is concentrated unnecessarily: likely `milk` in velouté, risotto, brown butter, purées, and cheesecake; `tree_nuts` in pecans and possibly amaretti; `molluscs` in scallops; plus possible `eggs` and `cereals_containing_gluten` in cheesecake or crumble components.\n\nA specific fix: replace the fish course with “roasted hen-of-the-woods, charred radicchio, apple-cider glaze, toasted buckwheat,” and change the dessert garnish from pecan brittle to pumpkin-seed praline. This gives both the regular and vegetarian menus a more distinct, savory-bitter course with real texture and acid, breaking the purée pattern while still feeling autumnal; it also removes `molluscs` and `tree_nuts` from two pressure points, making a nut-free path much more viable. From a cost perspective, scallops are a high-cost ingredient that can strain a 30% target food-cost percentage unless priced aggressively, whereas mushrooms, radicchio, apple, and buckwheat generally give a healthier margin and better menu balance.",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_criterion": [
              {
                "id": "diagnosis",
                "weight": 0.4,
                "score": 1.0,
                "note": "It clearly identifies multiple concrete weaknesses: repeated pumpkin/squash, heavy purée/creamy texture monotony, lack of vegetarian viability, and nut-free/allergen issues."
              },
              {
                "id": "remediation",
                "weight": 0.3,
                "score": 1.0,
                "note": "It proposes specific, actionable substitutions with named ingredients and techniques that directly address monotony, allergen exposure, and cost pressure."
              },
              {
                "id": "justification",
                "weight": 0.3,
                "score": 1.0,
                "note": "The reasoning is mechanism-based, citing structural progression, texture contrast, acid relief, allergen exposure points, and food-cost implications relative to a 30% target."
              }
            ],
            "summary": "The candidate follows the required three-paragraph free-text format and avoids generic praise. It delivers a strong critique with multiple defensible weaknesses, a concrete seasonal revision, and clear OGS-style justification grounded in flow, allergen management, and cost structure.",
            "judge_model": "gpt-5.4",
            "judge_prompt_hash": "756db6a487714bca",
            "judge_validated": false
          },
          "metrics": {
            "input_tokens": 1251,
            "output_tokens": 411,
            "latency_ms": 8043.2,
            "usd": 0.005674,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.99,
      "score_summary": {
        "mean": 0.99,
        "stdev": 0.0173,
        "sem": 0.01,
        "ci95_lo": 0.9704,
        "ci95_hi": 1.0096,
        "n": 3
      },
      "score_if_valid": 0.99,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1251.0,
        "output_tokens_mean": 427.667,
        "latency_ms_mean": 9367.2,
        "usd_sum": 0.017522
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ar-001",
      "category": "vocabulary",
      "capability": "ingredient_classification",
      "difficulty": "easy",
      "cuisine_region": "middle_eastern",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.ingredient_categories"
      ],
      "prompt_template_hash": "94e19d81d80368a1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n---\n\nباستخدام مفردات OGS لتصنيفات المكونات، صنّف الزعتر بكلمة واحدة.\n\nINPUT:\n{\n  \"item\": \"زعتر\"\n}\n\nANSWER FORMAT: one_of:protein,vegetable,fruit,grain,dairy,fat,herb,spice,fungus,legume,nut,seed,sweetener,acid,salt,other",
      "expected": {
        "value": "herb",
        "aliases": [
          "spice"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "herb",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herb",
            "expected": "herb",
            "aliases": [
              "spice"
            ]
          },
          "metrics": {
            "input_tokens": 342,
            "output_tokens": 5,
            "latency_ms": 757.2,
            "usd": 0.000478,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "herb",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herb",
            "expected": "herb",
            "aliases": [
              "spice"
            ]
          },
          "metrics": {
            "input_tokens": 342,
            "output_tokens": 5,
            "latency_ms": 711.6,
            "usd": 0.000478,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "herb",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herb",
            "expected": "herb",
            "aliases": [
              "spice"
            ]
          },
          "metrics": {
            "input_tokens": 342,
            "output_tokens": 5,
            "latency_ms": 754.5,
            "usd": 0.000478,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 342.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 741.1,
        "usd_sum": 0.001434
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ar-002",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "middle_eastern",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "6b39ec376be689ad",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nالمنسف هو الطبق الوطني لأي بلد عربي؟ أجب بكلمة واحدة.\n\nINPUT:\n{\n  \"dish\": \"المنسف\"\n}\n\nANSWER FORMAT: exact:country",
      "expected": {
        "value": "jordan",
        "aliases": [
          "الأردن",
          "jordan (الأردن)"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "الأردن",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "الأردن",
            "expected": "jordan",
            "aliases": [
              "الأردن",
              "jordan (الأردن)"
            ]
          },
          "metrics": {
            "input_tokens": 259,
            "output_tokens": 6,
            "latency_ms": 781.5,
            "usd": 0.000384,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "الأردن",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "الأردن",
            "expected": "jordan",
            "aliases": [
              "الأردن",
              "jordan (الأردن)"
            ]
          },
          "metrics": {
            "input_tokens": 259,
            "output_tokens": 6,
            "latency_ms": 746.8,
            "usd": 0.000384,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "الأردن",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "الأردن",
            "expected": "jordan",
            "aliases": [
              "الأردن",
              "jordan (الأردن)"
            ]
          },
          "metrics": {
            "input_tokens": 259,
            "output_tokens": 6,
            "latency_ms": 747.5,
            "usd": 0.000384,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 259.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 758.6,
        "usd_sum": 0.001152
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-es-001",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "5fe5c7f1f85e3675",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nEl pulpo a feira es un plato tradicional de qué región de España? Responde con una palabra.\n\nINPUT:\n{\n  \"dish\": \"pulpo a feira\"\n}\n\nANSWER FORMAT: exact:region name",
      "expected": {
        "value": "galicia",
        "aliases": [
          "galícia",
          "galiza"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Galicia",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "galicia",
            "expected": "galicia",
            "aliases": [
              "galícia",
              "galiza"
            ]
          },
          "metrics": {
            "input_tokens": 263,
            "output_tokens": 5,
            "latency_ms": 831.5,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Galicia",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "galicia",
            "expected": "galicia",
            "aliases": [
              "galícia",
              "galiza"
            ]
          },
          "metrics": {
            "input_tokens": 263,
            "output_tokens": 5,
            "latency_ms": 535.8,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Galicia",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "galicia",
            "expected": "galicia",
            "aliases": [
              "galícia",
              "galiza"
            ]
          },
          "metrics": {
            "input_tokens": 263,
            "output_tokens": 5,
            "latency_ms": 726.5,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 263.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 697.933,
        "usd_sum": 0.001137
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-es-002",
      "category": "vocabulary",
      "capability": "cooking_method_classification",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "9f03b90711bec40f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nUsando el vocabulario OGS de métodos de cocción, ¿cuál describe mejor la preparación clásica del cochinillo segoviano? Responde con una palabra.\n\nINPUT:\n{\n  \"item\": \"cochinillo segoviano\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "roasted"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 396,
            "output_tokens": 5,
            "latency_ms": 753.1,
            "usd": 0.000545,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 396,
            "output_tokens": 5,
            "latency_ms": 695.9,
            "usd": 0.000545,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 396,
            "output_tokens": 5,
            "latency_ms": 839.0,
            "usd": 0.000545,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 396.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 762.667,
        "usd_sum": 0.001635
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-fr-001",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "8d252ae3fb170e15",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nLa bouillabaisse est une préparation traditionnelle de quelle ville française ? Répondez par un mot.\n\nINPUT:\n{\n  \"dish\": \"bouillabaisse\"\n}\n\nANSWER FORMAT: exact:city name",
      "expected": {
        "value": "marseille"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Marseille",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "marseille",
            "expected": "marseille",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 263,
            "output_tokens": 5,
            "latency_ms": 710.1,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Marseille",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "marseille",
            "expected": "marseille",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 263,
            "output_tokens": 5,
            "latency_ms": 688.0,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Marseille",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "marseille",
            "expected": "marseille",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 263,
            "output_tokens": 5,
            "latency_ms": 688.1,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 263.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 695.4,
        "usd_sum": 0.001137
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-fr-002",
      "category": "sensory",
      "capability": "basic_tastes_profile",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "96a07aeab44536ab",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nPour une sauce beurre blanc classique, donnez le profil sur l'échelle OGS des cinq saveurs [sweet, sour, salty, bitter, umami] sous la forme d'un tableau JSON de cinq nombres entre 0 et 10.\n\nINPUT:\n{\n  \"dish\": \"sauce beurre blanc classique\"\n}\n\nANSWER FORMAT: json:array of 5 numbers",
      "expected": {
        "values": [
          1,
          5,
          4,
          1,
          3
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[0.5, 5.5, 2.5, 0.5, 1.0]",
          "compliant": true,
          "score": 0.4,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 1,
                "predicted": 5.5,
                "expected": 5,
                "score": 0.6667
              },
              {
                "index": 2,
                "predicted": 2.5,
                "expected": 4,
                "score": 0.0
              },
              {
                "index": 3,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 4,
                "predicted": 1.0,
                "expected": 3,
                "score": 0.0
              }
            ],
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 372,
            "output_tokens": 28,
            "latency_ms": 1028.9,
            "usd": 0.000745,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[0.5, 6.5, 3, 0.5, 1]",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 1,
                "predicted": 6.5,
                "expected": 5,
                "score": 0.0
              },
              {
                "index": 2,
                "predicted": 3,
                "expected": 4,
                "score": 0.3333
              },
              {
                "index": 3,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 4,
                "predicted": 1,
                "expected": 3,
                "score": 0.0
              }
            ],
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 372,
            "output_tokens": 24,
            "latency_ms": 949.2,
            "usd": 0.000705,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[1,6,3,0.5,1.5]",
          "compliant": true,
          "score": 0.4667,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 1,
                "expected": 1,
                "score": 1.0
              },
              {
                "index": 1,
                "predicted": 6,
                "expected": 5,
                "score": 0.3333
              },
              {
                "index": 2,
                "predicted": 3,
                "expected": 4,
                "score": 0.3333
              },
              {
                "index": 3,
                "predicted": 0.5,
                "expected": 1,
                "score": 0.6667
              },
              {
                "index": 4,
                "predicted": 1.5,
                "expected": 3,
                "score": 0.0
              }
            ],
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 372,
            "output_tokens": 18,
            "latency_ms": 823.1,
            "usd": 0.000645,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.4,
      "score_summary": {
        "mean": 0.4,
        "stdev": 0.0667,
        "sem": 0.0385,
        "ci95_lo": 0.3245,
        "ci95_hi": 0.4755,
        "n": 3
      },
      "score_if_valid": 0.4,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 372.0,
        "output_tokens_mean": 23.333,
        "latency_ms_mean": 933.733,
        "usd_sum": 0.002095
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-it-001",
      "category": "regional",
      "capability": "region_identification",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "c3aae88ebfcbe08b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nI pici sono una pasta fresca tipica di quale regione italiana? Rispondi con una parola.\n\nINPUT:\n{\n  \"dish\": \"pici cacio e pepe\"\n}\n\nANSWER FORMAT: exact:regione",
      "expected": {
        "value": "toscana"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Toscana",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "toscana",
            "expected": "toscana",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 265,
            "output_tokens": 6,
            "latency_ms": 766.8,
            "usd": 0.000391,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Toscana",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "toscana",
            "expected": "toscana",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 265,
            "output_tokens": 6,
            "latency_ms": 765.9,
            "usd": 0.000391,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Toscana",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "toscana",
            "expected": "toscana",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 265,
            "output_tokens": 6,
            "latency_ms": 786.2,
            "usd": 0.000391,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 265.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 772.967,
        "usd_sum": 0.001173
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-it-002",
      "category": "pairing",
      "capability": "beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.beverage_types",
        "vocab.match_types"
      ],
      "prompt_template_hash": "30919f613b1fbcd6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nScegli l'abbinamento più classico per l'ossobuco alla milanese fra: barolo, riesling, prosecco, lambrusco. Rispondi con una parola.\n\nINPUT:\n{\n  \"dish\": \"ossobuco alla milanese\"\n}\n\nANSWER FORMAT: exact:one of the given options",
      "expected": {
        "value": "barolo"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "barolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "barolo",
            "expected": "barolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 5,
            "latency_ms": 608.1,
            "usd": 0.00046,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "barolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "barolo",
            "expected": "barolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 5,
            "latency_ms": 644.1,
            "usd": 0.00046,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "barolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "barolo",
            "expected": "barolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 5,
            "latency_ms": 1257.5,
            "usd": 0.00046,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 328.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 836.567,
        "usd_sum": 0.00138
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ja-001",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "ff1f43723314705c",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nOGSの香りファミリー一覧から、柚子（ゆず）の主たる香りファミリーを一語で答えてください。\n\nINPUT:\n{\n  \"item\": \"柚子\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "fruit"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 332,
            "output_tokens": 4,
            "latency_ms": 689.0,
            "usd": 0.000455,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 332,
            "output_tokens": 4,
            "latency_ms": 578.5,
            "usd": 0.000455,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 332,
            "output_tokens": 4,
            "latency_ms": 674.1,
            "usd": 0.000455,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 332.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 647.2,
        "usd_sum": 0.001365
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:ml-ja-002",
      "category": "pairing",
      "capability": "beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.beverage_types",
        "vocab.match_types"
      ],
      "prompt_template_hash": "f68d4882814e0ecf",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\n握り寿司（赤身・白身・光物）のコース全体に最も安全に合う飲料のタイプを、次から一つ選んでください: sake, red_wine, stout_beer, bourbon.\n\nINPUT:\n{\n  \"dish\": \"握り寿司コース\"\n}\n\nANSWER FORMAT: exact:one of the given options",
      "expected": {
        "value": "sake"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "sake",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sake",
            "expected": "sake",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 338,
            "output_tokens": 5,
            "latency_ms": 738.2,
            "usd": 0.000473,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "sake",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sake",
            "expected": "sake",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 338,
            "output_tokens": 5,
            "latency_ms": 683.9,
            "usd": 0.000473,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "sake",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sake",
            "expected": "sake",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 338,
            "output_tokens": 5,
            "latency_ms": 651.5,
            "usd": 0.000473,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 338.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 691.2,
        "usd_sum": 0.001419
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-001",
      "category": "nutrition",
      "capability": "energy_calculation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "94fadc6f5c8c5968",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA portion contains 40 g carbohydrate, 20 g protein, 15 g fat. Return total energy in kcal as a single integer, using the EU Annex XIV macro factors.\n\nINPUT:\n{\n  \"carb_g\": 40,\n  \"protein_g\": 20,\n  \"fat_g\": 15\n}\n\nANSWER FORMAT: numeric:integer kcal",
      "expected": {
        "value": 375,
        "tolerance": 5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "375",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 375.0,
            "expected": 375.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 815,
            "output_tokens": 4,
            "latency_ms": 781.2,
            "usd": 0.001059,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "375",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 375.0,
            "expected": 375.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 815,
            "output_tokens": 4,
            "latency_ms": 774.2,
            "usd": 0.001059,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "375",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 375.0,
            "expected": 375.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 815,
            "output_tokens": 4,
            "latency_ms": 561.8,
            "usd": 0.001059,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 815.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 705.733,
        "usd_sum": 0.003177
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-002",
      "category": "nutrition",
      "capability": "energy_calculation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "4b141daaadd9af57",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA 150 mL glass of red wine contains approximately 14 g alcohol, 4 g carbohydrate, no protein, no fat. Return energy in kcal as a single integer.\n\nINPUT:\n{\n  \"carb_g\": 4,\n  \"protein_g\": 0,\n  \"fat_g\": 0,\n  \"alcohol_g\": 14\n}\n\nANSWER FORMAT: numeric:integer kcal",
      "expected": {
        "value": 114,
        "tolerance": 5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "114",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 114.0,
            "expected": 114.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 824,
            "output_tokens": 4,
            "latency_ms": 712.7,
            "usd": 0.00107,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "114",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 114.0,
            "expected": 114.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 824,
            "output_tokens": 4,
            "latency_ms": 723.4,
            "usd": 0.00107,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "114",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 114.0,
            "expected": 114.0,
            "tolerance": 5.0
          },
          "metrics": {
            "input_tokens": 824,
            "output_tokens": 4,
            "latency_ms": 784.0,
            "usd": 0.00107,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 824.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 740.033,
        "usd_sum": 0.00321
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-003",
      "category": "nutrition",
      "capability": "per_100g_estimate",
      "difficulty": "medium",
      "cuisine_region": "mediterranean",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "518ee70ef096f9ec",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA simple olive-oil vinaigrette is 75 g olive oil + 25 g red wine vinegar (vinegar ≈ 20 kcal/100 g). Return energy per 100 g of finished dressing as a single integer kcal.\n\nINPUT:\n{\n  \"olive_oil_g\": 75,\n  \"vinegar_g\": 25\n}\n\nANSWER FORMAT: numeric:integer kcal per 100 g",
      "expected": {
        "value": 668,
        "tolerance": 15
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "668",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 668.0,
            "expected": 668.0,
            "tolerance": 15.0
          },
          "metrics": {
            "input_tokens": 825,
            "output_tokens": 4,
            "latency_ms": 746.9,
            "usd": 0.001071,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "668",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 668.0,
            "expected": 668.0,
            "tolerance": 15.0
          },
          "metrics": {
            "input_tokens": 825,
            "output_tokens": 4,
            "latency_ms": 672.5,
            "usd": 0.001071,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "668",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 668.0,
            "expected": 668.0,
            "tolerance": 15.0
          },
          "metrics": {
            "input_tokens": 825,
            "output_tokens": 4,
            "latency_ms": 1002.6,
            "usd": 0.001071,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 825.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 807.333,
        "usd_sum": 0.003213
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-004",
      "category": "nutrition",
      "capability": "protein_estimate",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "42947503c82b308d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nEstimate the protein (g) in a 180 g cooked portion of skinless chicken-breast filet assuming 30% cooking mass loss. Respond with one integer.\n\nINPUT:\n{\n  \"raw_mass_g_target_cooked\": 180,\n  \"cooking_loss\": 0.3\n}\n\nANSWER FORMAT: numeric:integer g",
      "expected": {
        "value": 59,
        "tolerance": 4
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "59",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 59.0,
            "expected": 59.0,
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 811,
            "output_tokens": 4,
            "latency_ms": 754.9,
            "usd": 0.001054,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "59",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 59.0,
            "expected": 59.0,
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 811,
            "output_tokens": 4,
            "latency_ms": 628.8,
            "usd": 0.001054,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "59",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 59.0,
            "expected": 59.0,
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 811,
            "output_tokens": 4,
            "latency_ms": 861.9,
            "usd": 0.001054,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 811.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 748.533,
        "usd_sum": 0.003162
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-005",
      "category": "nutrition",
      "capability": "macro_vector",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "375ca5d44d346632",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA portion contains 85 g dry pasta (75 C / 13 P / 1.5 F per 100 g), 10 g butter (0.1 C / 0.9 P / 81 F per 100 g) and 15 g parmesan (3.2 C / 35 P / 29 F per 100 g). Return [carb_g, protein_g, fat_g, kcal] for the portion as a JSON array of four numbers.\n\nINPUT:\n{\n  \"pasta_g\": 85,\n  \"butter_g\": 10,\n  \"parmesan_g\": 15\n}\n\nANSWER FORMAT: json:array [carb, protein, fat, kcal]",
      "expected": {
        "values": [
          64.3,
          16.3,
          13.7,
          446
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[64.03, 16.42, 13.68, 438]",
          "compliant": true,
          "score": 0.7158,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 64.03,
                "expected": 64.3,
                "score": 0.91
              },
              {
                "index": 1,
                "predicted": 16.42,
                "expected": 16.3,
                "score": 0.96
              },
              {
                "index": 2,
                "predicted": 13.68,
                "expected": 13.7,
                "score": 0.9933
              },
              {
                "index": 3,
                "predicted": 438,
                "expected": 446,
                "score": 0.0
              }
            ],
            "tolerance": 3.0
          },
          "metrics": {
            "input_tokens": 889,
            "output_tokens": 21,
            "latency_ms": 1275.4,
            "usd": 0.001321,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[64.23, 16.61, 12.38, 432]",
          "compliant": true,
          "score": 0.6083,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 64.23,
                "expected": 64.3,
                "score": 0.9767
              },
              {
                "index": 1,
                "predicted": 16.61,
                "expected": 16.3,
                "score": 0.8967
              },
              {
                "index": 2,
                "predicted": 12.38,
                "expected": 13.7,
                "score": 0.56
              },
              {
                "index": 3,
                "predicted": 432,
                "expected": 446,
                "score": 0.0
              }
            ],
            "tolerance": 3.0
          },
          "metrics": {
            "input_tokens": 889,
            "output_tokens": 21,
            "latency_ms": 925.9,
            "usd": 0.001321,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[64.23, 16.61, 13.8, 443]",
          "compliant": true,
          "score": 0.71,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 64.23,
                "expected": 64.3,
                "score": 0.9767
              },
              {
                "index": 1,
                "predicted": 16.61,
                "expected": 16.3,
                "score": 0.8967
              },
              {
                "index": 2,
                "predicted": 13.8,
                "expected": 13.7,
                "score": 0.9667
              },
              {
                "index": 3,
                "predicted": 443,
                "expected": 446,
                "score": 0.0
              }
            ],
            "tolerance": 3.0
          },
          "metrics": {
            "input_tokens": 889,
            "output_tokens": 21,
            "latency_ms": 917.1,
            "usd": 0.001321,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.678,
      "score_summary": {
        "mean": 0.678,
        "stdev": 0.0605,
        "sem": 0.0349,
        "ci95_lo": 0.6096,
        "ci95_hi": 0.7465,
        "n": 3
      },
      "score_if_valid": 0.678,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 889.0,
        "output_tokens_mean": 21.0,
        "latency_ms_mean": 1039.467,
        "usd_sum": 0.003963
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-006",
      "category": "nutrition",
      "capability": "energy_calculation",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "1ba6175ed99ad486",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA mocktail uses 180 mL fresh orange juice (approx 9 g carbohydrate per 100 g) plus 12 g granulated sugar. Return kcal as a single integer.\n\nINPUT:\n{\n  \"oj_ml\": 180,\n  \"sugar_g\": 12\n}\n\nANSWER FORMAT: numeric:integer kcal",
      "expected": {
        "value": 111,
        "tolerance": 6
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "112",
          "compliant": true,
          "score": 0.8333,
          "detail": {
            "predicted": 112.0,
            "expected": 111.0,
            "tolerance": 6.0
          },
          "metrics": {
            "input_tokens": 808,
            "output_tokens": 4,
            "latency_ms": 717.6,
            "usd": 0.00105,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "112",
          "compliant": true,
          "score": 0.8333,
          "detail": {
            "predicted": 112.0,
            "expected": 111.0,
            "tolerance": 6.0
          },
          "metrics": {
            "input_tokens": 808,
            "output_tokens": 4,
            "latency_ms": 521.3,
            "usd": 0.00105,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "111",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 111.0,
            "expected": 111.0,
            "tolerance": 6.0
          },
          "metrics": {
            "input_tokens": 808,
            "output_tokens": 4,
            "latency_ms": 662.7,
            "usd": 0.00105,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.8889,
      "score_summary": {
        "mean": 0.8889,
        "stdev": 0.0962,
        "sem": 0.0556,
        "ci95_lo": 0.78,
        "ci95_hi": 0.9978,
        "n": 3
      },
      "score_if_valid": 0.8889,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 808.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 633.867,
        "usd_sum": 0.00315
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-007",
      "category": "nutrition",
      "capability": "fibre_rule",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "b9e5625ed0828c94",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nWhat kcal/g factor does the OGS nutrition reference assign to dietary fibre? Respond with a single number.\n\nANSWER FORMAT: numeric:decimal kcal/g",
      "expected": {
        "value": 2,
        "tolerance": 0.1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.1
          },
          "metrics": {
            "input_tokens": 772,
            "output_tokens": 4,
            "latency_ms": 733.8,
            "usd": 0.001005,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.1
          },
          "metrics": {
            "input_tokens": 772,
            "output_tokens": 4,
            "latency_ms": 675.2,
            "usd": 0.001005,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.1
          },
          "metrics": {
            "input_tokens": 772,
            "output_tokens": 4,
            "latency_ms": 500.0,
            "usd": 0.001005,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 772.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 636.333,
        "usd_sum": 0.003015
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-008",
      "category": "nutrition",
      "capability": "kj_conversion",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "95ef839d58e96bbc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nConvert 250 kcal to kJ using the canonical OGS/EU conversion factor. Respond with a single integer.\n\nINPUT:\n{\n  \"kcal\": 250\n}\n\nANSWER FORMAT: numeric:integer kJ",
      "expected": {
        "value": 1046,
        "tolerance": 10
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "1046",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 1046.0,
            "expected": 1046.0,
            "tolerance": 10.0
          },
          "metrics": {
            "input_tokens": 786,
            "output_tokens": 5,
            "latency_ms": 710.9,
            "usd": 0.001032,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "1046",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 1046.0,
            "expected": 1046.0,
            "tolerance": 10.0
          },
          "metrics": {
            "input_tokens": 786,
            "output_tokens": 5,
            "latency_ms": 710.4,
            "usd": 0.001032,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "1046",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 1046.0,
            "expected": 1046.0,
            "tolerance": 10.0
          },
          "metrics": {
            "input_tokens": 786,
            "output_tokens": 5,
            "latency_ms": 666.5,
            "usd": 0.001032,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 786.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 695.933,
        "usd_sum": 0.003096
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-009",
      "category": "nutrition",
      "capability": "macro_vector",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "641416c8b33c6dbc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nA simple Alfredo sauce is 100 g butter + 150 g heavy cream (36% fat: 2.8 C / 2 P / 37 F) + 60 g parmesan (3.2 C / 35 P / 29 F), reduced to 250 g finished weight. Return [carb_per100g, protein_per100g, fat_per100g, kcal_per100g] as a JSON array of four numbers.\n\nINPUT:\n{\n  \"butter_g\": 100,\n  \"cream_g\": 150,\n  \"parmesan_g\": 60,\n  \"finished_g\": 250\n}\n\nANSWER FORMAT: json:array [carb, protein, fat, kcal] per 100 g",
      "expected": {
        "values": [
          2.5,
          9.6,
          55.2,
          545
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[1.8,4.7,53.3,514]",
          "compliant": true,
          "score": 0.3375,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 1.8,
                "expected": 2.5,
                "score": 0.825
              },
              {
                "index": 1,
                "predicted": 4.7,
                "expected": 9.6,
                "score": 0.0
              },
              {
                "index": 2,
                "predicted": 53.3,
                "expected": 55.2,
                "score": 0.525
              },
              {
                "index": 3,
                "predicted": 514,
                "expected": 545,
                "score": 0.0
              }
            ],
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 894,
            "output_tokens": 18,
            "latency_ms": 980.5,
            "usd": 0.001298,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[2.0, 9.3, 54.0, 548]",
          "compliant": true,
          "score": 0.6875,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 2.0,
                "expected": 2.5,
                "score": 0.875
              },
              {
                "index": 1,
                "predicted": 9.3,
                "expected": 9.6,
                "score": 0.925
              },
              {
                "index": 2,
                "predicted": 54.0,
                "expected": 55.2,
                "score": 0.7
              },
              {
                "index": 3,
                "predicted": 548,
                "expected": 545,
                "score": 0.25
              }
            ],
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 894,
            "output_tokens": 21,
            "latency_ms": 922.7,
            "usd": 0.001328,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[1.6,4.8,50.4,487]",
          "compliant": true,
          "score": 0.1938,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 1.6,
                "expected": 2.5,
                "score": 0.775
              },
              {
                "index": 1,
                "predicted": 4.8,
                "expected": 9.6,
                "score": 0.0
              },
              {
                "index": 2,
                "predicted": 50.4,
                "expected": 55.2,
                "score": 0.0
              },
              {
                "index": 3,
                "predicted": 487,
                "expected": 545,
                "score": 0.0
              }
            ],
            "tolerance": 4.0
          },
          "metrics": {
            "input_tokens": 894,
            "output_tokens": 18,
            "latency_ms": 982.2,
            "usd": 0.001298,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.4063,
      "score_summary": {
        "mean": 0.4063,
        "stdev": 0.2539,
        "sem": 0.1466,
        "ci95_lo": 0.1189,
        "ci95_hi": 0.6936,
        "n": 3
      },
      "score_if_valid": 0.4063,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 894.0,
        "output_tokens_mean": 19.0,
        "latency_ms_mean": 961.8,
        "usd_sum": 0.003924
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:nut-010",
      "category": "nutrition",
      "capability": "numeric_compare",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.nutrition"
      ],
      "prompt_template_hash": "fd7edc20921469ff",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS nutrition reference (per-100 g macro densities)\n\nMacronutrient energy values (EU Regulation 1169/2011 Annex XIV):\n- carbohydrate: **4 kcal/g** (17 kJ/g)\n- protein: **4 kcal/g** (17 kJ/g)\n- fat: **9 kcal/g** (37 kJ/g)\n- alcohol: **7 kcal/g** (29 kJ/g)\n- fibre: **2 kcal/g** (8 kJ/g)\n\nExpress nutrition values per 100 g of the finished component unless explicitly stated otherwise. Round kcal to the nearest integer.\n\nCommon ingredient macros (per 100 g, approximate, for reasoning only):\n- olive oil: 0 C / 0 P / 100 F → 884 kcal\n- butter: 0.1 C / 0.9 P / 81 F → 717 kcal\n- chicken breast (raw, skinless): 0 C / 23 P / 1.2 F → 110 kcal\n- salmon (raw, Atlantic): 0 C / 20 P / 13 F → 208 kcal\n- dry pasta: 75 C / 13 P / 1.5 F → 371 kcal\n- cooked white rice: 28 C / 2.7 P / 0.3 F → 130 kcal\n- granulated sugar: 100 C / 0 P / 0 F → 387 kcal\n- all-purpose flour: 76 C / 10 P / 1 F → 364 kcal\n- whole milk: 4.8 C / 3.4 P / 3.3 F → 61 kcal\n- heavy cream (36%): 2.8 C / 2 P / 37 F → 345 kcal\n- eggs (whole, raw): 1.1 C / 12.6 P / 9.5 F → 143 kcal\n- parmesan: 3.2 C / 35 P / 29 F → 420 kcal\n- tomato (raw): 3.9 C / 0.9 P / 0.2 F → 18 kcal\n- onion (raw): 9 C / 1.1 P / 0.1 F → 40 kcal\n\nWhen summing nutrition across a recipe, scale each ingredient by its mass in grams, then divide by total finished mass to get per-100 g values. Assume no mass loss from evaporation unless the task says otherwise.\n\n---\n\nFrom the set {olive oil, butter, granulated sugar, dry pasta}, return the ingredient with the highest kcal per gram. Respond with one short name.\n\nINPUT:\n{\n  \"set\": [\n    \"olive oil\",\n    \"butter\",\n    \"granulated sugar\",\n    \"dry pasta\"\n  ]\n}\n\nANSWER FORMAT: exact:string",
      "expected": {
        "value": "olive oil",
        "aliases": [
          "olive-oil"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "olive oil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "olive oil",
            "expected": "olive oil",
            "aliases": [
              "olive-oil"
            ]
          },
          "metrics": {
            "input_tokens": 812,
            "output_tokens": 5,
            "latency_ms": 638.7,
            "usd": 0.001065,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "olive oil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "olive oil",
            "expected": "olive oil",
            "aliases": [
              "olive-oil"
            ]
          },
          "metrics": {
            "input_tokens": 812,
            "output_tokens": 5,
            "latency_ms": 1135.4,
            "usd": 0.001065,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "olive oil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "olive oil",
            "expected": "olive oil",
            "aliases": [
              "olive-oil"
            ]
          },
          "metrics": {
            "input_tokens": 812,
            "output_tokens": 5,
            "latency_ms": 498.4,
            "usd": 0.001065,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 812.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 757.5,
        "usd_sum": 0.003195
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-001",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "easy",
      "cuisine_region": "american",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "3fa6a67d5bfe7f1a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage option for the dish below from the candidates. Respond with just the candidate id (e.g. 'a').\n\nINPUT:\n{\n  \"dish\": \"60-day dry-aged ribeye, medium-rare, served with peppercorn jus\",\n  \"candidates\": {\n    \"a\": \"young Barolo DOCG (high tannin, high acidity, full body)\",\n    \"b\": \"off-dry German Kabinett Riesling\",\n    \"c\": \"unoaked Muscadet sur lie\",\n    \"d\": \"sweet Pedro Ximenez sherry\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 377,
            "output_tokens": 4,
            "latency_ms": 686.6,
            "usd": 0.000511,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 377,
            "output_tokens": 4,
            "latency_ms": 657.4,
            "usd": 0.000511,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 377,
            "output_tokens": 4,
            "latency_ms": 639.5,
            "usd": 0.000511,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 377.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 661.167,
        "usd_sum": 0.001533
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-002",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "4b85899cd6206acb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below from the candidates. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"fresh Kumamoto oysters on the half shell, lemon and mignonette\",\n  \"candidates\": {\n    \"a\": \"oaked California Chardonnay (14.5% ABV, buttery)\",\n    \"b\": \"Muscadet Sevre-et-Maine sur lie\",\n    \"c\": \"Amarone della Valpolicella\",\n    \"d\": \"bourbon old fashioned\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "b",
        "ranking": [
          "b",
          "a",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 364,
            "output_tokens": 4,
            "latency_ms": 754.1,
            "usd": 0.000495,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 364,
            "output_tokens": 4,
            "latency_ms": 636.3,
            "usd": 0.000495,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 364,
            "output_tokens": 4,
            "latency_ms": 810.1,
            "usd": 0.000495,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 364.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 733.5,
        "usd_sum": 0.001485
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-003",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "mexican",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "41ed3ee840823502",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"classic Chengdu-style mapo tofu (heavy doubanjiang, Sichuan peppercorn, ground pork, high chilli heat)\",\n  \"candidates\": {\n    \"a\": \"15% ABV Zinfandel\",\n    \"b\": \"off-dry Riesling Kabinett (~10% ABV)\",\n    \"c\": \"Islay single malt Scotch (Lagavulin 16)\",\n    \"d\": \"Brunello di Montalcino Riserva\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "b",
        "ranking": [
          "b",
          "d",
          "a",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 377,
            "output_tokens": 4,
            "latency_ms": 687.0,
            "usd": 0.000511,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 377,
            "output_tokens": 4,
            "latency_ms": 740.7,
            "usd": 0.000511,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 377,
            "output_tokens": 4,
            "latency_ms": 622.0,
            "usd": 0.000511,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 377.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 683.233,
        "usd_sum": 0.001533
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-004",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "82a4b41d0e3d5668",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"warm goat cheese crouton on mesclun with walnut vinaigrette\",\n  \"candidates\": {\n    \"a\": \"Sancerre (Sauvignon Blanc, Loire)\",\n    \"b\": \"Napa Cabernet Sauvignon Reserve\",\n    \"c\": \"Vintage Port\",\n    \"d\": \"lagered pilsner\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "b",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 702.5,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 741.1,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 595.3,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 348.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 679.633,
        "usd_sum": 0.001425
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-005",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "441705237fe65e54",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"warm dark chocolate fondant (70% cacao), molten centre, vanilla ice cream\",\n  \"candidates\": {\n    \"a\": \"dry Sancerre Sauvignon Blanc\",\n    \"b\": \"vintage Port\",\n    \"c\": \"bone-dry Brut Nature Champagne\",\n    \"d\": \"Muscadet Sevre-et-Maine\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "b",
        "ranking": [
          "b",
          "c",
          "a",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 351,
            "output_tokens": 4,
            "latency_ms": 1661.0,
            "usd": 0.000479,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 351,
            "output_tokens": 4,
            "latency_ms": 843.9,
            "usd": 0.000479,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "b",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "b",
            "expected_best": "b",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 351,
            "output_tokens": 4,
            "latency_ms": 801.4,
            "usd": 0.000479,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 351.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1102.1,
        "usd_sum": 0.001437
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-006",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "b195ee1dfe7968ab",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage to pair across a premium edomae sushi omakase. Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"premium edomae sushi omakase (akami, toro, kohada, ikura, uni, anago)\",\n  \"candidates\": {\n    \"a\": \"Junmai Daiginjo sake (well-chilled)\",\n    \"b\": \"heavily oaked California Chardonnay\",\n    \"c\": \"Barolo Riserva\",\n    \"d\": \"Islay single malt\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 4,
            "latency_ms": 594.0,
            "usd": 0.000501,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 4,
            "latency_ms": 632.1,
            "usd": 0.000501,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 4,
            "latency_ms": 694.4,
            "usd": 0.000501,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 369.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 640.167,
        "usd_sum": 0.001503
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-007",
      "category": "pairing",
      "capability": "regional_pairing",
      "difficulty": "hard",
      "cuisine_region": "spanish",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "81bf14f26f4fa761",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for a classic paella valenciana (rabbit, chicken, garrofó beans, saffron). Respond with just the candidate id.\n\nINPUT:\n{\n  \"dish\": \"authentic paella valenciana\",\n  \"candidates\": {\n    \"a\": \"young Rioja crianza (Tempranillo blend, light oak)\",\n    \"b\": \"Amarone della Valpolicella Riserva\",\n    \"c\": \"Islay peated single malt\",\n    \"d\": \"off-dry Moscato d'Asti\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 746.0,
            "usd": 0.000503,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 788.5,
            "usd": 0.000503,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 5025.5,
            "usd": 0.000503,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 370.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 2186.667,
        "usd_sum": 0.001509
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-008",
      "category": "pairing",
      "capability": "pairing_grade_reasoning",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "51ad14586a063779",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nA pairing's overall_score is computed as 68. Using the OGS grade scale, return the letter grade (ogs-match.md §5). Respond with just the grade token.\n\nINPUT:\n{\n  \"overall_score\": 68\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "D"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "D",
            "expected": "D",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 383,
            "output_tokens": 4,
            "latency_ms": 945.6,
            "usd": 0.000519,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "D",
            "expected": "D",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 383,
            "output_tokens": 4,
            "latency_ms": 738.6,
            "usd": 0.000519,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "D",
            "expected": "D",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 383,
            "output_tokens": 4,
            "latency_ms": 1081.1,
            "usd": 0.000519,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 383.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 921.767,
        "usd_sum": 0.001557
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-009",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "d7ade64ce48bfbe4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below from the candidates. Respond with just the candidate id (e.g. 'a').\n\nINPUT:\n{\n  \"dish\": \"classic Neapolitan pizza Margherita (tomato, buffalo mozzarella, basil)\",\n  \"candidates\": {\n    \"a\": \"bright Chianti Classico (Sangiovese, high acid, medium tannin)\",\n    \"b\": \"heavily oaked California Chardonnay\",\n    \"c\": \"sweet ice wine\",\n    \"d\": \"Islay peated whisky\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "d",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 629.3,
            "usd": 0.000503,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 656.0,
            "usd": 0.000503,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 4,
            "latency_ms": 620.5,
            "usd": 0.000503,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 370.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 635.267,
        "usd_sum": 0.001509
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-010",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "chinese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "637717fdf3577204",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below.\n\nINPUT:\n{\n  \"dish\": \"mapo tofu with doubanjiang and Sichuan peppercorn, medium-hot\",\n  \"candidates\": {\n    \"a\": \"off-dry German Riesling Spätlese\",\n    \"b\": \"high-alcohol (15%) Zinfandel\",\n    \"c\": \"bone-dry Fino sherry\",\n    \"d\": \"straight rye whiskey\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "d",
          "b"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 349,
            "output_tokens": 4,
            "latency_ms": 942.0,
            "usd": 0.000476,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 349,
            "output_tokens": 4,
            "latency_ms": 623.9,
            "usd": 0.000476,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 349,
            "output_tokens": 4,
            "latency_ms": 764.8,
            "usd": 0.000476,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 349.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 776.9,
        "usd_sum": 0.001428
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-011",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "indian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "2b655f8b1ea64df8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish below.\n\nINPUT:\n{\n  \"dish\": \"butter chicken (murgh makhani) in rich tomato-cream sauce with fenugreek\",\n  \"candidates\": {\n    \"a\": \"off-dry Gewurztraminer\",\n    \"b\": \"young tannic Cabernet Sauvignon\",\n    \"c\": \"unoaked Muscadet\",\n    \"d\": \"Islay single malt\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 345,
            "output_tokens": 4,
            "latency_ms": 853.1,
            "usd": 0.000471,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 345,
            "output_tokens": 4,
            "latency_ms": 694.2,
            "usd": 0.000471,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 345,
            "output_tokens": 4,
            "latency_ms": 680.7,
            "usd": 0.000471,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 345.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 742.667,
        "usd_sum": 0.001413
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-012",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "ethiopian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "3fe344c62017694c",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for the dish.\n\nINPUT:\n{\n  \"dish\": \"doro wat: chicken stewed in berbere-spiced ghee, served with injera\",\n  \"candidates\": {\n    \"a\": \"juicy, low-tannin Beaujolais (Gamay) served slightly chilled\",\n    \"b\": \"high-alcohol Amarone\",\n    \"c\": \"bone-dry sparkling Prosecco Brut\",\n    \"d\": \"heavily peated Islay whisky\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 610.2,
            "usd": 0.00049,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 714.8,
            "usd": 0.00049,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 705.2,
            "usd": 0.00049,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 360.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 676.733,
        "usd_sum": 0.00147
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-013",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "argentinian",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "4d30c6e439c9c85f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage.\n\nINPUT:\n{\n  \"dish\": \"grilled ribeye over wood embers, served with chimichurri\",\n  \"candidates\": {\n    \"a\": \"old-vine Mendoza Malbec, medium-plus tannin, black fruit\",\n    \"b\": \"unoaked Chablis\",\n    \"c\": \"off-dry Moscato\",\n    \"d\": \"dry Manzanilla sherry\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "b",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 4,
            "latency_ms": 657.6,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 4,
            "latency_ms": 802.4,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 4,
            "latency_ms": 730.2,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 346.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 730.067,
        "usd_sum": 0.001416
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-014",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "848f37f91777acfb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for otoro nigiri topped with a brush of soy.\n\nINPUT:\n{\n  \"dish\": \"bluefin tuna otoro nigiri, brushed soy\",\n  \"candidates\": {\n    \"a\": \"junmai daiginjo sake, chilled\",\n    \"b\": \"young Chinon Cabernet Franc\",\n    \"c\": \"dry Riesling Kabinett\",\n    \"d\": \"sweet Sauternes\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 345,
            "output_tokens": 4,
            "latency_ms": 745.7,
            "usd": 0.000471,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 345,
            "output_tokens": 4,
            "latency_ms": 715.9,
            "usd": 0.000471,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 345,
            "output_tokens": 4,
            "latency_ms": 587.6,
            "usd": 0.000471,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 345.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 683.067,
        "usd_sum": 0.001413
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-015",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "hard",
      "cuisine_region": "moroccan",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "f5ecf9dcd59f0d0a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage.\n\nINPUT:\n{\n  \"dish\": \"slow-cooked lamb shoulder tagine with preserved lemon and green olives\",\n  \"candidates\": {\n    \"a\": \"Southern Rhône Grenache-Syrah blend (Gigondas), moderate tannin\",\n    \"b\": \"unoaked Muscadet\",\n    \"c\": \"Champagne Blanc de Blancs\",\n    \"d\": \"young high-tannin Barolo\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "c",
          "b"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 4,
            "latency_ms": 735.4,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 4,
            "latency_ms": 628.6,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 4,
            "latency_ms": 778.4,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 346.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 714.133,
        "usd_sum": 0.001416
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-016",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "vietnamese",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "3dd7c0170877ac9b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for a bowl of Vietnamese beef pho.\n\nINPUT:\n{\n  \"dish\": \"pho bo: beef broth with star anise, cinnamon, rice noodles, rare beef, basil, lime\",\n  \"candidates\": {\n    \"a\": \"cold Vietnamese lager-style beer (e.g. Saigon or 333)\",\n    \"b\": \"tannic Cabernet Sauvignon\",\n    \"c\": \"heavily oaked Chardonnay\",\n    \"d\": \"sweet Tokaji Aszú 5 puttonyos\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "b",
          "c",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 365,
            "output_tokens": 4,
            "latency_ms": 805.6,
            "usd": 0.000496,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 365,
            "output_tokens": 4,
            "latency_ms": 1784.6,
            "usd": 0.000496,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 365,
            "output_tokens": 4,
            "latency_ms": 665.3,
            "usd": 0.000496,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 365.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1085.167,
        "usd_sum": 0.001488
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-017",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "8f60336584f33baf",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for 24-month aged Comté served on its own.\n\nINPUT:\n{\n  \"dish\": \"24-month aged Comté, served at room temperature\",\n  \"candidates\": {\n    \"a\": \"Jura Vin Jaune (Savagnin, oxidative, nutty)\",\n    \"b\": \"young oaky Cabernet Sauvignon\",\n    \"c\": \"dry Muscadet\",\n    \"d\": \"Sauternes\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "b",
          "c"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 739.0,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 864.0,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 640.6,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 348.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 747.867,
        "usd_sum": 0.001425
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-018",
      "category": "pairing",
      "capability": "regional_pairing",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "9209873bc05779b7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage for a fragrant, moderately spicy Thai green curry with chicken and Thai basil.\n\nINPUT:\n{\n  \"dish\": \"Thai green curry with chicken, Thai basil, kaffir lime\",\n  \"candidates\": {\n    \"a\": \"off-dry New Zealand Sauvignon Blanc with residual sugar\",\n    \"b\": \"heavy Port wine\",\n    \"c\": \"young tannic Bordeaux\",\n    \"d\": \"Islay peated whisky\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "c",
          "b",
          "d"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 349,
            "output_tokens": 4,
            "latency_ms": 629.1,
            "usd": 0.000476,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 349,
            "output_tokens": 4,
            "latency_ms": 749.2,
            "usd": 0.000476,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 349,
            "output_tokens": 4,
            "latency_ms": 684.3,
            "usd": 0.000476,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 349.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 687.533,
        "usd_sum": 0.001428
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-019",
      "category": "pairing",
      "capability": "best_beverage_selection",
      "difficulty": "medium",
      "cuisine_region": "mexican",
      "scorer": "ranked_selection",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "da735293a9d3102b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nPick the single best beverage.\n\nINPUT:\n{\n  \"dish\": \"Oaxacan mole negro enchiladas (rich chocolate-chile-spice sauce over chicken)\",\n  \"candidates\": {\n    \"a\": \"medium-tannin Garnacha from Aragón\",\n    \"b\": \"unoaked Muscadet\",\n    \"c\": \"very sweet late-harvest Riesling\",\n    \"d\": \"bone-dry Brut Champagne\"\n  }\n}\n\nANSWER FORMAT: one_of:a,b,c,d",
      "expected": {
        "value": "a",
        "ranking": [
          "a",
          "d",
          "c",
          "b"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "d",
          "compliant": true,
          "score": 0.15,
          "detail": {
            "predicted": "d",
            "expected_best": "a",
            "rank": 1
          },
          "metrics": {
            "input_tokens": 344,
            "output_tokens": 4,
            "latency_ms": 708.2,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "d",
          "compliant": true,
          "score": 0.15,
          "detail": {
            "predicted": "d",
            "expected_best": "a",
            "rank": 1
          },
          "metrics": {
            "input_tokens": 344,
            "output_tokens": 4,
            "latency_ms": 696.2,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "a",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a",
            "expected_best": "a",
            "rank": 0
          },
          "metrics": {
            "input_tokens": 344,
            "output_tokens": 4,
            "latency_ms": 546.3,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.4333,
      "score_summary": {
        "mean": 0.4333,
        "stdev": 0.4907,
        "sem": 0.2833,
        "ci95_lo": -0.122,
        "ci95_hi": 0.9887,
        "n": 3
      },
      "score_if_valid": 0.4333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 344.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 650.233,
        "usd_sum": 0.00141
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:pair-020",
      "category": "pairing",
      "capability": "pairing_grade_reasoning",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "bac251ebb1c064ef",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nA pairing's overall_score is computed as 96. Using the OGS grade scale, return the letter grade.\n\nINPUT:\n{\n  \"overall_score\": 96\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "A+"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "A+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a+",
            "expected": "a+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 5,
            "latency_ms": 740.1,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "A+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a+",
            "expected": "a+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 5,
            "latency_ms": 803.3,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "A+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "a+",
            "expected": "a+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 5,
            "latency_ms": 726.8,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 370.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 756.733,
        "usd_sum": 0.001539
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-001",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "1df867dd3a55af5b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat is the ONLY grape variety permitted in Barolo DOCG? Respond with just the variety name.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "nebbiolo"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 245,
            "output_tokens": 6,
            "latency_ms": 762.4,
            "usd": 0.000366,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 245,
            "output_tokens": 6,
            "latency_ms": 733.4,
            "usd": 0.000366,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 245,
            "output_tokens": 6,
            "latency_ms": 683.8,
            "usd": 0.000366,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 245.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 726.533,
        "usd_sum": 0.001098
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-002",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "24a95f8a1c0a0a45",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat single grape variety is used for Chablis AOC? Respond with just the variety name.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "chardonnay"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Chardonnay",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chardonnay",
            "expected": "chardonnay",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 245,
            "output_tokens": 5,
            "latency_ms": 751.9,
            "usd": 0.000356,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Chardonnay",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chardonnay",
            "expected": "chardonnay",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 245,
            "output_tokens": 5,
            "latency_ms": 752.4,
            "usd": 0.000356,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Chardonnay",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "chardonnay",
            "expected": "chardonnay",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 245,
            "output_tokens": 5,
            "latency_ms": 781.6,
            "usd": 0.000356,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 245.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 761.967,
        "usd_sum": 0.001068
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-003",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "vietnamese",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "15d36148f8b853e6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhich country is phở originally from? Respond with just the country name.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "vietnam"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Vietnam",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "vietnam",
            "expected": "vietnam",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 240,
            "output_tokens": 4,
            "latency_ms": 676.3,
            "usd": 0.00034,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Vietnam",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "vietnam",
            "expected": "vietnam",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 240,
            "output_tokens": 4,
            "latency_ms": 668.8,
            "usd": 0.00034,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Vietnam",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "vietnam",
            "expected": "vietnam",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 240,
            "output_tokens": 4,
            "latency_ms": 873.7,
            "usd": 0.00034,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 240.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 739.6,
        "usd_sum": 0.00102
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-004",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "e332af036a2bd591",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the classic Alsace wine traditionally paired with ripe, stinky Muenster cheese. One word, grape variety only.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "gewurztraminer"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Gewurztraminer",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "gewurztraminer",
            "expected": "gewurztraminer",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 250,
            "output_tokens": 7,
            "latency_ms": 816.2,
            "usd": 0.000383,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "gewurztraminer",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "gewurztraminer",
            "expected": "gewurztraminer",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 250,
            "output_tokens": 7,
            "latency_ms": 738.0,
            "usd": 0.000383,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Gewurztraminer",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "gewurztraminer",
            "expected": "gewurztraminer",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 250,
            "output_tokens": 7,
            "latency_ms": 1202.3,
            "usd": 0.000383,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 250.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 918.833,
        "usd_sum": 0.001149
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-005",
      "category": "regional",
      "capability": "french_classical_cuisine",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "4aa2c63ffdeeacfb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the French mother sauce that is made from a brown roux, brown stock and mirepoix, traditionally enriched with tomato. One word, lowercase.\n\nANSWER FORMAT: string",
      "expected": {
        "value": "espagnole"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "espagnole",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "espagnole",
            "expected": "espagnole",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 6,
            "latency_ms": 707.6,
            "usd": 0.00038,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "espagnole",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "espagnole",
            "expected": "espagnole",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 6,
            "latency_ms": 826.9,
            "usd": 0.00038,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "espagnole",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "espagnole",
            "expected": "espagnole",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 6,
            "latency_ms": 772.3,
            "usd": 0.00038,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 256.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 768.933,
        "usd_sum": 0.00114
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-006",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "medium",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "979e7579b34d9797",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat is the name of the official Rioja classification for a red wine aged at least 2 years (1 in oak) and 3+ years total? Respond with just the term.\n\nINPUT:\n{\n  \"question\": \"official Rioja classification for red wine aged 2+ years (1+ in oak), 3+ total\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "reserva"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Reserva",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "reserva",
            "expected": "reserva",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 294,
            "output_tokens": 4,
            "latency_ms": 674.4,
            "usd": 0.000407,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Reserva",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "reserva",
            "expected": "reserva",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 294,
            "output_tokens": 4,
            "latency_ms": 601.6,
            "usd": 0.000407,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Reserva",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "reserva",
            "expected": "reserva",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 294,
            "output_tokens": 4,
            "latency_ms": 620.8,
            "usd": 0.000407,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 294.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 632.267,
        "usd_sum": 0.001221
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-007",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "korean",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "23a69016b1082553",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nFrom which country does kimchi originate? Respond with just the country name in English.\n\nINPUT:\n{\n  \"dish\": \"kimchi\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "korea",
        "aliases": [
          "south korea",
          "republic of korea"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Korea",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "korea",
            "expected": "korea",
            "aliases": [
              "south korea",
              "republic of korea"
            ]
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 5,
            "latency_ms": 634.1,
            "usd": 0.000369,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Korea",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "korea",
            "expected": "korea",
            "aliases": [
              "south korea",
              "republic of korea"
            ]
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 5,
            "latency_ms": 605.7,
            "usd": 0.000369,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Korea",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "korea",
            "expected": "korea",
            "aliases": [
              "south korea",
              "republic of korea"
            ]
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 5,
            "latency_ms": 708.7,
            "usd": 0.000369,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 255.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 649.5,
        "usd_sum": 0.001107
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-008",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "0a12eb057eb17654",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat is the single grape variety required for Barbaresco DOCG?\n\nINPUT:\n{\n  \"wine\": \"Barbaresco DOCG\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "nebbiolo"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 6,
            "latency_ms": 742.7,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 6,
            "latency_ms": 726.6,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Nebbiolo",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "nebbiolo",
            "expected": "nebbiolo",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 6,
            "latency_ms": 703.6,
            "usd": 0.000379,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 255.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 724.3,
        "usd_sum": 0.001137
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-009",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "peruvian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "767d178c3cce3333",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhich country is classic ceviche most commonly attributed to? Respond with the country name.\n\nINPUT:\n{\n  \"dish\": \"ceviche\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "peru"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Peru",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "peru",
            "expected": "peru",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 5,
            "latency_ms": 604.5,
            "usd": 0.000369,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Peru",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "peru",
            "expected": "peru",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 5,
            "latency_ms": 663.6,
            "usd": 0.000369,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Peru",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "peru",
            "expected": "peru",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 255,
            "output_tokens": 5,
            "latency_ms": 528.0,
            "usd": 0.000369,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 255.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 598.7,
        "usd_sum": 0.001107
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-010",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "1001169bc92aaacd",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country does paella valenciana originate? Respond with the country name.\n\nINPUT:\n{\n  \"dish\": \"paella valenciana\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "spain"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Spain",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spain",
            "expected": "spain",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 4,
            "latency_ms": 651.6,
            "usd": 0.00036,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Spain",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spain",
            "expected": "spain",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 4,
            "latency_ms": 587.3,
            "usd": 0.00036,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Spain",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spain",
            "expected": "spain",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 4,
            "latency_ms": 731.1,
            "usd": 0.00036,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 256.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 656.667,
        "usd_sum": 0.00108
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-011",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "571e5074e66e6a11",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the classic French white wine region traditionally paired with raw oysters. Respond with just the region name.\n\nINPUT:\n{\n  \"dish\": \"raw oysters on the half shell\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "chablis",
        "aliases": [
          "muscadet"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Muscadet",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "muscadet",
            "expected": "chablis",
            "aliases": [
              "muscadet"
            ]
          },
          "metrics": {
            "input_tokens": 262,
            "output_tokens": 6,
            "latency_ms": 652.7,
            "usd": 0.000387,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Muscadet",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "muscadet",
            "expected": "chablis",
            "aliases": [
              "muscadet"
            ]
          },
          "metrics": {
            "input_tokens": 262,
            "output_tokens": 6,
            "latency_ms": 598.8,
            "usd": 0.000387,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Muscadet",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "muscadet",
            "expected": "chablis",
            "aliases": [
              "muscadet"
            ]
          },
          "metrics": {
            "input_tokens": 262,
            "output_tokens": 6,
            "latency_ms": 688.4,
            "usd": 0.000387,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 262.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 646.633,
        "usd_sum": 0.001161
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-012",
      "category": "regional",
      "capability": "french_classical_cuisine",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "d9792be4d5ca7335",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the French mother sauce made from clarified butter, egg yolks, and lemon juice.\n\nINPUT:\n{\n  \"description\": \"emulsion of clarified butter, egg yolks, and lemon juice\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "hollandaise"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Hollandaise",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hollandaise",
            "expected": "hollandaise",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 267,
            "output_tokens": 7,
            "latency_ms": 736.8,
            "usd": 0.000404,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Hollandaise",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hollandaise",
            "expected": "hollandaise",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 267,
            "output_tokens": 7,
            "latency_ms": 677.9,
            "usd": 0.000404,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Hollandaise",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hollandaise",
            "expected": "hollandaise",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 267,
            "output_tokens": 7,
            "latency_ms": 720.3,
            "usd": 0.000404,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 267.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 711.667,
        "usd_sum": 0.001212
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-013",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "c831a091be5025f9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country did edomae-style sushi originate?\n\nINPUT:\n{\n  \"dish\": \"edomae sushi\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "japan"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Japan",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "japan",
            "expected": "japan",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 251,
            "output_tokens": 4,
            "latency_ms": 628.0,
            "usd": 0.000354,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Japan",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "japan",
            "expected": "japan",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 251,
            "output_tokens": 4,
            "latency_ms": 591.1,
            "usd": 0.000354,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Japan",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "japan",
            "expected": "japan",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 251,
            "output_tokens": 4,
            "latency_ms": 683.0,
            "usd": 0.000354,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 251.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 634.033,
        "usd_sum": 0.001062
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-014",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "moroccan",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "7d761cbfb86342af",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country does the slow-cooked dish tagine originate?\n\nINPUT:\n{\n  \"dish\": \"tagine (clay-pot slow-cooked stew)\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "morocco"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Morocco",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "morocco",
            "expected": "morocco",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 259,
            "output_tokens": 5,
            "latency_ms": 743.0,
            "usd": 0.000374,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Morocco",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "morocco",
            "expected": "morocco",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 259,
            "output_tokens": 5,
            "latency_ms": 746.2,
            "usd": 0.000374,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Morocco",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "morocco",
            "expected": "morocco",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 259,
            "output_tokens": 5,
            "latency_ms": 666.1,
            "usd": 0.000374,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 259.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 718.433,
        "usd_sum": 0.001122
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-015",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "2f35d24bb9d58118",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhat single grape variety is used for white Sancerre AOC?\n\nINPUT:\n{\n  \"wine\": \"Sancerre AOC blanc\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "sauvignon blanc",
        "aliases": [
          "sauvignon_blanc",
          "sauvignon"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Sauvignon Blanc",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauvignon blanc",
            "expected": "sauvignon blanc",
            "aliases": [
              "sauvignon_blanc",
              "sauvignon"
            ]
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 7,
            "latency_ms": 770.7,
            "usd": 0.00039,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Sauvignon Blanc",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauvignon blanc",
            "expected": "sauvignon blanc",
            "aliases": [
              "sauvignon_blanc",
              "sauvignon"
            ]
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 7,
            "latency_ms": 751.8,
            "usd": 0.00039,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Sauvignon Blanc",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauvignon blanc",
            "expected": "sauvignon blanc",
            "aliases": [
              "sauvignon_blanc",
              "sauvignon"
            ]
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 7,
            "latency_ms": 665.0,
            "usd": 0.00039,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 256.0,
        "output_tokens_mean": 7.0,
        "latency_ms_mean": 729.167,
        "usd_sum": 0.00117
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-016",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "easy",
      "cuisine_region": "mexican",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "efb489c56faf463a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nIn which country does mole poblano originate?\n\nINPUT:\n{\n  \"dish\": \"mole poblano\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "mexico"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Mexico",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mexico",
            "expected": "mexico",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 249,
            "output_tokens": 4,
            "latency_ms": 576.1,
            "usd": 0.000351,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Mexico",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mexico",
            "expected": "mexico",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 249,
            "output_tokens": 4,
            "latency_ms": 704.5,
            "usd": 0.000351,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Mexico",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "mexico",
            "expected": "mexico",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 249,
            "output_tokens": 4,
            "latency_ms": 604.4,
            "usd": 0.000351,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 249.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 628.333,
        "usd_sum": 0.001053
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-017",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "53530e4634162afc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the Alsatian white wine traditionally served with choucroute garnie.\n\nINPUT:\n{\n  \"dish\": \"choucroute garnie\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "riesling"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Riesling",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "riesling",
            "expected": "riesling",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 6,
            "latency_ms": 750.2,
            "usd": 0.00038,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Riesling",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "riesling",
            "expected": "riesling",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 6,
            "latency_ms": 752.4,
            "usd": 0.00038,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Riesling",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "riesling",
            "expected": "riesling",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 256,
            "output_tokens": 6,
            "latency_ms": 545.7,
            "usd": 0.00038,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 256.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 682.767,
        "usd_sum": 0.00114
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-018",
      "category": "regional",
      "capability": "dish_origin",
      "difficulty": "medium",
      "cuisine_region": "brazilian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "28a636bdc82633de",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nWhich country claims feijoada completa as a national dish?\n\nINPUT:\n{\n  \"dish\": \"feijoada completa\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "brazil"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Brazil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "brazil",
            "expected": "brazil",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 252,
            "output_tokens": 4,
            "latency_ms": 995.6,
            "usd": 0.000355,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Brazil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "brazil",
            "expected": "brazil",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 252,
            "output_tokens": 4,
            "latency_ms": 589.7,
            "usd": 0.000355,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Brazil",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "brazil",
            "expected": "brazil",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 252,
            "output_tokens": 4,
            "latency_ms": 794.7,
            "usd": 0.000355,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 252.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 793.333,
        "usd_sum": 0.001065
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-019",
      "category": "regional",
      "capability": "wine_region_knowledge",
      "difficulty": "hard",
      "cuisine_region": "hungarian",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "2c5869ce479f65cc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nTokaji Aszú's vineyards are centred in a single country. Name that country in English.\n\nINPUT:\n{\n  \"wine\": \"Tokaji Aszú\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "hungary"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Hungary",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hungary",
            "expected": "hungary",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 261,
            "output_tokens": 5,
            "latency_ms": 5286.3,
            "usd": 0.000376,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Hungary",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hungary",
            "expected": "hungary",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 261,
            "output_tokens": 5,
            "latency_ms": 915.2,
            "usd": 0.000376,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Hungary",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "hungary",
            "expected": "hungary",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 261,
            "output_tokens": 5,
            "latency_ms": 990.5,
            "usd": 0.000376,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 261.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 2397.333,
        "usd_sum": 0.001128
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:region-020",
      "category": "regional",
      "capability": "classic_regional_pairing",
      "difficulty": "hard",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [],
      "prompt_template_hash": "aa3761bdbf2d490e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n---\n\nName the Galician white wine grape traditionally paired with pulpo á feira (boiled octopus with pimentón).\n\nINPUT:\n{\n  \"dish\": \"pulpo á feira — Galician boiled octopus with pimentón and olive oil\"\n}\n\nANSWER FORMAT: one_word",
      "expected": {
        "value": "albarino",
        "aliases": [
          "albariño"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "Albariño",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "albariño",
            "expected": "albarino",
            "aliases": [
              "albariño"
            ]
          },
          "metrics": {
            "input_tokens": 277,
            "output_tokens": 6,
            "latency_ms": 660.5,
            "usd": 0.000406,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "Albariño",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "albariño",
            "expected": "albarino",
            "aliases": [
              "albariño"
            ]
          },
          "metrics": {
            "input_tokens": 277,
            "output_tokens": 6,
            "latency_ms": 723.9,
            "usd": 0.000406,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "Albariño",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "albariño",
            "expected": "albarino",
            "aliases": [
              "albariño"
            ]
          },
          "metrics": {
            "input_tokens": 277,
            "output_tokens": 6,
            "latency_ms": 653.1,
            "usd": 0.000406,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 277.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 679.167,
        "usd_sum": 0.001218
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-001",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "08845fcf7f8beb09",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"salmon and tuna nigiri\",\n  \"beverage\": \"young Bordeaux Left Bank (Cabernet Sauvignon, high tannin)\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_TANNIN_FISH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_TANNIN_FISH",
            "expected": "RISK_TANNIN_FISH",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 636,
            "output_tokens": 10,
            "latency_ms": 685.4,
            "usd": 0.000895,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_TANNIN_FISH",
            "expected": "RISK_TANNIN_FISH",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 636,
            "output_tokens": 10,
            "latency_ms": 708.4,
            "usd": 0.000895,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_TANNIN_FISH",
            "expected": "RISK_TANNIN_FISH",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 636,
            "output_tokens": 10,
            "latency_ms": 681.2,
            "usd": 0.000895,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 636.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 691.667,
        "usd_sum": 0.002685
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-002",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "7a188b424515258f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"Thai green curry with bird's eye chilli, high heat\",\n  \"beverage\": \"15.5% ABV California Zinfandel\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SPICE_ALCOHOL"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SPICE_ALCOHOL",
            "expected": "RISK_SPICE_ALCOHOL",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 638,
            "output_tokens": 10,
            "latency_ms": 824.6,
            "usd": 0.000898,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SPICE_ALCOHOL",
            "expected": "RISK_SPICE_ALCOHOL",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 638,
            "output_tokens": 10,
            "latency_ms": 742.7,
            "usd": 0.000898,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SPICE_ALCOHOL",
            "expected": "RISK_SPICE_ALCOHOL",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 638,
            "output_tokens": 10,
            "latency_ms": 754.0,
            "usd": 0.000898,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 638.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 773.767,
        "usd_sum": 0.002694
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-003",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "a60fa6dd5c1ff452",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code for the pairing. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"delicate flounder sashimi with ponzu\",\n  \"beverage\": \"heavily new-oaked California Chardonnay, 14% ABV, buttery ML\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_OAK_MASKS"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_OAK_MASKS",
            "expected": "RISK_OAK_MASKS",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 641,
            "output_tokens": 9,
            "latency_ms": 554.4,
            "usd": 0.000891,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_OAK_MASKS",
            "expected": "RISK_OAK_MASKS",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 641,
            "output_tokens": 9,
            "latency_ms": 683.3,
            "usd": 0.000891,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_OAK_MASKS",
            "expected": "RISK_OAK_MASKS",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 641,
            "output_tokens": 9,
            "latency_ms": 664.3,
            "usd": 0.000891,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 641.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 634.0,
        "usd_sum": 0.002673
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-004",
      "category": "risk",
      "capability": "risk_severity",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "16286de7fea14ae1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nUsing OGS risk severity levels (low/medium/high), classify the RISK_ALCOHOL_OVERPOWERS severity for the pairing below. Respond with just the severity token.\n\nINPUT:\n{\n  \"dish\": \"pan-seared halibut with beurre blanc, very delicate flavours\",\n  \"beverage\": \"15.5% ABV Amarone della Valpolicella\"\n}\n\nANSWER FORMAT: one_of:low,medium,high",
      "expected": {
        "value": "high"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 603,
            "output_tokens": 4,
            "latency_ms": 848.6,
            "usd": 0.000794,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 603,
            "output_tokens": 4,
            "latency_ms": 766.1,
            "usd": 0.000794,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 603,
            "output_tokens": 4,
            "latency_ms": 1521.9,
            "usd": 0.000794,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 603.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1045.533,
        "usd_sum": 0.002382
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-005",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "cd44f8ac361108e8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code for the pairing. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"caramel panna cotta with honey\",\n  \"beverage\": \"bone-dry Brut Nature Champagne (zero dosage)\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SWEETNESS_CONFLICT"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SWEETNESS_CONFLICT",
            "expected": "RISK_SWEETNESS_CONFLICT",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 630,
            "output_tokens": 11,
            "latency_ms": 943.9,
            "usd": 0.000897,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SWEETNESS_CONFLICT",
            "expected": "RISK_SWEETNESS_CONFLICT",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 630,
            "output_tokens": 11,
            "latency_ms": 1536.2,
            "usd": 0.000897,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "RISK_SWEETNESS_CONFLICT",
            "expected": "RISK_SWEETNESS_CONFLICT",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 630,
            "output_tokens": 11,
            "latency_ms": 730.0,
            "usd": 0.000897,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 630.0,
        "output_tokens_mean": 11.0,
        "latency_ms_mean": 1070.033,
        "usd_sum": 0.002691
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-006",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "28cb271326a709c7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"raw tuna sashimi with a light soy-wasabi dip\",\n  \"beverage\": \"young tannic Napa Cabernet Sauvignon\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_TANNIN_FISH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 636,
            "output_tokens": 10,
            "latency_ms": 755.0,
            "usd": 0.000895,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 636,
            "output_tokens": 10,
            "latency_ms": 829.0,
            "usd": 0.000895,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 636,
            "output_tokens": 10,
            "latency_ms": 681.3,
            "usd": 0.000895,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 636.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 755.1,
        "usd_sum": 0.002685
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-007",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "93c46eac23eca249",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"spicy Thai green curry with chilies and coconut milk\",\n  \"beverage\": \"15.5% abv California Zinfandel\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SPICE_ALCOHOL"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 639,
            "output_tokens": 10,
            "latency_ms": 656.0,
            "usd": 0.000899,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 639,
            "output_tokens": 10,
            "latency_ms": 870.6,
            "usd": 0.000899,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 639,
            "output_tokens": 10,
            "latency_ms": 705.9,
            "usd": 0.000899,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 639.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 744.167,
        "usd_sum": 0.002697
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-008",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "2409e606629eebf9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"grilled escarole salad with bitter radicchio and walnuts\",\n  \"beverage\": \"bone-dry bitter amontillado sherry\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_BITTER_AMPLIFY"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 641,
            "output_tokens": 10,
            "latency_ms": 737.8,
            "usd": 0.000901,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 641,
            "output_tokens": 10,
            "latency_ms": 1739.0,
            "usd": 0.000901,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 641,
            "output_tokens": 10,
            "latency_ms": 759.7,
            "usd": 0.000901,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 641.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 1078.833,
        "usd_sum": 0.002703
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-009",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "a00498fd7fce7cb7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"raw scallop crudo with yuzu\",\n  \"beverage\": \"heavily oaked California Chardonnay, new French barrels\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_OAK_MASKS"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_oak_masks",
            "expected": "risk_oak_masks",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 637,
            "output_tokens": 9,
            "latency_ms": 775.0,
            "usd": 0.000886,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_oak_masks",
            "expected": "risk_oak_masks",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 637,
            "output_tokens": 9,
            "latency_ms": 664.1,
            "usd": 0.000886,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_OAK_MASKS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_oak_masks",
            "expected": "risk_oak_masks",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 637,
            "output_tokens": 9,
            "latency_ms": 756.8,
            "usd": 0.000886,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 637.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 731.967,
        "usd_sum": 0.002658
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-010",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "b28af5fc40407ce0",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"warm clear beef consommé\",\n  \"beverage\": \"sweet late-harvest Moscato d'Asti\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SWEETNESS_CONFLICT"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 633,
            "output_tokens": 11,
            "latency_ms": 775.4,
            "usd": 0.000901,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 633,
            "output_tokens": 11,
            "latency_ms": 833.1,
            "usd": 0.000901,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 633,
            "output_tokens": 11,
            "latency_ms": 867.3,
            "usd": 0.000901,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 633.0,
        "output_tokens_mean": 11.0,
        "latency_ms_mean": 825.267,
        "usd_sum": 0.002703
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-011",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "fba6f85afa81595f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"frisée salad with aggressive Dijon vinaigrette (1:2 vinegar to oil)\",\n  \"beverage\": \"bone-dry Brut Nature Champagne\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_ACID_CLASH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_ACID_CLASH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_acid_clash",
            "expected": "risk_acid_clash",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 640,
            "output_tokens": 9,
            "latency_ms": 760.8,
            "usd": 0.00089,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_ACID_CLASH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_acid_clash",
            "expected": "risk_acid_clash",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 640,
            "output_tokens": 9,
            "latency_ms": 731.5,
            "usd": 0.00089,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_ACID_CLASH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_acid_clash",
            "expected": "risk_acid_clash",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 640,
            "output_tokens": 9,
            "latency_ms": 769.9,
            "usd": 0.00089,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 640.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 754.067,
        "usd_sum": 0.00267
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-012",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "italian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "2f3a4360a39afdc9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"steamed turbot with beurre blanc\",\n  \"beverage\": \"young powerful Barolo\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_INTENSITY_MISMATCH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 629,
            "output_tokens": 10,
            "latency_ms": 783.7,
            "usd": 0.000886,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 629,
            "output_tokens": 10,
            "latency_ms": 667.5,
            "usd": 0.000886,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 629,
            "output_tokens": 10,
            "latency_ms": 819.3,
            "usd": 0.000886,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 629.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 756.833,
        "usd_sum": 0.002658
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-013",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "390e90cae060ada8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"yellowtail sashimi\",\n  \"beverage\": \"overproof 63% abv Jamaican rum\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_ALCOHOL_OVERPOWERS"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_ALCOHOL_OVERPOWERS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_alcohol_overpowers",
            "expected": "risk_alcohol_overpowers",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 631,
            "output_tokens": 12,
            "latency_ms": 774.9,
            "usd": 0.000909,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_ALCOHOL_OVERPOWERS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_alcohol_overpowers",
            "expected": "risk_alcohol_overpowers",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 631,
            "output_tokens": 12,
            "latency_ms": 705.8,
            "usd": 0.000909,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_ALCOHOL_OVERPOWERS",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_alcohol_overpowers",
            "expected": "risk_alcohol_overpowers",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 631,
            "output_tokens": 12,
            "latency_ms": 678.9,
            "usd": 0.000909,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 631.0,
        "output_tokens_mean": 12.0,
        "latency_ms_mean": 719.867,
        "usd_sum": 0.002727
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-014",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "d83b8b2ecabe9907",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"gin-cured salmon gravlax with dill\",\n  \"beverage\": \"young tannic Chianti Classico\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_TANNIN_FISH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 634,
            "output_tokens": 10,
            "latency_ms": 1348.8,
            "usd": 0.000893,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 634,
            "output_tokens": 10,
            "latency_ms": 637.5,
            "usd": 0.000893,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_TANNIN_FISH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_tannin_fish",
            "expected": "risk_tannin_fish",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 634,
            "output_tokens": 10,
            "latency_ms": 762.3,
            "usd": 0.000893,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 634.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 916.2,
        "usd_sum": 0.002679
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-015",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "1e83097d483e0639",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"poached lobster with drawn butter\",\n  \"beverage\": \"aggressive 80-IBU West Coast IPA\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_BITTER_AMPLIFY"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 634,
            "output_tokens": 10,
            "latency_ms": 4896.5,
            "usd": 0.000893,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 634,
            "output_tokens": 10,
            "latency_ms": 778.7,
            "usd": 0.000893,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_BITTER_AMPLIFY",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_bitter_amplify",
            "expected": "risk_bitter_amplify",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 634,
            "output_tokens": 10,
            "latency_ms": 746.8,
            "usd": 0.000893,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 634.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 2140.667,
        "usd_sum": 0.002679
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-016",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "easy",
      "cuisine_region": "mexican",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "b2e02272ca87b2d4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"habanero-heavy Yucatán salsa\",\n  \"beverage\": \"15% abv Zinfandel\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SPICE_ALCOHOL"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 633,
            "output_tokens": 10,
            "latency_ms": 673.6,
            "usd": 0.000891,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 633,
            "output_tokens": 10,
            "latency_ms": 638.1,
            "usd": 0.000891,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SPICE_ALCOHOL",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_spice_alcohol",
            "expected": "risk_spice_alcohol",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 633,
            "output_tokens": 10,
            "latency_ms": 824.4,
            "usd": 0.000891,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 633.0,
        "output_tokens_mean": 10.0,
        "latency_ms_mean": 712.033,
        "usd_sum": 0.002673
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-017",
      "category": "risk",
      "capability": "risk_severity",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "12372c8de13add55",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nUsing OGS risk severity levels (low/medium/high), classify the risk severity of the described pairing. Respond with just the severity level.\n\nINPUT:\n{\n  \"dish\": \"simple grilled branzino with lemon\",\n  \"beverage\": \"heavily oaked California Chardonnay\",\n  \"risk_code\": \"RISK_OAK_MASKS\"\n}\n\nANSWER FORMAT: one_of:low,medium,high",
      "expected": {
        "value": "medium"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "high",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "high",
            "expected": "medium",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 596,
            "output_tokens": 4,
            "latency_ms": 2070.3,
            "usd": 0.000785,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "high",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "high",
            "expected": "medium",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 596,
            "output_tokens": 4,
            "latency_ms": 806.0,
            "usd": 0.000785,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "high",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "high",
            "expected": "medium",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 596,
            "output_tokens": 4,
            "latency_ms": 704.4,
            "usd": 0.000785,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 596.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1193.567,
        "usd_sum": 0.002355
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-018",
      "category": "risk",
      "capability": "risk_severity",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "7a97c002792dfad9",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nUsing OGS risk severity levels, classify the severity of a tannic red wine with raw anchovies.\n\nINPUT:\n{\n  \"dish\": \"raw marinated anchovies\",\n  \"beverage\": \"young tannic Cabernet Sauvignon\",\n  \"risk_code\": \"RISK_TANNIN_FISH\"\n}\n\nANSWER FORMAT: one_of:low,medium,high",
      "expected": {
        "value": "high"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 586,
            "output_tokens": 4,
            "latency_ms": 851.1,
            "usd": 0.000772,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 586,
            "output_tokens": 4,
            "latency_ms": 625.3,
            "usd": 0.000772,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "high",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "high",
            "expected": "high",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 586,
            "output_tokens": 4,
            "latency_ms": 818.1,
            "usd": 0.000772,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 586.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 764.833,
        "usd_sum": 0.002316
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-019",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "german",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "0e3d6c3f9d86af0b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"raw Kumamoto oyster\",\n  \"beverage\": \"off-dry Spätlese Riesling with 45 g/L residual sugar\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_SWEETNESS_CONFLICT"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_sweetness_conflict",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 638,
            "output_tokens": 11,
            "latency_ms": 736.5,
            "usd": 0.000907,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_sweetness_conflict",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 638,
            "output_tokens": 11,
            "latency_ms": 1160.1,
            "usd": 0.000907,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_SWEETNESS_CONFLICT",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_sweetness_conflict",
            "expected": "risk_sweetness_conflict",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 638,
            "output_tokens": 11,
            "latency_ms": 696.2,
            "usd": 0.000907,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 638.0,
        "output_tokens_mean": 11.0,
        "latency_ms_mean": 864.267,
        "usd_sum": 0.002721
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:risk-020",
      "category": "risk",
      "capability": "risk_identification",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.explanation_codes.negative"
      ],
      "prompt_template_hash": "2dd649d25c606ee7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (negative / risk only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n---\n\nReturn the single most important OGS negative explanation-code (risk) for the pairing below. Respond with just the code ID.\n\nINPUT:\n{\n  \"dish\": \"slow-braised oxtail stew with dark jus\",\n  \"beverage\": \"very pale Provençal rosé, 12.5% abv\"\n}\n\nANSWER FORMAT: one_of:RISK_TANNIN_FISH,RISK_ALCOHOL_OVERPOWERS,RISK_BITTER_AMPLIFY,RISK_OAK_MASKS,RISK_INTENSITY_MISMATCH,RISK_ACID_CLASH,RISK_SWEETNESS_CONFLICT,RISK_SPICE_ALCOHOL",
      "expected": {
        "value": "RISK_INTENSITY_MISMATCH"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 642,
            "output_tokens": 11,
            "latency_ms": 619.1,
            "usd": 0.000913,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 642,
            "output_tokens": 11,
            "latency_ms": 880.3,
            "usd": 0.000913,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "RISK_INTENSITY_MISMATCH",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "risk_intensity_mismatch",
            "expected": "risk_intensity_mismatch",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 642,
            "output_tokens": 11,
            "latency_ms": 757.1,
            "usd": 0.000913,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 642.0,
        "output_tokens_mean": 11.0,
        "latency_ms_mean": 752.167,
        "usd_sum": 0.002739
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-001",
      "category": "safety",
      "capability": "cooking_temperature",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "8e37b185cfaed2aa",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhat is the minimum core temperature (°C) required to safely cook poultry under the OGS HACCP reference? Respond with a single number.\n\nINPUT:\n{\n  \"product\": \"whole chicken\"\n}\n\nANSWER FORMAT: numeric:integer or decimal °C",
      "expected": {
        "value": 75,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 794,
            "output_tokens": 4,
            "latency_ms": 751.5,
            "usd": 0.001032,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 794,
            "output_tokens": 4,
            "latency_ms": 966.7,
            "usd": 0.001032,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 794,
            "output_tokens": 4,
            "latency_ms": 610.0,
            "usd": 0.001032,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 794.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 776.067,
        "usd_sum": 0.003096
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-002",
      "category": "safety",
      "capability": "cooking_temperature",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "d38d6470a18f7e97",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nA restaurant serves minced-beef smash burgers. What minimum core temperature (°C) should the patties reach? Respond with a single number.\n\nINPUT:\n{\n  \"product\": \"minced beef patty\"\n}\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": 75,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 795,
            "output_tokens": 4,
            "latency_ms": 617.1,
            "usd": 0.001034,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 795,
            "output_tokens": 4,
            "latency_ms": 664.3,
            "usd": 0.001034,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 795,
            "output_tokens": 4,
            "latency_ms": 634.5,
            "usd": 0.001034,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 795.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 638.633,
        "usd_sum": 0.003102
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-003",
      "category": "safety",
      "capability": "danger_zone",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "97f1d8e6fc44b9e2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nThe bacterial danger zone is defined between two temperatures in °C. Respond as 'LOW-HIGH' in °C (integers), e.g. '10-50'.\n\nANSWER FORMAT: exact:LOW-HIGH °C",
      "expected": {
        "value": "5-63",
        "aliases": [
          "5 to 63",
          "5-63°C",
          "5 C - 63 C"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "5-63 °C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "5-63 °c",
            "expected": "5-63",
            "aliases": [
              "5 to 63",
              "5-63°C",
              "5 C - 63 C"
            ]
          },
          "metrics": {
            "input_tokens": 788,
            "output_tokens": 8,
            "latency_ms": 740.4,
            "usd": 0.001065,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "5-63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "5-63",
            "expected": "5-63",
            "aliases": [
              "5 to 63",
              "5-63°C",
              "5 C - 63 C"
            ]
          },
          "metrics": {
            "input_tokens": 788,
            "output_tokens": 6,
            "latency_ms": 809.1,
            "usd": 0.001045,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "5-63 °C",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "5-63 °c",
            "expected": "5-63",
            "aliases": [
              "5 to 63",
              "5-63°C",
              "5 C - 63 C"
            ]
          },
          "metrics": {
            "input_tokens": 788,
            "output_tokens": 8,
            "latency_ms": 967.6,
            "usd": 0.001065,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 788.0,
        "output_tokens_mean": 7.333,
        "latency_ms_mean": 839.033,
        "usd_sum": 0.003175
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-004",
      "category": "safety",
      "capability": "cooling_rule",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "9e5770b6c7036769",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nUnder UK FSA guidance, hot food must be cooled from 63 °C down to 21 °C within how many hours? Respond with a single integer.\n\nANSWER FORMAT: numeric:integer hours",
      "expected": {
        "value": 2,
        "tolerance": 0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 783,
            "output_tokens": 4,
            "latency_ms": 1099.7,
            "usd": 0.001019,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 783,
            "output_tokens": 4,
            "latency_ms": 669.3,
            "usd": 0.001019,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "2",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 2.0,
            "expected": 2.0,
            "tolerance": 0.0
          },
          "metrics": {
            "input_tokens": 783,
            "output_tokens": 4,
            "latency_ms": 996.1,
            "usd": 0.001019,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 783.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 921.7,
        "usd_sum": 0.003057
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-005",
      "category": "safety",
      "capability": "reheating",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "cb2ff73059921b59",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nTo what minimum core temperature (°C) must a previously cooked and chilled dish be reheated under OGS HACCP? Respond with a single number.\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": 75,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 782,
            "output_tokens": 4,
            "latency_ms": 1118.2,
            "usd": 0.001018,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 782,
            "output_tokens": 4,
            "latency_ms": 632.3,
            "usd": 0.001018,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "75",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 75.0,
            "expected": 75.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 782,
            "output_tokens": 4,
            "latency_ms": 638.2,
            "usd": 0.001018,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 782.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 796.233,
        "usd_sum": 0.003054
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-006",
      "category": "safety",
      "capability": "hot_holding",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "b3c9f3a848d65390",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhat minimum temperature (°C) must hot food be held at for service? Respond with a single number.\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": 63,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 63.0,
            "expected": 63.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 773,
            "output_tokens": 4,
            "latency_ms": 700.0,
            "usd": 0.001006,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 63.0,
            "expected": 63.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 773,
            "output_tokens": 4,
            "latency_ms": 787.1,
            "usd": 0.001006,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "63",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 63.0,
            "expected": 63.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 773,
            "output_tokens": 4,
            "latency_ms": 739.8,
            "usd": 0.001006,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 773.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 742.3,
        "usd_sum": 0.003018
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-007",
      "category": "safety",
      "capability": "cold_chain",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "f120f4a38737e981",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhat is the maximum temperature (°C) for frozen storage under OGS HACCP? Respond with a single number (negative).\n\nANSWER FORMAT: numeric:integer °C",
      "expected": {
        "value": -18,
        "tolerance": 1
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "-18",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": -18.0,
            "expected": -18.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 777,
            "output_tokens": 5,
            "latency_ms": 899.2,
            "usd": 0.001021,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "-18",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": -18.0,
            "expected": -18.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 777,
            "output_tokens": 5,
            "latency_ms": 4016.4,
            "usd": 0.001021,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "-18",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": -18.0,
            "expected": -18.0,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 777,
            "output_tokens": 5,
            "latency_ms": 809.0,
            "usd": 0.001021,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 777.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 1908.2,
        "usd_sum": 0.003063
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-008",
      "category": "safety",
      "capability": "thawing_method",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "9b8f975bd042c7a0",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nWhich of the following thawing methods is explicitly NOT permitted under OGS HACCP: 'in the fridge at 4 °C', 'under cold running water', 'in the microwave just before cooking', 'at room temperature on the counter'? Respond with the exact method phrase.\n\nINPUT:\n{\n  \"methods\": [\n    \"in the fridge at 4 °C\",\n    \"under cold running water\",\n    \"in the microwave just before cooking\",\n    \"at room temperature on the counter\"\n  ]\n}\n\nANSWER FORMAT: exact:method phrase",
      "expected": {
        "value": "at room temperature on the counter",
        "aliases": [
          "room temperature",
          "on the counter",
          "room-temperature thawing"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "at room temperature on the counter",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "at room temperature on the counter",
            "expected": "at room temperature on the counter",
            "aliases": [
              "room temperature",
              "on the counter",
              "room-temperature thawing"
            ]
          },
          "metrics": {
            "input_tokens": 853,
            "output_tokens": 9,
            "latency_ms": 661.4,
            "usd": 0.001156,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "at room temperature on the counter",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "at room temperature on the counter",
            "expected": "at room temperature on the counter",
            "aliases": [
              "room temperature",
              "on the counter",
              "room-temperature thawing"
            ]
          },
          "metrics": {
            "input_tokens": 853,
            "output_tokens": 9,
            "latency_ms": 797.5,
            "usd": 0.001156,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "at room temperature on the counter",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "at room temperature on the counter",
            "expected": "at room temperature on the counter",
            "aliases": [
              "room temperature",
              "on the counter",
              "room-temperature thawing"
            ]
          },
          "metrics": {
            "input_tokens": 853,
            "output_tokens": 9,
            "latency_ms": 905.7,
            "usd": 0.001156,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 853.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 788.2,
        "usd_sum": 0.003468
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-009",
      "category": "safety",
      "capability": "ccp_identification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "set_f1",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "9f4b49c1c20e4376",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nFor a cook-chill dish of chicken curry with rice, identify the Critical Control Points from the OGS stage list. Return a comma-separated list using the OGS CCP stage IDs.\n\nINPUT:\n{\n  \"dish\": \"cook-chill chicken curry with rice\"\n}\n\nANSWER FORMAT: set:comma-separated CCP stage IDs",
      "expected": {
        "values": [
          "cooking",
          "cooling",
          "storage_cold",
          "reheating",
          "hot_holding"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "receiving,storage_cold,prep,cooking,cooling,storage_cold,reheating,hot_holding,service",
          "compliant": true,
          "score": 0.7692,
          "detail": {
            "predicted": [
              "cooking",
              "cooling",
              "hot_holding",
              "prep",
              "receiving",
              "reheating",
              "service",
              "storage_cold"
            ],
            "expected": [
              "cooking",
              "cooling",
              "hot_holding",
              "reheating",
              "storage_cold"
            ],
            "precision": 0.625,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 807,
            "output_tokens": 28,
            "latency_ms": 1143.3,
            "usd": 0.001289,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "receiving,storage_cold,prep,cooking,cooling,storage_cold,reheating,hot_holding,service",
          "compliant": true,
          "score": 0.7692,
          "detail": {
            "predicted": [
              "cooking",
              "cooling",
              "hot_holding",
              "prep",
              "receiving",
              "reheating",
              "service",
              "storage_cold"
            ],
            "expected": [
              "cooking",
              "cooling",
              "hot_holding",
              "reheating",
              "storage_cold"
            ],
            "precision": 0.625,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 807,
            "output_tokens": 28,
            "latency_ms": 886.9,
            "usd": 0.001289,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "receiving,storage_cold,prep,cooking,cooling,storage_cold,reheating,hot_holding,service",
          "compliant": true,
          "score": 0.7692,
          "detail": {
            "predicted": [
              "cooking",
              "cooling",
              "hot_holding",
              "prep",
              "receiving",
              "reheating",
              "service",
              "storage_cold"
            ],
            "expected": [
              "cooking",
              "cooling",
              "hot_holding",
              "reheating",
              "storage_cold"
            ],
            "precision": 0.625,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 807,
            "output_tokens": 28,
            "latency_ms": 1043.7,
            "usd": 0.001289,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.7692,
      "score_summary": {
        "mean": 0.7692,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.7692,
        "ci95_hi": 0.7692,
        "n": 3
      },
      "score_if_valid": 0.7692,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 807.0,
        "output_tokens_mean": 28.0,
        "latency_ms_mean": 1024.633,
        "usd_sum": 0.003867
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-010",
      "category": "safety",
      "capability": "cooking_temperature",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "eea6814938f18565",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nA bistro serves a rare whole-muscle ribeye (seared exterior, internal 46 °C). Under OGS HACCP, is this permitted? Answer yes or no.\n\nINPUT:\n{\n  \"product\": \"whole-muscle ribeye\",\n  \"core_c\": 46\n}\n\nANSWER FORMAT: exact:yes|no",
      "expected": {
        "value": "yes"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "yes",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "yes",
            "expected": "yes",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 811,
            "output_tokens": 4,
            "latency_ms": 867.6,
            "usd": 0.001054,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "yes",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "yes",
            "expected": "yes",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 811,
            "output_tokens": 4,
            "latency_ms": 852.0,
            "usd": 0.001054,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "yes",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "yes",
            "expected": "yes",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 811,
            "output_tokens": 4,
            "latency_ms": 662.8,
            "usd": 0.001054,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 811.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 794.133,
        "usd_sum": 0.003162
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-011",
      "category": "safety",
      "capability": "at_risk_groups",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "edc2981cec11c39e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nA children's hospital must serve hollandaise sauce. Which is the ONE compliant option: 'fresh shell eggs at 63 °C', 'fresh shell eggs at 70 °C', 'pasteurised egg yolk', 'raw shell eggs with lemon juice'? Respond with the exact phrase.\n\nINPUT:\n{\n  \"options\": [\n    \"fresh shell eggs at 63 °C\",\n    \"fresh shell eggs at 70 °C\",\n    \"pasteurised egg yolk\",\n    \"raw shell eggs with lemon juice\"\n  ]\n}\n\nANSWER FORMAT: exact:option phrase",
      "expected": {
        "value": "pasteurised egg yolk",
        "aliases": [
          "pasteurised egg",
          "pasteurized egg yolk",
          "pasteurized egg"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "pasteurised egg yolk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "pasteurised egg yolk",
            "expected": "pasteurised egg yolk",
            "aliases": [
              "pasteurised egg",
              "pasteurized egg yolk",
              "pasteurized egg"
            ]
          },
          "metrics": {
            "input_tokens": 860,
            "output_tokens": 9,
            "latency_ms": 791.0,
            "usd": 0.001165,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "pasteurised egg yolk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "pasteurised egg yolk",
            "expected": "pasteurised egg yolk",
            "aliases": [
              "pasteurised egg",
              "pasteurized egg yolk",
              "pasteurized egg"
            ]
          },
          "metrics": {
            "input_tokens": 860,
            "output_tokens": 9,
            "latency_ms": 895.7,
            "usd": 0.001165,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "pasteurised egg yolk",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "pasteurised egg yolk",
            "expected": "pasteurised egg yolk",
            "aliases": [
              "pasteurised egg",
              "pasteurized egg yolk",
              "pasteurized egg"
            ]
          },
          "metrics": {
            "input_tokens": 860,
            "output_tokens": 9,
            "latency_ms": 816.4,
            "usd": 0.001165,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 860.0,
        "output_tokens_mean": 9.0,
        "latency_ms_mean": 834.367,
        "usd_sum": 0.003495
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:safe-012",
      "category": "safety",
      "capability": "numeric_vector",
      "difficulty": "hard",
      "cuisine_region": "global",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "rules.haccp",
        "vocab.haccp_ccp_stages"
      ],
      "prompt_template_hash": "2ba4113ffe26b2fc",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### HACCP temperature and time rules (food-safety reference)\n\nTemperatures are expressed in degrees Celsius.\n\n**Danger zone**: 5 °C to 63 °C — limit total exposure to 4 hours.\n\n**Cold chain**:\n- Chilled storage: 0 to 5 °C\n- Frozen storage: ≤ -18 °C\n- Thawing: in fridge at ≤ 5 °C, or under cold running water, or in   microwave immediately before cooking. Never at room temperature.\n\n**Cooking core temperatures (minimum, hold for ≥15 s)**:\n- Poultry, reformed meat (burgers, sausages), stuffed meat: **≥ 75 °C**\n- Minced beef (burgers served medium+ from whole muscle after searing   all external surfaces to 75 °C is acceptable): **≥ 75 °C** as a   default; 70 °C for 2 min or 65 °C for 10 min are equivalent.\n- Whole-muscle beef/lamb (steak/roast) cooked to order: any internal   temperature acceptable if the external surface has been seared.\n- Pork: **≥ 63 °C** held 3 min (USDA) or 75 °C (EU retail default).\n- Fish (non-sushi-grade): **≥ 63 °C** held 15 s.\n- Shell eggs for at-risk groups: **≥ 70 °C** or use pasteurised egg.\n\n**Hot holding**: ≥ 63 °C.\n**Cooling**: 63 °C → 21 °C within 2 hours, then 21 °C → 5 °C within 4 more hours (UK FSA), or 57 °C → 21 °C within 2 hours and 21 °C → 5 °C within 4 hours (FDA).\n**Reheating**: reach **≥ 75 °C** (UK) / 74 °C (US) throughout, within 2 hours. Reheat once only.\n\n**CCP stages**: `receiving`, `storage_cold`, `storage_dry`, `thawing`, `prep`, `cooking`, `hot_holding`, `cooling`, `reheating`, `service`, `cleaning_sanitising`.\n\n### HACCP CCP stages\n\nreceiving, storage_cold, storage_dry, thawing, prep, cooking, hot_holding, cooling, reheating, service, cleaning_sanitising\n\n---\n\nReturn the canonical HACCP reference temperatures in °C for this dish flow (poultry cook-chill operation): [cold_storage_max, poultry_core_cook_min, hot_hold_min, reheat_min, frozen_storage_max]. Respond with a JSON array of five numbers only, in that order.\n\nINPUT:\n{\n  \"flow\": \"poultry cook-chill\"\n}\n\nANSWER FORMAT: json:array of 5 numbers",
      "expected": {
        "values": [
          5,
          75,
          63,
          75,
          -18
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[5, 75, 63, 75, -18]",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 5,
                "expected": 5,
                "score": 1.0
              },
              {
                "index": 1,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 2,
                "predicted": 63,
                "expected": 63,
                "score": 1.0
              },
              {
                "index": 3,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 4,
                "predicted": -18,
                "expected": -18,
                "score": 1.0
              }
            ],
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 829,
            "output_tokens": 18,
            "latency_ms": 970.1,
            "usd": 0.001216,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[5, 75, 63, 75, -18]",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 5,
                "expected": 5,
                "score": 1.0
              },
              {
                "index": 1,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 2,
                "predicted": 63,
                "expected": 63,
                "score": 1.0
              },
              {
                "index": 3,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 4,
                "predicted": -18,
                "expected": -18,
                "score": 1.0
              }
            ],
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 829,
            "output_tokens": 18,
            "latency_ms": 840.8,
            "usd": 0.001216,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[5, 75, 63, 75, -18]",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_position": [
              {
                "index": 0,
                "predicted": 5,
                "expected": 5,
                "score": 1.0
              },
              {
                "index": 1,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 2,
                "predicted": 63,
                "expected": 63,
                "score": 1.0
              },
              {
                "index": 3,
                "predicted": 75,
                "expected": 75,
                "score": 1.0
              },
              {
                "index": 4,
                "predicted": -18,
                "expected": -18,
                "score": 1.0
              }
            ],
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 829,
            "output_tokens": 18,
            "latency_ms": 843.8,
            "usd": 0.001216,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 829.0,
        "output_tokens_mean": 18.0,
        "latency_ms_mean": 884.9,
        "usd_sum": 0.003648
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-001",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "913118356da8150b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the umami intensity of a 36-month aged Parmigiano-Reggiano on the OGS 0-10 intensity scale (0 = not present, 10 = extreme). Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"Parmigiano-Reggiano 36 months\",\n  \"attribute\": \"umami\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 8.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "9",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 9.0,
            "expected": 8.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 4,
            "latency_ms": 1125.9,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "9",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 9.0,
            "expected": 8.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 4,
            "latency_ms": 722.8,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "9",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 9.0,
            "expected": 8.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 378,
            "output_tokens": 4,
            "latency_ms": 698.8,
            "usd": 0.000513,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 378.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 849.167,
        "usd_sum": 0.001539
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-002",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "2690028d89b60292",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the sour intensity of undiluted fresh lemon juice on the OGS 0-10 intensity scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"fresh lemon juice (undiluted)\",\n  \"attribute\": \"sour\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 9.5,
        "tolerance": 1.0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "9",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": 9.0,
            "expected": 9.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 596.4,
            "usd": 0.00049,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "9",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": 9.0,
            "expected": 9.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 748.4,
            "usd": 0.00049,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "9",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": 9.0,
            "expected": 9.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 360,
            "output_tokens": 4,
            "latency_ms": 616.6,
            "usd": 0.00049,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.5,
      "score_summary": {
        "mean": 0.5,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.5,
        "ci95_hi": 0.5,
        "n": 3
      },
      "score_if_valid": 0.5,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 360.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 653.8,
        "usd_sum": 0.00147
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-003",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "b09502eb4fd5d61f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the tannin intensity of a young (under 5 years old) Barolo on the OGS 0-10 intensity scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"beverage\": \"young Barolo, <5 years\",\n  \"attribute\": \"tannin\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 8.5,
        "tolerance": 1.0
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "8",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": 8.0,
            "expected": 8.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 775.2,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "9",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": 9.0,
            "expected": 8.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 914.5,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "8",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "predicted": 8.0,
            "expected": 8.5,
            "tolerance": 1.0
          },
          "metrics": {
            "input_tokens": 348,
            "output_tokens": 4,
            "latency_ms": 733.3,
            "usd": 0.000475,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.5,
      "score_summary": {
        "mean": 0.5,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.5,
        "ci95_hi": 0.5,
        "n": 3
      },
      "score_if_valid": 0.5,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 348.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 807.667,
        "usd_sum": 0.001425
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-004",
      "category": "sensory",
      "capability": "basic_taste_vector",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "94f009713056d52f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nReturn the five basic tastes for a ripe, in-season strawberry, each on the OGS 0-10 intensity scale. Respond as a single JSON object with keys sweet, sour, salty, bitter, umami.\n\nINPUT:\n{\n  \"ingredient\": \"ripe, in-season strawberry\"\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"object\",\n  \"required\": [\n    \"sweet\",\n    \"sour\",\n    \"salty\",\n    \"bitter\",\n    \"umami\"\n  ],\n  \"properties\": {\n    \"sweet\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"sour\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"salty\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"bitter\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"umami\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    }\n  },\n  \"additionalProperties\": false\n}",
      "expected": {
        "values": {
          "sweet": 7.0,
          "sour": 4.0,
          "salty": 0.5,
          "bitter": 1.0,
          "umami": 0.5
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"sweet\":7.5,\"sour\":3.5,\"salty\":0,\"bitter\":0.5,\"umami\":0}",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 3.5,
                "expected": 4.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              },
              "bitter": {
                "predicted": 0.5,
                "expected": 1.0,
                "score": 0.75
              },
              "umami": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 566,
            "output_tokens": 34,
            "latency_ms": 1309.1,
            "usd": 0.001048,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"sweet\":7.5,\"sour\":3.5,\"salty\":0,\"bitter\":0.5,\"umami\":0}",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 3.5,
                "expected": 4.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              },
              "bitter": {
                "predicted": 0.5,
                "expected": 1.0,
                "score": 0.75
              },
              "umami": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 566,
            "output_tokens": 34,
            "latency_ms": 1005.0,
            "usd": 0.001048,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"sweet\":7,\"sour\":4,\"salty\":0,\"bitter\":0.5,\"umami\":0}",
          "compliant": true,
          "score": 0.85,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7,
                "expected": 7.0,
                "score": 1.0
              },
              "sour": {
                "predicted": 4,
                "expected": 4.0,
                "score": 1.0
              },
              "salty": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              },
              "bitter": {
                "predicted": 0.5,
                "expected": 1.0,
                "score": 0.75
              },
              "umami": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 566,
            "output_tokens": 30,
            "latency_ms": 1049.1,
            "usd": 0.001007,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.7833,
      "score_summary": {
        "mean": 0.7833,
        "stdev": 0.0577,
        "sem": 0.0333,
        "ci95_lo": 0.718,
        "ci95_hi": 0.8487,
        "n": 3
      },
      "score_if_valid": 0.7833,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 566.0,
        "output_tokens_mean": 32.667,
        "latency_ms_mean": 1121.067,
        "usd_sum": 0.003103
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-005",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "american",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "c950ec9851ad52bf",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the body of a New World (California) oaked Chardonnay at 14% ABV on the OGS 0-10 body scale (0 = water-thin, 10 = syrupy). Respond with only a single decimal number.\n\nINPUT:\n{\n  \"beverage\": \"oaked California Chardonnay, 14% ABV\",\n  \"attribute\": \"body\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 7.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "7",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 7.0,
            "expected": 7.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 364,
            "output_tokens": 4,
            "latency_ms": 734.0,
            "usd": 0.000495,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "7",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 7.0,
            "expected": 7.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 364,
            "output_tokens": 4,
            "latency_ms": 869.5,
            "usd": 0.000495,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "7",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 7.0,
            "expected": 7.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 364,
            "output_tokens": 4,
            "latency_ms": 714.0,
            "usd": 0.000495,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 364.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 772.5,
        "usd_sum": 0.001485
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-006",
      "category": "sensory",
      "capability": "dominant_aromatic_family",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "d68ef8562addd55d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nIdentify the single dominant aromatic FAMILY (from the OGS vocabulary) of a classic Marlborough Sauvignon Blanc. Respond with just the family name.\n\nINPUT:\n{\n  \"beverage\": \"Marlborough Sauvignon Blanc\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "vegetal"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fruit",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "fruit",
            "expected": "vegetal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 329,
            "output_tokens": 4,
            "latency_ms": 708.0,
            "usd": 0.000451,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "fruit",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "fruit",
            "expected": "vegetal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 329,
            "output_tokens": 4,
            "latency_ms": 667.1,
            "usd": 0.000451,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "fruit",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "fruit",
            "expected": "vegetal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 329,
            "output_tokens": 4,
            "latency_ms": 693.6,
            "usd": 0.000451,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 329.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 689.567,
        "usd_sum": 0.001353
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-007",
      "category": "sensory",
      "capability": "structural_vector",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "658c14cce434cc0a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nReturn the OGS structural attributes of a Brut Champagne on the 0-10 scale. Respond as a JSON object with keys body, acidity_perception, tannin, astringency, alcohol_heat, effervescence.\n\nINPUT:\n{\n  \"beverage\": \"Brut non-vintage Champagne\"\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"object\",\n  \"required\": [\n    \"body\",\n    \"acidity_perception\",\n    \"tannin\",\n    \"astringency\",\n    \"alcohol_heat\",\n    \"effervescence\"\n  ],\n  \"properties\": {\n    \"body\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"acidity_perception\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"tannin\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"astringency\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"alcohol_heat\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"effervescence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    }\n  },\n  \"additionalProperties\": false\n}",
      "expected": {
        "values": {
          "body": 4.5,
          "acidity_perception": 8.0,
          "tannin": 0.5,
          "astringency": 1.0,
          "alcohol_heat": 3.5,
          "effervescence": 9.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"body\":4,\"acidity_perception\":8,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":3,\"effervescence\":9}",
          "compliant": true,
          "score": 0.875,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 4,
                "expected": 4.5,
                "score": 0.75
              },
              "acidity_perception": {
                "predicted": 8,
                "expected": 8.0,
                "score": 1.0
              },
              "tannin": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              },
              "astringency": {
                "predicted": 1,
                "expected": 1.0,
                "score": 1.0
              },
              "alcohol_heat": {
                "predicted": 3,
                "expected": 3.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 593,
            "output_tokens": 38,
            "latency_ms": 1158.6,
            "usd": 0.001121,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"body\":4,\"acidity_perception\":8,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":3,\"effervescence\":9}",
          "compliant": true,
          "score": 0.875,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 4,
                "expected": 4.5,
                "score": 0.75
              },
              "acidity_perception": {
                "predicted": 8,
                "expected": 8.0,
                "score": 1.0
              },
              "tannin": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              },
              "astringency": {
                "predicted": 1,
                "expected": 1.0,
                "score": 1.0
              },
              "alcohol_heat": {
                "predicted": 3,
                "expected": 3.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 593,
            "output_tokens": 38,
            "latency_ms": 1128.2,
            "usd": 0.001121,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"body\":4,\"acidity_perception\":8,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":3,\"effervescence\":9}",
          "compliant": true,
          "score": 0.875,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 4,
                "expected": 4.5,
                "score": 0.75
              },
              "acidity_perception": {
                "predicted": 8,
                "expected": 8.0,
                "score": 1.0
              },
              "tannin": {
                "predicted": 0,
                "expected": 0.5,
                "score": 0.75
              },
              "astringency": {
                "predicted": 1,
                "expected": 1.0,
                "score": 1.0
              },
              "alcohol_heat": {
                "predicted": 3,
                "expected": 3.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 593,
            "output_tokens": 38,
            "latency_ms": 1117.2,
            "usd": 0.001121,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.875,
      "score_summary": {
        "mean": 0.875,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.875,
        "ci95_hi": 0.875,
        "n": 3
      },
      "score_if_valid": 0.875,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 593.0,
        "output_tokens_mean": 38.0,
        "latency_ms_mean": 1134.667,
        "usd_sum": 0.003363
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-008",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "873c6c2f9608e43f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the bitter intensity of raw radicchio on the OGS 0-10 intensity scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"raw radicchio\",\n  \"attribute\": \"bitter\"\n}\n\nANSWER FORMAT: integer_0_10",
      "expected": {
        "value": 6.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "7",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 7.0,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 355,
            "output_tokens": 4,
            "latency_ms": 590.4,
            "usd": 0.000484,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "7",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 7.0,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 355,
            "output_tokens": 4,
            "latency_ms": 725.4,
            "usd": 0.000484,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "7",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 7.0,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 355,
            "output_tokens": 4,
            "latency_ms": 597.9,
            "usd": 0.000484,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 355.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 637.9,
        "usd_sum": 0.001452
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-009",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "523e04f39252cb4b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the umami intensity of a plain 10-minute cold-steep kombu dashi on the OGS 0-10 scale. Respond with only a single decimal number.\n\nINPUT:\n{\n  \"ingredient\": \"kombu dashi (cold steep 10 min, 10 g/L Hokkaido kombu)\",\n  \"attribute\": \"umami\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 6.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "4.5",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": 4.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 379,
            "output_tokens": 6,
            "latency_ms": 785.4,
            "usd": 0.000534,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "5.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 5.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 379,
            "output_tokens": 6,
            "latency_ms": 701.1,
            "usd": 0.000534,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "5.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 5.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 379,
            "output_tokens": 6,
            "latency_ms": 860.8,
            "usd": 0.000534,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.2222,
      "score_summary": {
        "mean": 0.2222,
        "stdev": 0.1924,
        "sem": 0.1111,
        "ci95_lo": 0.0044,
        "ci95_hi": 0.44,
        "n": 3
      },
      "score_if_valid": 0.2222,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 379.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 782.433,
        "usd_sum": 0.001602
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-010",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "1813cae53ba17a84",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the bitter intensity of a square of unsweetened 85% cacao dark chocolate on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"single-origin 85% dark chocolate\",\n  \"attribute\": \"bitter\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 6.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 357,
            "output_tokens": 6,
            "latency_ms": 679.1,
            "usd": 0.000506,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 357,
            "output_tokens": 6,
            "latency_ms": 791.7,
            "usd": 0.000506,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "7.5",
          "compliant": true,
          "score": 0.3333,
          "detail": {
            "predicted": 7.5,
            "expected": 6.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 357,
            "output_tokens": 6,
            "latency_ms": 872.7,
            "usd": 0.000506,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.3333,
        "ci95_hi": 0.3333,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 357.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 781.167,
        "usd_sum": 0.001518
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-011",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "1886b2e009716d1d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the perceived acidity of freshly squeezed yuzu juice on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"fresh yuzu juice\",\n  \"attribute\": \"acidity_perception\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 9.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "9.0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 330,
            "output_tokens": 6,
            "latency_ms": 1422.1,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "9.0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 330,
            "output_tokens": 6,
            "latency_ms": 883.6,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "9.0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 330,
            "output_tokens": 6,
            "latency_ms": 649.2,
            "usd": 0.000472,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 330.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 984.967,
        "usd_sum": 0.001416
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-012",
      "category": "sensory",
      "capability": "basic_taste_vector",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "8d40a83f1b501685",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nReturn the five basic tastes of fresh-squeezed Valencia orange juice on the OGS 0-10 scale as a JSON object with keys sweet, sour, salty, bitter, umami. Respond with ONLY the JSON.\n\nINPUT:\n{\n  \"ingredient\": \"freshly squeezed Valencia orange juice\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "values": {
          "sweet": 7.0,
          "sour": 5.5,
          "salty": 0.0,
          "bitter": 0.5,
          "umami": 0.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"sweet\":7,\"sour\":5.5,\"salty\":0,\"bitter\":1.5,\"umami\":0}",
          "compliant": true,
          "score": 0.9,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7,
                "expected": 7.0,
                "score": 1.0
              },
              "sour": {
                "predicted": 5.5,
                "expected": 5.5,
                "score": 1.0
              },
              "salty": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              },
              "bitter": {
                "predicted": 1.5,
                "expected": 0.5,
                "score": 0.5
              },
              "umami": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 363,
            "output_tokens": 32,
            "latency_ms": 963.1,
            "usd": 0.000774,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"sweet\":7,\"sour\":5.5,\"salty\":0,\"bitter\":1.5,\"umami\":0}",
          "compliant": true,
          "score": 0.9,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7,
                "expected": 7.0,
                "score": 1.0
              },
              "sour": {
                "predicted": 5.5,
                "expected": 5.5,
                "score": 1.0
              },
              "salty": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              },
              "bitter": {
                "predicted": 1.5,
                "expected": 0.5,
                "score": 0.5
              },
              "umami": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 363,
            "output_tokens": 32,
            "latency_ms": 1804.2,
            "usd": 0.000774,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"sweet\":7,\"sour\":6,\"salty\":0,\"bitter\":1.5,\"umami\":0}",
          "compliant": true,
          "score": 0.85,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 7,
                "expected": 7.0,
                "score": 1.0
              },
              "sour": {
                "predicted": 6,
                "expected": 5.5,
                "score": 0.75
              },
              "salty": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              },
              "bitter": {
                "predicted": 1.5,
                "expected": 0.5,
                "score": 0.5
              },
              "umami": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 363,
            "output_tokens": 30,
            "latency_ms": 895.4,
            "usd": 0.000754,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.8833,
      "score_summary": {
        "mean": 0.8833,
        "stdev": 0.0289,
        "sem": 0.0167,
        "ci95_lo": 0.8507,
        "ci95_hi": 0.916,
        "n": 3
      },
      "score_if_valid": 0.8833,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 363.0,
        "output_tokens_mean": 31.333,
        "latency_ms_mean": 1220.9,
        "usd_sum": 0.002302
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-013",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "middle_eastern",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "691f59088f3e98f3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the sweet intensity of a fresh Medjool date on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"Medjool date, soft and ripe\",\n  \"attribute\": \"sweet\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 9.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "9.0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 351,
            "output_tokens": 6,
            "latency_ms": 782.3,
            "usd": 0.000499,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "9.0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 351,
            "output_tokens": 6,
            "latency_ms": 704.4,
            "usd": 0.000499,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "9.0",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 9.0,
            "expected": 9.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 351,
            "output_tokens": 6,
            "latency_ms": 782.7,
            "usd": 0.000499,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 351.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 756.467,
        "usd_sum": 0.001497
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-014",
      "category": "sensory",
      "capability": "dominant_aromatic_family",
      "difficulty": "medium",
      "cuisine_region": "scottish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "c4a83a79265c0e9f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nIdentify the single dominant aromatic FAMILY (OGS vocabulary) of a heavily-peated Islay single malt (e.g. Ardbeg 10).\n\nINPUT:\n{\n  \"item\": \"heavily peated Islay single malt whisky, ~50 ppm phenols\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "wood"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "earth",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "earth",
            "expected": "wood",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 341,
            "output_tokens": 4,
            "latency_ms": 665.4,
            "usd": 0.000466,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "earth",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "earth",
            "expected": "wood",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 341,
            "output_tokens": 4,
            "latency_ms": 716.6,
            "usd": 0.000466,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "earth",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "earth",
            "expected": "wood",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 341,
            "output_tokens": 4,
            "latency_ms": 765.9,
            "usd": 0.000466,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 341.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 715.967,
        "usd_sum": 0.001398
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-015",
      "category": "sensory",
      "capability": "structural_vector",
      "difficulty": "hard",
      "cuisine_region": "italian",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "66d64ca2254c09ee",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nReturn the OGS structural attributes of a 5-year-old Amarone della Valpolicella on the 0-10 scale. Respond with a JSON object with keys body, acidity_perception, tannin, astringency, alcohol_heat, effervescence. JSON only.\n\nINPUT:\n{\n  \"item\": \"Amarone della Valpolicella Classico, 5 years old, 16% abv\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "values": {
          "body": 9.0,
          "acidity_perception": 6.0,
          "tannin": 7.0,
          "astringency": 5.5,
          "alcohol_heat": 8.5,
          "effervescence": 0.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"body\":9,\"acidity_perception\":6,\"tannin\":6,\"astringency\":5,\"alcohol_heat\":8,\"effervescence\":0}",
          "compliant": true,
          "score": 0.8333,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              },
              "acidity_perception": {
                "predicted": 6,
                "expected": 6.0,
                "score": 1.0
              },
              "tannin": {
                "predicted": 6,
                "expected": 7.0,
                "score": 0.5
              },
              "astringency": {
                "predicted": 5,
                "expected": 5.5,
                "score": 0.75
              },
              "alcohol_heat": {
                "predicted": 8,
                "expected": 8.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 38,
            "latency_ms": 1161.5,
            "usd": 0.000843,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"body\":9,\"acidity_perception\":5,\"tannin\":6,\"astringency\":5,\"alcohol_heat\":8,\"effervescence\":0}",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              },
              "acidity_perception": {
                "predicted": 5,
                "expected": 6.0,
                "score": 0.5
              },
              "tannin": {
                "predicted": 6,
                "expected": 7.0,
                "score": 0.5
              },
              "astringency": {
                "predicted": 5,
                "expected": 5.5,
                "score": 0.75
              },
              "alcohol_heat": {
                "predicted": 8,
                "expected": 8.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 38,
            "latency_ms": 1221.6,
            "usd": 0.000843,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"body\":9,\"acidity_perception\":5,\"tannin\":6,\"astringency\":5,\"alcohol_heat\":8,\"effervescence\":0}",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "per_key": {
              "body": {
                "predicted": 9,
                "expected": 9.0,
                "score": 1.0
              },
              "acidity_perception": {
                "predicted": 5,
                "expected": 6.0,
                "score": 0.5
              },
              "tannin": {
                "predicted": 6,
                "expected": 7.0,
                "score": 0.5
              },
              "astringency": {
                "predicted": 5,
                "expected": 5.5,
                "score": 0.75
              },
              "alcohol_heat": {
                "predicted": 8,
                "expected": 8.5,
                "score": 0.75
              },
              "effervescence": {
                "predicted": 0,
                "expected": 0.0,
                "score": 1.0
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 370,
            "output_tokens": 38,
            "latency_ms": 1118.3,
            "usd": 0.000843,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.7778,
      "score_summary": {
        "mean": 0.7778,
        "stdev": 0.0481,
        "sem": 0.0278,
        "ci95_lo": 0.7233,
        "ci95_hi": 0.8322,
        "n": 3
      },
      "score_if_valid": 0.7778,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 370.0,
        "output_tokens_mean": 38.0,
        "latency_ms_mean": 1167.133,
        "usd_sum": 0.002529
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-016",
      "category": "sensory",
      "capability": "basic_taste_estimation",
      "difficulty": "medium",
      "cuisine_region": "spanish",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "70318c6fd3b033d2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nEstimate the umami intensity of 24-month acorn-finished jamón ibérico de bellota on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"ingredient\": \"jamón ibérico de bellota, 24 months, acorn-finished\",\n  \"attribute\": \"umami\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 8.0,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 8.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 6,
            "latency_ms": 813.0,
            "usd": 0.000521,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 8.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 6,
            "latency_ms": 1425.6,
            "usd": 0.000521,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "8.5",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 8.5,
            "expected": 8.0,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 369,
            "output_tokens": 6,
            "latency_ms": 739.6,
            "usd": 0.000521,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 369.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 992.733,
        "usd_sum": 0.001563
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-017",
      "category": "sensory",
      "capability": "dominant_aromatic_family",
      "difficulty": "medium",
      "cuisine_region": "thai",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "c3cedde89152ce76",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nIdentify the single dominant aromatic FAMILY of fresh Thai basil (horapa) leaves.\n\nINPUT:\n{\n  \"item\": \"fresh Thai basil (horapa), whole leaves\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "herbal"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 320,
            "output_tokens": 5,
            "latency_ms": 726.2,
            "usd": 0.00045,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 320,
            "output_tokens": 5,
            "latency_ms": 717.5,
            "usd": 0.00045,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 320,
            "output_tokens": 5,
            "latency_ms": 697.5,
            "usd": 0.00045,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 320.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 713.733,
        "usd_sum": 0.00135
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-018",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "7d31de11173a7fd2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the body of a dry Provençal rosé (e.g. Côtes de Provence Grenache-led) on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"item\": \"dry Provençal rosé, 12.5% abv, minimal skin contact\",\n  \"attribute\": \"body\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 3.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "3.5",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": 3.5,
            "expected": 3.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 352,
            "output_tokens": 6,
            "latency_ms": 934.4,
            "usd": 0.0005,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "4.0",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 4.0,
            "expected": 3.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 352,
            "output_tokens": 6,
            "latency_ms": 1007.8,
            "usd": 0.0005,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "4.0",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 4.0,
            "expected": 3.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 352,
            "output_tokens": 6,
            "latency_ms": 752.8,
            "usd": 0.0005,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.7778,
      "score_summary": {
        "mean": 0.7778,
        "stdev": 0.1924,
        "sem": 0.1111,
        "ci95_lo": 0.56,
        "ci95_hi": 0.9956,
        "n": 3
      },
      "score_if_valid": 0.7778,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 352.0,
        "output_tokens_mean": 6.0,
        "latency_ms_mean": 898.333,
        "usd_sum": 0.0015
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-019",
      "category": "sensory",
      "capability": "basic_taste_vector",
      "difficulty": "hard",
      "cuisine_region": "korean",
      "scorer": "numeric_vector_mae",
      "context_needs": [
        "scales.basic_tastes"
      ],
      "prompt_template_hash": "2e7df899a0148ebd",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n---\n\nReturn the five basic tastes of well-fermented napa kimchi (6 weeks at 4°C after 2 days at room temp) as a JSON object.\n\nINPUT:\n{\n  \"ingredient\": \"napa kimchi, 6 weeks fermented, ripe\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "values": {
          "sweet": 1.0,
          "sour": 7.0,
          "salty": 6.0,
          "bitter": 1.5,
          "umami": 7.0
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"sweet\":2,\"sour\":8,\"salty\":6,\"bitter\":1,\"umami\":6}",
          "compliant": true,
          "score": 0.65,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 2,
                "expected": 1.0,
                "score": 0.5
              },
              "sour": {
                "predicted": 8,
                "expected": 7.0,
                "score": 0.5
              },
              "salty": {
                "predicted": 6,
                "expected": 6.0,
                "score": 1.0
              },
              "bitter": {
                "predicted": 1,
                "expected": 1.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 6,
                "expected": 7.0,
                "score": 0.5
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 28,
            "latency_ms": 2842.0,
            "usd": 0.000723,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"sweet\":1.5,\"sour\":7.5,\"salty\":6.5,\"bitter\":1,\"umami\":5.5}",
          "compliant": true,
          "score": 0.65,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 1.5,
                "expected": 1.0,
                "score": 0.75
              },
              "sour": {
                "predicted": 7.5,
                "expected": 7.0,
                "score": 0.75
              },
              "salty": {
                "predicted": 6.5,
                "expected": 6.0,
                "score": 0.75
              },
              "bitter": {
                "predicted": 1,
                "expected": 1.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 5.5,
                "expected": 7.0,
                "score": 0.25
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 36,
            "latency_ms": 931.4,
            "usd": 0.000803,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"sweet\":2,\"sour\":8,\"salty\":6,\"bitter\":1,\"umami\":6}",
          "compliant": true,
          "score": 0.65,
          "detail": {
            "per_key": {
              "sweet": {
                "predicted": 2,
                "expected": 1.0,
                "score": 0.5
              },
              "sour": {
                "predicted": 8,
                "expected": 7.0,
                "score": 0.5
              },
              "salty": {
                "predicted": 6,
                "expected": 6.0,
                "score": 1.0
              },
              "bitter": {
                "predicted": 1,
                "expected": 1.5,
                "score": 0.75
              },
              "umami": {
                "predicted": 6,
                "expected": 7.0,
                "score": 0.5
              }
            },
            "tolerance": 2.0
          },
          "metrics": {
            "input_tokens": 354,
            "output_tokens": 28,
            "latency_ms": 1214.3,
            "usd": 0.000723,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.65,
      "score_summary": {
        "mean": 0.65,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.65,
        "ci95_hi": 0.65,
        "n": 3
      },
      "score_if_valid": 0.65,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 354.0,
        "output_tokens_mean": 30.667,
        "latency_ms_mean": 1662.567,
        "usd_sum": 0.002249
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:sense-020",
      "category": "sensory",
      "capability": "structural_estimation",
      "difficulty": "medium",
      "cuisine_region": "portuguese",
      "scorer": "numeric_mae",
      "context_needs": [
        "scales.structural"
      ],
      "prompt_template_hash": "ca5b9ab85fe71acb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n---\n\nEstimate the effervescence of a typical young Vinho Verde (traditional style with slight spritz) on the OGS 0-10 scale.\n\nINPUT:\n{\n  \"item\": \"young Vinho Verde, slightly spritzig\",\n  \"attribute\": \"effervescence\"\n}\n\nANSWER FORMAT: decimal_0_10",
      "expected": {
        "value": 2.5,
        "tolerance": 1.5
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "3",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 2.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 344,
            "output_tokens": 4,
            "latency_ms": 720.4,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "3.0",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 2.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 344,
            "output_tokens": 6,
            "latency_ms": 888.5,
            "usd": 0.00049,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "3",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": 3.0,
            "expected": 2.5,
            "tolerance": 1.5
          },
          "metrics": {
            "input_tokens": 344,
            "output_tokens": 4,
            "latency_ms": 631.0,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 344.0,
        "output_tokens_mean": 4.667,
        "latency_ms_mean": 746.633,
        "usd_sum": 0.00143
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-001",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "4523d68b7da194f7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'fleur de sel' as an OGS ingredient. It MUST validate against the OGS ingredient schema (schema/ingredient.schema.json). Respond with ONLY the JSON object, no markdown fences.\n\nINPUT:\n{\n  \"ingredient_name\": \"Fleur de sel\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"fleur-de-sel\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:fleur-de-sel",
          "ogs_version": "0.1.0",
          "category": "salt"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:fleur-de-sel\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"fr\",\n  \"available_languages\": [\"fr\", \"en\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"Fleur de sel\",\n        \"language\": \"fr\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Fleur de sel\",\n        \"language\": \"en\",\n        \"role\": \"loan\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"Hand-harvested surface sea salt crystals prized for their delicate texture, clean salinity, and subtle marine character.\",\n    \"fr\": \"Cristaux de sel marin de surface récoltés à la main, appréciés pour leur texture délicate, leur salinité nette et leur léger caractère marin.\"\n  },\n  \"category\": \"salt\",\n  \"subcategory\": \"sea salt\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 0,\n      \"sour\": 0,\n      \"salty\": 10,\n      \"bitter\": 0,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 1,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"sea spray\",\n        \"item\": \"brine\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"minerality\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\"delicate\", \"crystalline\", \"flaky\"],\n    \"finish_length\": 6,\n    \"complexity\": 3.5\n  },\n  \"origin\": {\n    \"region\": {\n      \"fr\": \"Côtes atlantiques françaises\",\n      \"en\": \"French Atlantic coast\"\n    },\n    \"season\": {\n      \"fr\": \"Été\",\n      \"en\": \"Summer\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Traditionally collected from the thin surface layer of sea salt pans during warm, dry, lightly breezy conditions.\",\n      \"fr\": \"Traditionnellement récolté à la surface des marais salants par temps chaud, sec et légèrement venteux.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Used primarily as a finishing salt for its fragile crystals and quick-dissolving crunch.\",\n    \"fr\": \"Utilisé principalement comme sel de finition pour ses cristaux fragiles et son croquant qui fond rapidement.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS generated entry\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:fleur-de-sel",
                "expected": "ogs:core:ingredient:fleur-de-sel",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "salt",
                "expected": "salt",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2513,
            "output_tokens": 649,
            "latency_ms": 9187.4,
            "usd": 0.009631,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:fleur-de-sel\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"fr\",\n  \"available_languages\": [\"fr\", \"en\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"Fleur de sel\",\n        \"language\": \"fr\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Fleur de sel\",\n        \"language\": \"en\",\n        \"role\": \"loan\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"Hand-harvested sea salt crystals formed on the surface of evaporating salt ponds, prized for delicate texture and clean saline taste.\"\n  },\n  \"category\": \"salt\",\n  \"subcategory\": \"sea salt\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 0,\n      \"sour\": 0,\n      \"salty\": 10,\n      \"bitter\": 0.5,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 2,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"sea breeze\",\n        \"item\": \"brine\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"wet stone\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\"crystalline\", \"delicate\", \"flaky\"],\n    \"finish_length\": 6,\n    \"complexity\": 3.5\n  },\n  \"origin\": {\n    \"region\": {\n      \"fr\": \"Guérande, Camargue et autres marais salants côtiers\"\n    },\n    \"season\": {\n      \"en\": \"Typically harvested during warm, dry summer periods\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Character reflects sea water composition, wind exposure, and traditional salt-pan harvesting methods.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Often used as a finishing salt because its fragile crystals provide a gentle crunch and dissolve quickly on the palate.\"\n  },\n  \"metadata\": {\n    \"source\": \"Open Gastronomy generated entry\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:fleur-de-sel",
                "expected": "ogs:core:ingredient:fleur-de-sel",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "salt",
                "expected": "salt",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2513,
            "output_tokens": 546,
            "latency_ms": 7455.8,
            "usd": 0.008601,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:fleur-de-sel\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"fr\",\n  \"available_languages\": [\"fr\", \"en\"],\n  \"name\": {\n    \"fr\": \"Fleur de sel\",\n    \"en\": \"Fleur de sel\"\n  },\n  \"description\": {\n    \"en\": \"A delicate hand-harvested sea salt collected as thin crystals from the surface of salt pans, prized for its clean salinity, subtle marine character, and crisp texture.\",\n    \"fr\": \"Un sel de mer délicat récolté à la main sous forme de fins cristaux à la surface des marais salants, apprécié pour sa salinité nette, son léger caractère marin et sa texture croquante.\"\n  },\n  \"category\": \"salt\",\n  \"subcategory\": \"sea salt\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 0,\n      \"sour\": 0,\n      \"salty\": 10,\n      \"bitter\": 0,\n      \"umami\": 0.5\n    },\n    \"structural\": {\n      \"body\": 1,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,\n      \"astringency\": 0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"sea breeze\",\n        \"item\": \"brine\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"wet stone\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\"crystalline\", \"crisp\", \"delicate\"],\n    \"finish_length\": 6,\n    \"complexity\": 3\n  },\n  \"origin\": {\n    \"region\": {\n      \"fr\": \"Marais salants côtiers\",\n      \"en\": \"Coastal salt marshes\"\n    },\n    \"season\": {\n      \"fr\": \"Récolte estivale par temps sec\",\n      \"en\": \"Summer harvest in dry weather\"\n    },\n    \"terroir_notes\": {\n      \"fr\": \"Le vent, l'évaporation solaire et la composition minérale des bassins influencent la taille des cristaux et les nuances marines.\",\n      \"en\": \"Wind, solar evaporation, and basin mineral composition influence crystal formation and subtle marine nuances.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Typically used as a finishing salt rather than for bulk seasoning.\",\n    \"fr\": \"S'utilise généralement comme sel de finition plutôt que pour l'assaisonnement de base.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS generated entry\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:fleur-de-sel",
                "expected": "ogs:core:ingredient:fleur-de-sel",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "salt",
                "expected": "salt",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2513,
            "output_tokens": 632,
            "latency_ms": 8898.0,
            "usd": 0.009461,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2513.0,
        "output_tokens_mean": 609.0,
        "latency_ms_mean": 8513.733,
        "usd_sum": 0.027693
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-002",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "367b58556573440d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing a Sancerre (Sauvignon Blanc, Loire Valley) that validates against the OGS beverage schema. Respond with ONLY the JSON object, no markdown fences.\n\nINPUT:\n{\n  \"beverage_name\": \"Sancerre\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"sancerre-generic\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:sancerre-generic",
          "ogs_version": "0.1.0",
          "type": "wine"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:beverage:sancerre-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"fr\"],\n  \"name\": {\n    \"en\": \"Sancerre\",\n    \"fr\": \"Sancerre\"\n  },\n  \"description\": {\n    \"en\": \"A generic Sancerre white wine from the Loire Valley, typically made from Sauvignon Blanc and known for crisp acidity, light to medium body, and citrus, green fruit, herbal, and mineral notes.\",\n    \"fr\": \"Un vin blanc générique de Sancerre dans la vallée de la Loire, généralement élaboré à partir de Sauvignon Blanc et connu pour son acidité vive, son corps léger à moyen, et ses notes d'agrumes, de fruits verts, d'herbes et de minéralité.\"\n  },\n  \"type\": \"wine\",\n  \"subtype\": \"white wine\",\n  \"grape_varieties\": [\"Sauvignon Blanc\"],\n  \"region\": {\n    \"en\": \"Loire Valley\",\n    \"fr\": \"Vallée de la Loire\"\n  },\n  \"appellation\": {\n    \"en\": \"Sancerre\",\n    \"fr\": \"Sancerre\"\n  },\n  \"abv\": 12.5,\n  \"residual_sugar\": 2.0,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 10\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1,\n      \"sour\": 7.5,\n      \"salty\": 0.5,\n      \"bitter\": 1.5,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 4,\n      \"acidity_perception\": 8,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 2,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"grapefruit\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"zest\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"tree fruit\",\n        \"item\": \"green apple\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"fresh herbs\",\n        \"item\": \"grass\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"fresh herbs\",\n        \"item\": \"gooseberry leaf\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"flint\",\n        \"intensity\": 5.5\n      }\n    ],\n    \"texture\": [\"crisp\", \"lean\", \"clean\"],\n    \"finish_length\": 6.5,\n    \"complexity\": 6\n  },\n  \"notes\": {\n    \"en\": \"Representative profile of a classic dry Sancerre Blanc.\",\n    \"fr\": \"Profil représentatif d'un Sancerre Blanc sec classique.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS benchmark generated example\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:sancerre-generic",
                "expected": "ogs:core:beverage:sancerre-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2514,
            "output_tokens": 824,
            "latency_ms": 10173.0,
            "usd": 0.011383,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"id\":\"ogs:core:beverage:sancerre-generic\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"name\":\"Sancerre\",\"type\":\"wine\",\"subtype\":\"white wine\",\"grape_varieties\":[\"Sauvignon Blanc\"],\"region\":\"Loire Valley\",\"appellation\":\"Sancerre\",\"sensory_profile\":{\"basic_tastes\":{\"sweet\":1,\"sour\":7,\"salty\":0,\"bitter\":2,\"umami\":0},\"structural\":{\"body\":4,\"acidity_perception\":8,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":3,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lemon\",\"qualifier\":\"fresh\",\"intensity\":7},{\"family\":\"fruit\",\"group\":\"orchard\",\"item\":\"green apple\",\"qualifier\":\"fresh\",\"intensity\":6},{\"family\":\"fruit\",\"group\":\"tropical\",\"item\":\"grapefruit\",\"qualifier\":\"zest\",\"intensity\":6},{\"family\":\"herbal\",\"group\":\"fresh herb\",\"item\":\"grass\",\"qualifier\":\"fresh\",\"intensity\":5},{\"family\":\"vegetal\",\"group\":\"green\",\"item\":\"bell pepper\",\"qualifier\":\"fresh\",\"intensity\":3},{\"family\":\"earth\",\"group\":\"mineral\",\"item\":\"flint\",\"intensity\":5}],\"texture\":[\"crisp\",\"lean\",\"steely\"],\"finish_length\":6,\"complexity\":6},\"notes\":\"A classic Loire Valley Sauvignon Blanc with high acidity, light-to-medium body, citrus and green fruit character, and a distinct mineral edge.\"}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:sancerre-generic",
                "expected": "ogs:core:beverage:sancerre-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2514,
            "output_tokens": 340,
            "latency_ms": 4661.5,
            "usd": 0.006542,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:beverage:sancerre-generic\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\",\"fr\"],\"name\":{\"en\":\"Sancerre\"},\"description\":{\"en\":\"A generic Sancerre wine from the Loire Valley, typically made from Sauvignon Blanc and known for crisp acidity, citrus, green fruit, and mineral notes.\"},\"type\":\"wine\",\"subtype\":\"white\",\"grape_varieties\":[\"Sauvignon Blanc\"],\"region\":{\"en\":\"Loire Valley\"},\"appellation\":{\"en\":\"Sancerre\"},\"sensory_profile\":{\"basic_tastes\":{\"sweet\":1,\"sour\":7.5,\"salty\":0.5,\"bitter\":1.5,\"umami\":0},\"structural\":{\"body\":4.5,\"acidity_perception\":8,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":3,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lemon\",\"qualifier\":\"fresh\",\"intensity\":6.5},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"grapefruit\",\"qualifier\":\"zest\",\"intensity\":5.5},{\"family\":\"fruit\",\"group\":\"orchard\",\"item\":\"green apple\",\"qualifier\":\"fresh\",\"intensity\":5.5},{\"family\":\"vegetal\",\"group\":\"green\",\"item\":\"grass\",\"qualifier\":\"fresh\",\"intensity\":4.5},{\"family\":\"herbal\",\"group\":\"herb\",\"item\":\"gooseberry leaf\",\"qualifier\":\"fresh\",\"intensity\":3.5},{\"family\":\"earth\",\"group\":\"mineral\",\"item\":\"flint\",\"intensity\":5}],\"texture\":[\"crisp\",\"lean\",\"dry\"],\"finish_length\":6.5,\"complexity\":6},\"notes\":{\"en\":\"Commonly served chilled; classic examples are dry, bright, and mineral-driven.\"},\"metadata\":{\"source\":\"OGS benchmark synthesis\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:sancerre-generic",
                "expected": "ogs:core:beverage:sancerre-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2514,
            "output_tokens": 414,
            "latency_ms": 6873.6,
            "usd": 0.007282,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2514.0,
        "output_tokens_mean": 526.0,
        "latency_ms_mean": 7236.033,
        "usd_sum": 0.025207
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-003",
      "category": "structured_output",
      "capability": "schema_valid_pairing",
      "difficulty": "expert",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.pairing",
        "vocab.match_types",
        "vocab.explanation_codes",
        "vocab.grade_scale",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "2740dc70d150e12a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS pairing schema (`schema/pairing.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/pairing.schema.json\",\n  \"title\": \"OGS Pairing\",\n  \"description\": \"A scored, explained pairing between a dish and a beverage.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"dish_ref\",\n    \"beverage_ref\",\n    \"match_types\",\n    \"overall_score\",\n    \"grade\",\n    \"explanation_codes\",\n    \"confidence\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:pairing:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"dish_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"beverage_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"match_types\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/match_type_entry\"\n      }\n    },\n    \"overall_score\": {\n      \"type\": \"integer\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"grade\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"A+\",\n        \"A\",\n        \"B+\",\n        \"B\",\n        \"C+\",\n        \"C\",\n        \"D\",\n        \"F\"\n      ]\n    },\n    \"explanation_codes\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/explanation_code_entry\"\n      }\n    },\n    \"risks\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/risk_entry\"\n      }\n    },\n    \"confidence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 1\n    },\n    \"evidence\": {\n      \"$ref\": \"#/$defs/evidence\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"match_type_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"type\",\n        \"score\"\n      ],\n      \"properties\": {\n        \"type\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"harmony\",\n            \"contrast\",\n            \"bridge\",\n            \"intensity_alignment\",\n            \"structural_balance\",\n            \"refresh\"\n          ]\n        },\n        \"score\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 100\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"primary\": {\n          \"type\": \"boolean\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"explanation_code_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"weight\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 1\n        },\n        \"detail\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"risk_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\",\n        \"severity\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"severity\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"low\",\n            \"medium\",\n            \"high\"\n          ]\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"mitigation\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"evidence\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"expert_panel\",\n            \"instrumental\",\n            \"computed\",\n            \"estimated\"\n          ]\n        },\n        \"sources\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Explanation Codes\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS pairing JSON document for the described match. It MUST validate against the OGS pairing schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"pairing_slug\": \"brie-champagne\",\n  \"dish_ref\": \"ogs:core:dish:brie-on-baguette\",\n  \"beverage_ref\": \"ogs:core:beverage:brut-champagne-nv\",\n  \"intent\": \"A classic refresh pairing driven by effervescence and acidity against creamy brie.\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/pairing.schema.json",
        "required_fields": {
          "id": "ogs:core:pairing:brie-champagne",
          "ogs_version": "0.1.0",
          "dish_ref": "ogs:core:dish:brie-on-baguette",
          "beverage_ref": "ogs:core:beverage:brut-champagne-nv"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:pairing:brie-champagne\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"name\":\"Brie on Baguette with Brut Champagne\",\"dish_ref\":\"ogs:core:dish:brie-on-baguette\",\"beverage_ref\":\"ogs:core:beverage:brut-champagne-nv\",\"match_types\":[{\"type\":\"refresh\",\"score\":95,\"primary\":true,\"description\":\"Effervescence and acidity cleanse the palate against the creamy, rich brie.\"},{\"type\":\"structural_balance\",\"score\":90,\"description\":\"The wine's acidity balances the cheese's fat and richness.\"},{\"type\":\"contrast\",\"score\":88,\"description\":\"A crisp, lean sparkling wine contrasts effectively with the soft, creamy texture of brie.\"},{\"type\":\"harmony\",\"score\":84,\"description\":\"The pairing is classic and balanced, with comparable elegance and complexity.\"},{\"type\":\"intensity_alignment\",\"score\":86,\"description\":\"Neither the brie nor the Champagne overwhelms the other; intensity is well matched.\"}],\"overall_score\":93,\"grade\":\"A\",\"explanation_codes\":[{\"code\":\"REFRESH_EFFERVESCE\",\"weight\":0.34},{\"code\":\"REFRESH_ACID\",\"weight\":0.26},{\"code\":\"STRUCT_ACIDITY_BALANCE\",\"weight\":0.2},{\"code\":\"CONTRAST_RICH_CRISP\",\"weight\":0.12},{\"code\":\"STRUCT_INTENSITY_MATCH\",\"weight\":0.08}],\"confidence\":0.92,\"evidence\":{\"method\":\"estimated\",\"notes\":\"Classic sparkling wine and creamy cheese pairing based on structural refresh, acidity, and texture contrast.\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:brie-champagne",
                "expected": "ogs:core:pairing:brie-champagne",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:brie-on-baguette",
                "expected": "ogs:core:dish:brie-on-baguette",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:brut-champagne-nv",
                "expected": "ogs:core:beverage:brut-champagne-nv",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3379,
            "output_tokens": 348,
            "latency_ms": 4907.8,
            "usd": 0.007704,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"id\":\"ogs:core:pairing:brie-champagne\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"name\":\"Brie on Baguette with Brut Champagne\",\"dish_ref\":\"ogs:core:dish:brie-on-baguette\",\"beverage_ref\":\"ogs:core:beverage:brut-champagne-nv\",\"match_types\":[{\"type\":\"refresh\",\"score\":95,\"primary\":true,\"description\":\"Effervescence and acidity refresh the palate against the creamy richness of brie.\"},{\"type\":\"structural_balance\",\"score\":90,\"description\":\"High acidity and light, brisk structure balance the soft, fatty cheese.\"},{\"type\":\"contrast\",\"score\":88,\"description\":\"The rich, creamy texture of brie is lifted by the crisp, dry sparkling wine.\"},{\"type\":\"harmony\",\"score\":84,\"description\":\"The pairing is classic and integrated, with neither element overwhelming the other.\"},{\"type\":\"intensity_alignment\",\"score\":86,\"description\":\"Both elements are moderate in intensity and remain clearly expressive.\"}],\"overall_score\":94,\"grade\":\"A\",\"explanation_codes\":[{\"code\":\"REFRESH_EFFERVESCE\",\"weight\":0.34},{\"code\":\"REFRESH_ACID\",\"weight\":0.28},{\"code\":\"CONTRAST_RICH_CRISP\",\"weight\":0.18},{\"code\":\"STRUCT_ACIDITY_BALANCE\",\"weight\":0.12},{\"code\":\"STRUCT_INTENSITY_MATCH\",\"weight\":0.08}],\"confidence\":0.93,\"evidence\":{\"method\":\"estimated\",\"notes\":\"Classic pairing logic based on creamy soft cheese balanced by brut sparkling wine acidity and mousse.\"},\"notes\":\"A classic refresh pairing driven by effervescence and acidity against creamy brie.\"}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:brie-champagne",
                "expected": "ogs:core:pairing:brie-champagne",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:brie-on-baguette",
                "expected": "ogs:core:dish:brie-on-baguette",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:brut-champagne-nv",
                "expected": "ogs:core:beverage:brut-champagne-nv",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3379,
            "output_tokens": 363,
            "latency_ms": 5958.6,
            "usd": 0.007854,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:pairing:brie-champagne\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"Brie on Baguette with Brut Champagne\",\n  \"description\": \"A classic pairing where Champagne's effervescence and acidity refresh the palate against creamy brie while matching the dish's elegance.\",\n  \"dish_ref\": \"ogs:core:dish:brie-on-baguette\",\n  \"beverage_ref\": \"ogs:core:beverage:brut-champagne-nv\",\n  \"match_types\": [\n    {\n      \"type\": \"refresh\",\n      \"score\": 94,\n      \"description\": \"Effervescence and acidity cleanse the palate after the rich, creamy cheese.\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 90,\n      \"description\": \"The wine's acidity balances the fat and creaminess of the brie.\",\n      \"primary\": false\n    },\n    {\n      \"type\": \"harmony\",\n      \"score\": 86,\n      \"description\": \"The pairing feels classic and elegant, with aligned complexity and texture.\",\n      \"primary\": false\n    },\n    {\n      \"type\": \"intensity_alignment\",\n      \"score\": 84,\n      \"description\": \"Neither the cheese nor the Champagne dominates; both are moderately intense.\",\n      \"primary\": false\n    }\n  ],\n  \"overall_score\": 93,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"REFRESH_EFFERVESCE\",\n      \"weight\": 0.34,\n      \"detail\": \"Champagne's bubbles scrub the palate between bites of creamy brie.\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.27,\n      \"detail\": \"Brisk acidity cuts through the cheese's richness and keeps the pairing lively.\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.2,\n      \"detail\": \"Acidity provides structural balance against fat and cream.\"\n    },\n    {\n      \"code\": \"HARMONY_COMPLEXITY_MATCH\",\n      \"weight\": 0.11,\n      \"detail\": \"Both elements are elegant and moderately complex, so neither outshines the other.\"\n    },\n    {\n      \"code\": \"STRUCT_INTENSITY_MATCH\",\n      \"weight\": 0.08,\n      \"detail\": \"The overall intensity is well aligned for a seamless pairing.\"\n    }\n  ],\n  \"confidence\": 0.94,\n  \"evidence\": {\n    \"method\": \"estimated\",\n    \"notes\": \"Assessment based on classic food and wine pairing principles for soft-ripened cheese and brut Champagne.\"\n  },\n  \"notes\": \"Best when the brie is served cool to lightly room temperature and the Champagne is well chilled.\"\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:brie-champagne",
                "expected": "ogs:core:pairing:brie-champagne",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:brie-on-baguette",
                "expected": "ogs:core:dish:brie-on-baguette",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:brut-champagne-nv",
                "expected": "ogs:core:beverage:brut-champagne-nv",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3379,
            "output_tokens": 633,
            "latency_ms": 8795.5,
            "usd": 0.010554,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3379.0,
        "output_tokens_mean": 448.0,
        "latency_ms_mean": 6553.967,
        "usd_sum": 0.026112
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-004",
      "category": "structured_output",
      "capability": "sensory_profile_json",
      "difficulty": "medium",
      "cuisine_region": "indian",
      "scorer": "json_match",
      "context_needs": [
        "scales.basic_tastes",
        "scales.structural",
        "vocab.aromatic_families",
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "9bd6860a8c5c69ec",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nProduce a JSON object representing the sensory profile for ripe mango. The object must have keys basic_tastes (with sweet, sour, salty, bitter, umami), structural (with body, acidity_perception, tannin, astringency, alcohol_heat, effervescence), and aromatics (array of entries with family, group, item, intensity). All intensities on 0-10. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"item\": \"ripe Alphonso mango, peak ripeness\"\n}\n\nANSWER FORMAT: json\n\nRESPONSE JSON SCHEMA:\n{\n  \"type\": \"object\",\n  \"required\": [\n    \"basic_tastes\",\n    \"structural\",\n    \"aromatics\"\n  ],\n  \"properties\": {\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"sweet\",\n        \"sour\",\n        \"salty\",\n        \"bitter\",\n        \"umami\"\n      ]\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"body\",\n        \"acidity_perception\",\n        \"tannin\",\n        \"astringency\",\n        \"alcohol_heat\",\n        \"effervescence\"\n      ]\n    },\n    \"aromatics\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"object\",\n        \"required\": [\n          \"family\",\n          \"group\",\n          \"item\",\n          \"intensity\"\n        ]\n      }\n    }\n  }\n}",
      "expected": {
        "value": {
          "basic_tastes": {
            "sweet": 8.5,
            "sour": 3.0,
            "salty": 0,
            "bitter": 0.5,
            "umami": 0.5
          },
          "structural": {
            "body": 5.5,
            "acidity_perception": 3.0,
            "tannin": 0,
            "astringency": 0.5,
            "alcohol_heat": 0,
            "effervescence": 0
          }
        },
        "required_keys": [
          "basic_tastes.sweet",
          "basic_tastes.sour",
          "structural.body",
          "structural.acidity_perception",
          "structural.tannin"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"basic_tastes\":{\"sweet\":8.5,\"sour\":3.5,\"salty\":0,\"bitter\":0.5,\"umami\":0.5},\"structural\":{\"body\":6.5,\"acidity_perception\":3.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"tropical\",\"item\":\"mango\",\"intensity\":9.5},{\"family\":\"fruit\",\"group\":\"stone fruit\",\"item\":\"peach\",\"intensity\":4},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"orange zest\",\"intensity\":2.5},{\"family\":\"floral\",\"group\":\"blossom\",\"item\":\"honeysuckle\",\"intensity\":2.5},{\"family\":\"confection\",\"group\":\"sweet\",\"item\":\"honey\",\"intensity\":3},{\"family\":\"herbal\",\"group\":\"fresh herb\",\"item\":\"green leaf\",\"intensity\":1.5}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.sweet": {
                "predicted": 8.5,
                "expected": 8.5,
                "match": true
              },
              "basic_tastes.sour": {
                "predicted": 3.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 6.5,
                "expected": 5.5,
                "match": true
              },
              "structural.acidity_perception": {
                "predicted": 3.5,
                "expected": 3.0,
                "match": true
              },
              "structural.tannin": {
                "predicted": 0,
                "expected": 0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 712,
            "output_tokens": 214,
            "latency_ms": 3614.2,
            "usd": 0.00303,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"basic_tastes\":{\"sweet\":8.5,\"sour\":3.5,\"salty\":0,\"bitter\":0.5,\"umami\":0.5},\"structural\":{\"body\":6.5,\"acidity_perception\":3.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"fresh\",\"item\":\"mango\",\"intensity\":9.5},{\"family\":\"fruit\",\"group\":\"overripe\",\"item\":\"tropical fruit\",\"intensity\":4},{\"family\":\"floral\",\"group\":\"fresh\",\"item\":\"blossom\",\"intensity\":3.5},{\"family\":\"confection\",\"group\":\"candied\",\"item\":\"honey\",\"intensity\":4.5},{\"family\":\"vegetal\",\"group\":\"fresh\",\"item\":\"green notes\",\"intensity\":1.5},{\"family\":\"spice\",\"group\":\"fresh\",\"item\":\"saffron\",\"intensity\":1.5},{\"family\":\"zest\",\"group\":\"fresh\",\"item\":\"citrus peel\",\"intensity\":1}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.sweet": {
                "predicted": 8.5,
                "expected": 8.5,
                "match": true
              },
              "basic_tastes.sour": {
                "predicted": 3.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 6.5,
                "expected": 5.5,
                "match": true
              },
              "structural.acidity_perception": {
                "predicted": 3.5,
                "expected": 3.0,
                "match": true
              },
              "structural.tannin": {
                "predicted": 0,
                "expected": 0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 712,
            "output_tokens": 236,
            "latency_ms": 3700.7,
            "usd": 0.00325,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"basic_tastes\":{\"sweet\":8.5,\"sour\":3.5,\"salty\":0,\"bitter\":0.5,\"umami\":0.5},\"structural\":{\"body\":6.5,\"acidity_perception\":3.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"tropical\",\"item\":\"mango\",\"intensity\":9.5},{\"family\":\"fruit\",\"group\":\"stone fruit\",\"item\":\"peach\",\"intensity\":4},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"orange zest\",\"intensity\":2.5},{\"family\":\"floral\",\"group\":\"blossom\",\"item\":\"honeysuckle\",\"intensity\":2.5},{\"family\":\"confection\",\"group\":\"sweet\",\"item\":\"honey\",\"intensity\":3},{\"family\":\"herbal\",\"group\":\"green\",\"item\":\"fresh herb\",\"intensity\":1.5}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.sweet": {
                "predicted": 8.5,
                "expected": 8.5,
                "match": true
              },
              "basic_tastes.sour": {
                "predicted": 3.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 6.5,
                "expected": 5.5,
                "match": true
              },
              "structural.acidity_perception": {
                "predicted": 3.5,
                "expected": 3.0,
                "match": true
              },
              "structural.tannin": {
                "predicted": 0,
                "expected": 0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 712,
            "output_tokens": 213,
            "latency_ms": 3254.1,
            "usd": 0.00302,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 712.0,
        "output_tokens_mean": 221.0,
        "latency_ms_mean": 3523.0,
        "usd_sum": 0.0093
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-005",
      "category": "structured_output",
      "capability": "schema_valid_dish",
      "difficulty": "expert",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.dish",
        "vocab.component_roles",
        "vocab.cooking_methods",
        "vocab.courses",
        "vocab.service_temperatures",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "9caf9bc474b2464d",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS dish schema (`schema/dish.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/dish.schema.json\",\n  \"title\": \"OGS Dish\",\n  \"description\": \"A composed dish with components, cooking methods, and overall sensory profile.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"components\",\n    \"overall_sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"cuisine\": {\n      \"type\": \"string\",\n      \"description\": \"Free-text cuisine name (v0.1) OR a cuisine entity reference matching ogs:<ns>:cuisine:<id> (v0.2+).\"\n    },\n    \"course\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"amuse\",\n        \"appetizer\",\n        \"soup\",\n        \"salad\",\n        \"fish\",\n        \"main\",\n        \"cheese\",\n        \"dessert\",\n        \"petit_four\"\n      ]\n    },\n    \"service_temperature\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"frozen\",\n        \"cold\",\n        \"cool\",\n        \"room\",\n        \"warm\",\n        \"hot\",\n        \"very_hot\"\n      ]\n    },\n    \"components\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/component\"\n      }\n    },\n    \"overall_sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"component\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"role\",\n        \"ingredient_ref\"\n      ],\n      \"properties\": {\n        \"role\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"main_protein\",\n            \"main_vegetable\",\n            \"starch\",\n            \"sauce\",\n            \"fat_component\",\n            \"acid_component\",\n            \"aromatic_garnish\",\n            \"seasoning\",\n            \"textural_element\"\n          ]\n        },\n        \"ingredient_ref\": {\n          \"type\": \"string\",\n          \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\"\n        },\n        \"cooking_method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"raw\",\n            \"seared\",\n            \"grilled\",\n            \"roasted\",\n            \"braised\",\n            \"poached\",\n            \"steamed\",\n            \"fried\",\n            \"confit\",\n            \"smoked\",\n            \"fermented\",\n            \"cured\",\n            \"pickled\",\n            \"dehydrated\",\n            \"sous_vide\",\n            \"blanched\",\n            \"sauteed\",\n            \"baked\",\n            \"charred\",\n            \"emulsified\"\n          ]\n        },\n        \"maillard_index\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"reduction_level\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"proportion\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"dominant\",\n            \"significant\",\n            \"accent\",\n            \"trace\"\n          ]\n        },\n        \"sensory_contribution\": {\n          \"type\": \"object\",\n          \"properties\": {\n            \"basic_tastes\": {\n              \"$ref\": \"#/$defs/basic_tastes\"\n            },\n            \"aromatics\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"$ref\": \"#/$defs/aromatic_entry\"\n              }\n            },\n            \"texture\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"type\": \"string\"\n              }\n            }\n          },\n          \"additionalProperties\": true\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS dish JSON document for 'moules marinieres' (mussels steamed with white wine, shallots, parsley, butter). It must validate against the OGS dish schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"dish_slug\": \"moules-marinieres\",\n  \"cuisine\": \"french\",\n  \"course\": \"main\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/dish.schema.json",
        "required_fields": {
          "id": "ogs:core:dish:moules-marinieres",
          "ogs_version": "0.1.0"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:dish:moules-marinieres\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"fr\",\n  \"available_languages\": [\"fr\", \"en\"],\n  \"name\": {\n    \"fr\": \"Moules marinières\",\n    \"en\": \"Mussels mariniere\"\n  },\n  \"description\": {\n    \"fr\": \"Moules cuites à la vapeur avec vin blanc, échalotes, persil et beurre.\",\n    \"en\": \"Mussels steamed with white wine, shallots, parsley, and butter.\"\n  },\n  \"cuisine\": \"french\",\n  \"course\": \"main\",\n  \"service_temperature\": \"hot\",\n  \"components\": [\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:mussels\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 4,\n          \"umami\": 6,\n          \"sweet\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"shellfish\",\n            \"item\": \"mussel\",\n            \"intensity\": 7\n          }\n        ],\n        \"texture\": [\"tender\", \"juicy\"]\n      },\n      \"notes\": {\n        \"en\": \"Primary shellfish component, gently steamed until just opened.\"\n      }\n    },\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:white-wine\",\n      \"cooking_method\": \"steamed\",\n      \"reduction_level\": 1,\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 4\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"orchard\",\n            \"item\": \"green apple\",\n            \"intensity\": 3\n          }\n        ],\n        \"texture\": [\"light\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:shallot\",\n      \"cooking_method\": \"sauteed\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 2,\n          \"pungent\": 0\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"shallot\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\"soft\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:parsley\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"bitter\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"soft herbs\",\n            \"item\": \"parsley\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"leafy\"]\n      }\n    },\n    {\n      \"role\": \"fat_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:butter\",\n      \"cooking_method\": \"emulsified\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"salty\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"dairy\",\n            \"group\": \"cultured fat\",\n            \"item\": \"butter\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\"silky\", \"rich\"]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 3.5,\n      \"salty\": 4.5,\n      \"bitter\": 1,\n      \"umami\": 6.5\n    },\n    \"structural\": {\n      \"body\": 4.5,\n      \"acidity_perception\": 3.5,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 1,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"shellfish\",\n        \"item\": \"mussel\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"allium\",\n        \"item\": \"shallot\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"soft herbs\",\n        \"item\": \"parsley\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"cultured fat\",\n        \"item\": \"butter\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"orchard\",\n        \"item\": \"green apple\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\"tender\", \"juicy\", \"brothy\", \"silky\"],\n    \"finish_length\": 5.5,\n    \"complexity\": 5.5\n  },\n  \"notes\": {\n    \"en\": \"Classic French preparation of mussels steamed open with white wine, shallots, parsley, and butter.\"\n  },\n  \"metadata\": {\n    \"source\": \"OpenAI OGS generation\"\n  }\n}",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "Additional properties are not allowed ('pungent' was unexpected)"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:moules-marinieres",
                "expected": "ogs:core:dish:moules-marinieres",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3079,
            "output_tokens": 1314,
            "latency_ms": 15792.0,
            "usd": 0.016989,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:dish:moules-marinieres\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"fr\",\n  \"available_languages\": [\"fr\", \"en\"],\n  \"name\": {\n    \"fr\": \"Moules marinières\",\n    \"en\": \"Mussels mariniere\"\n  },\n  \"description\": {\n    \"fr\": \"Moules cuites à la vapeur avec vin blanc, échalotes, persil et beurre.\",\n    \"en\": \"Mussels steamed with white wine, shallots, parsley, and butter.\"\n  },\n  \"cuisine\": \"french\",\n  \"course\": \"main\",\n  \"service_temperature\": \"hot\",\n  \"components\": [\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:mussels\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 4,\n          \"umami\": 7,\n          \"sweet\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"shellfish\",\n            \"item\": \"mussel\",\n            \"intensity\": 8\n          }\n        ],\n        \"texture\": [\"tender\", \"juicy\"]\n      }\n    },\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:white-wine\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 4,\n          \"bitter\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"orchard\",\n            \"item\": \"green apple\",\n            \"intensity\": 3\n          },\n          {\n            \"family\": \"floral\",\n            \"group\": \"white flower\",\n            \"item\": \"blossom\",\n            \"intensity\": 2\n          }\n        ],\n        \"texture\": [\"light\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:shallot\",\n      \"cooking_method\": \"sauteed\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 2,\n          \"pungent\": 0\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"shallot\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"soft\"]\n      },\n      \"notes\": \"Sweated in the cooking base.\"\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:parsley\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"bitter\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"soft herb\",\n            \"item\": \"parsley\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"leafy\"]\n      }\n    },\n    {\n      \"role\": \"fat_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:butter\",\n      \"cooking_method\": \"emulsified\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"salty\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"dairy\",\n            \"group\": \"cultured fat\",\n            \"item\": \"butter\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\"silky\", \"rich\"]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 3.5,\n      \"salty\": 4.5,\n      \"bitter\": 1,\n      \"umami\": 7\n    },\n    \"structural\": {\n      \"body\": 4.5,\n      \"acidity_perception\": 3.5,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 0.5,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"shellfish\",\n        \"item\": \"mussel\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"allium\",\n        \"item\": \"shallot\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"soft herb\",\n        \"item\": \"parsley\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"cultured fat\",\n        \"item\": \"butter\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"orchard\",\n        \"item\": \"green apple\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\"tender\", \"juicy\", \"brothy\", \"silky\"],\n    \"finish_length\": 5.5,\n    \"complexity\": 5\n  },\n  \"notes\": {\n    \"en\": \"Classic preparation of mussels steamed open and served in their cooking liquor.\",\n    \"fr\": \"Préparation classique de moules ouvertes à la vapeur et servies dans leur jus de cuisson.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS generated\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "Additional properties are not allowed ('pungent' was unexpected)"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:moules-marinieres",
                "expected": "ogs:core:dish:moules-marinieres",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3079,
            "output_tokens": 1347,
            "latency_ms": 15304.7,
            "usd": 0.017319,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:dish:moules-marinieres\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"fr\",\n  \"available_languages\": [\"fr\", \"en\"],\n  \"name\": {\n    \"fr\": \"Moules marinières\",\n    \"en\": \"Mussels marinière\"\n  },\n  \"description\": {\n    \"fr\": \"Moules cuites à la vapeur avec vin blanc, échalotes, persil et beurre.\",\n    \"en\": \"Mussels steamed with white wine, shallots, parsley, and butter.\"\n  },\n  \"cuisine\": \"french\",\n  \"course\": \"main\",\n  \"service_temperature\": \"hot\",\n  \"components\": [\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:mussels\",\n      \"cooking_method\": \"steamed\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 4,\n          \"umami\": 7,\n          \"sweet\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"shellfish\",\n            \"item\": \"mussel\",\n            \"intensity\": 8\n          }\n        ],\n        \"texture\": [\"tender\", \"juicy\"]\n      }\n    },\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:white-wine\",\n      \"cooking_method\": \"steamed\",\n      \"reduction_level\": 1,\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sour\": 4,\n          \"bitter\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"fruit\",\n            \"group\": \"orchard\",\n            \"item\": \"green apple\",\n            \"intensity\": 3\n          },\n          {\n            \"family\": \"floral\",\n            \"group\": \"white_flower\",\n            \"item\": \"blossom\",\n            \"intensity\": 2\n          }\n        ],\n        \"texture\": [\"light\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:shallot\",\n      \"cooking_method\": \"sauteed\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 2,\n          \"umami\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"shallot\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"soft\"]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:parsley\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"bitter\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"herbal\",\n            \"group\": \"leafy_herb\",\n            \"item\": \"parsley\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"fresh\"]\n      }\n    },\n    {\n      \"role\": \"fat_component\",\n      \"ingredient_ref\": \"ogs:core:ingredient:butter\",\n      \"cooking_method\": \"emulsified\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"salty\": 1\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"dairy\",\n            \"group\": \"cultured_fat\",\n            \"item\": \"butter\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\"silky\", \"rich\"]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 3.5,\n      \"salty\": 4.5,\n      \"bitter\": 1,\n      \"umami\": 7.5\n    },\n    \"structural\": {\n      \"body\": 5,\n      \"acidity_perception\": 4,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 0.5,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"marine\",\n        \"group\": \"shellfish\",\n        \"item\": \"mussel\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"allium\",\n        \"item\": \"shallot\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"leafy_herb\",\n        \"item\": \"parsley\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"cultured_fat\",\n        \"item\": \"butter\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"orchard\",\n        \"item\": \"green apple\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\"tender\", \"juicy\", \"silky\", \"brothy\"],\n    \"finish_length\": 6,\n    \"complexity\": 5.5\n  },\n  \"notes\": {\n    \"en\": \"Classic French mussel preparation served hot in its cooking broth.\",\n    \"fr\": \"Préparation classique de moules servies chaudes dans leur jus de cuisson.\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:moules-marinieres",
                "expected": "ogs:core:dish:moules-marinieres",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3079,
            "output_tokens": 1301,
            "latency_ms": 15935.5,
            "usd": 0.016859,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": -0.32,
        "ci95_hi": 0.9867,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3079.0,
        "output_tokens_mean": 1320.667,
        "latency_ms_mean": 15677.4,
        "usd_sum": 0.051167
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-006",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "medium",
      "cuisine_region": "iranian",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "bb4d0fcee2d1e751",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'saffron' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"saffron\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"saffron-stigma\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:saffron-stigma",
          "ogs_version": "0.1.0",
          "category": "spice"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:ingredient:saffron-stigma\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\"],\"name\":\"saffron\",\"description\":\"Dried stigmas of Crocus sativus used as a highly aromatic spice with floral, hay-like, honeyed, and earthy notes.\",\"category\":\"spice\",\"subcategory\":\"flower stigma\",\"sensory_profile\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0,\"salty\":0,\"bitter\":3,\"umami\":0},\"structural\":{\"body\":1,\"acidity_perception\":0,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"floral\",\"group\":\"flower\",\"item\":\"violet\",\"intensity\":4},{\"family\":\"earth\",\"group\":\"dry earth\",\"item\":\"hay\",\"intensity\":7},{\"family\":\"confection\",\"group\":\"sweet\",\"item\":\"honey\",\"intensity\":5},{\"family\":\"spice\",\"group\":\"warm spice\",\"item\":\"saffron\",\"intensity\":8}],\"texture\":[\"dry\",\"threadlike\",\"delicate\"],\"finish_length\":8,\"complexity\":9},\"origin\":{\"region\":\"Traditionally associated with Iran, Spain, India, and Greece.\",\"season\":\"Harvested in autumn.\",\"terroir_notes\":\"Quality varies with climate, soil, and drying methods, influencing intensity and purity of aroma.\"},\"notes\":\"Used in very small quantities for coloring and aroma in savory and sweet preparations.\",\"metadata\":{\"source\":\"OGS generated entry\",\"license\":\"CC-BY-4.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:saffron-stigma",
                "expected": "ogs:core:ingredient:saffron-stigma",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2501,
            "output_tokens": 341,
            "latency_ms": 5805.4,
            "usd": 0.006536,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:saffron-stigma\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"saffron\",\n  \"description\": \"Dried stigmas of Crocus sativus used as a highly aromatic and coloring spice.\",\n  \"category\": \"spice\",\n  \"subcategory\": \"flower stigma\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1,\n      \"sour\": 0,\n      \"salty\": 0,\n      \"bitter\": 3.5,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 1,\n      \"acidity_perception\": 0,\n      \"tannin\": 0,\n      \"astringency\": 1.5,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"floral\",\n        \"group\": \"dried flower\",\n        \"item\": \"saffron blossom\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 8.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"dry earth\",\n        \"item\": \"hay\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spice\",\n        \"item\": \"honeyed spice\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"dry herb\",\n        \"item\": \"tea-like\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\"dry\", \"threadlike\", \"delicate\"],\n    \"finish_length\": 8,\n    \"complexity\": 9\n  },\n  \"origin\": {\n    \"region\": \"Traditionally associated with Iran, Spain, Kashmir, and Greece.\",\n    \"season\": \"Harvested in autumn during crocus flowering.\",\n    \"terroir_notes\": \"Quality varies with climate, altitude, drying practice, and freshness.\"\n  },\n  \"notes\": \"Used in very small quantities for aroma, color, and subtle bitterness.\",\n  \"metadata\": {\n    \"source\": \"OGS generated entry\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:saffron-stigma",
                "expected": "ogs:core:ingredient:saffron-stigma",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2501,
            "output_tokens": 545,
            "latency_ms": 8074.1,
            "usd": 0.008576,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:ingredient:saffron-stigma\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\"],\"name\":\"saffron\",\"description\":\"Dried stigmas of Crocus sativus used as a spice, valued for intense color, floral aroma, and subtle bitter-honey notes.\",\"category\":\"spice\",\"subcategory\":\"flower stigma\",\"sensory_profile\":{\"basic_tastes\":{\"sweet\":1.5,\"sour\":0.5,\"salty\":0,\"bitter\":3.5,\"umami\":0.5},\"structural\":{\"body\":1,\"acidity_perception\":0.5,\"tannin\":0,\"astringency\":1.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"floral\",\"group\":\"flower\",\"item\":\"violet\",\"qualifier\":\"dried\",\"intensity\":6.5},{\"family\":\"spice\",\"group\":\"exotic spice\",\"item\":\"saffron\",\"intensity\":9.5},{\"family\":\"confection\",\"group\":\"sweet\",\"item\":\"honey\",\"intensity\":3.5},{\"family\":\"earth\",\"group\":\"dry earth\",\"item\":\"hay\",\"qualifier\":\"dried\",\"intensity\":5.5},{\"family\":\"herbal\",\"group\":\"tea herb\",\"item\":\"tea-like\",\"qualifier\":\"dried\",\"intensity\":3}],\"texture\":[\"dry\",\"threadlike\",\"delicate\"],\"finish_length\":7.5,\"complexity\":8.5},\"origin\":{\"region\":\"Traditionally cultivated in regions including Iran, India, Spain, and Greece.\",\"season\":\"Autumn harvest.\",\"terroir_notes\":\"Quality varies with climate, drying technique, and handling; often shows floral, hay-like, and honeyed nuances.\"},\"notes\":\"Used in very small quantities to provide golden color and a distinctive aromatic profile in savory and sweet preparations.\",\"metadata\":{\"source\":\"OpenAI generated\",\"license\":\"CC0-1.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:saffron-stigma",
                "expected": "ogs:core:ingredient:saffron-stigma",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2501,
            "output_tokens": 414,
            "latency_ms": 6836.2,
            "usd": 0.007266,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2501.0,
        "output_tokens_mean": 433.333,
        "latency_ms_mean": 6905.233,
        "usd_sum": 0.022378
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-007",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "0bc2e49ee6c1093e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'miso (red)' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"miso (red)\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"aka-miso\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:aka-miso",
          "ogs_version": "0.1.0",
          "category": "other"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:aka-miso\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"ja\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"miso (red)\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"aka miso\",\n        \"language\": \"ja\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"A dark, long-fermented Japanese soybean paste with pronounced savory depth, saltiness, and roasted, earthy notes.\"\n  },\n  \"category\": \"legume\",\n  \"subcategory\": \"fermented soybean paste\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 1,\n      \"salty\": 7.5,\n      \"bitter\": 1.5,\n      \"umami\": 8.5\n    },\n    \"structural\": {\n      \"body\": 6.5,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"fermented\",\n        \"item\": \"soybean paste\",\n        \"qualifier\": \"fermented\",\n        \"intensity\": 8.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"earthy\",\n        \"item\": \"earth\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"roasted\",\n        \"item\": \"toasted grain\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"caramelized\",\n        \"item\": \"molasses\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\"paste\", \"dense\", \"smooth\", \"slightly grainy\"],\n    \"finish_length\": 7.5,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Japan\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Red miso, often called aka miso, is typically fermented longer than lighter miso styles, yielding a darker color and more intense savory character.\"\n  },\n  \"metadata\": {\n    \"source\": \"Open Gastronomy Standard generated entry\"\n  }\n}",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:aka-miso",
                "expected": "ogs:core:ingredient:aka-miso",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "legume",
                "expected": "other",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 2501,
            "output_tokens": 595,
            "latency_ms": 8000.4,
            "usd": 0.009076,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:aka-miso\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"ja\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"miso (red)\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"aka miso\",\n        \"language\": \"ja\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"A dark, long-fermented Japanese soybean paste with pronounced savory depth, saltiness, and roasted, earthy notes.\"\n  },\n  \"category\": \"legume\",\n  \"subcategory\": \"fermented soybean paste\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 1,\n      \"salty\": 7.5,\n      \"bitter\": 1.5,\n      \"umami\": 9\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 1.5,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"fermented\",\n        \"item\": \"soy\",\n        \"qualifier\": \"fermented\",\n        \"intensity\": 9\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"earthy\",\n        \"item\": \"mushroom\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"roasted\",\n        \"item\": \"toast\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"caramelized\",\n        \"item\": \"molasses\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"nut\",\n        \"group\": \"legume\",\n        \"item\": \"soybean\",\n        \"intensity\": 6\n      }\n    ],\n    \"texture\": [\"paste\", \"dense\", \"smooth\", \"slightly grainy\"],\n    \"finish_length\": 8,\n    \"complexity\": 7.5\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Japan\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Typically produced from soybeans with rice or barley koji and salt; longer fermentation develops darker color and deeper savory intensity.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Red miso, often called aka miso, is generally aged longer than lighter miso styles and is commonly used in hearty soups, glazes, marinades, and braises.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS generated entry\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "'nut' is not one of ['fruit', 'floral', 'herbal', 'spice', 'earth', 'wood', 'dairy', 'savory', 'confection', 'vegetal', 'marine']"
            ],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:aka-miso",
                "expected": "ogs:core:ingredient:aka-miso",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "legume",
                "expected": "other",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 2501,
            "output_tokens": 682,
            "latency_ms": 9360.3,
            "usd": 0.009946,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:aka-miso\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"ja\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"miso (red)\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"aka miso\",\n        \"language\": \"ja\",\n        \"role\": \"alternate\"\n      }\n    ],\n    \"primary_language\": \"en\"\n  },\n  \"description\": {\n    \"en\": \"A dark, long-fermented Japanese soybean paste with a robust, salty, umami-rich character.\"\n  },\n  \"category\": \"legume\",\n  \"subcategory\": \"fermented soybean paste\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2,\n      \"sour\": 1,\n      \"salty\": 7.5,\n      \"bitter\": 1.5,\n      \"umami\": 8.5\n    },\n    \"structural\": {\n      \"body\": 7,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"fermented\",\n        \"item\": \"soybean paste\",\n        \"qualifier\": \"fermented\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"legume\",\n        \"item\": \"soybean\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"caramelized\",\n        \"item\": \"molasses\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"aged\",\n        \"item\": \"cedar\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\"paste\", \"dense\", \"smooth\"],\n    \"finish_length\": 7.5,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Japan\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Often associated with longer fermentation and deeper color development.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Commonly used in soups, marinades, glazes, and braises where strong savory depth is desired.\"\n  },\n  \"metadata\": {\n    \"source\": \"Open Gastronomy generated profile\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 0.5,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": false,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:aka-miso",
                "expected": "ogs:core:ingredient:aka-miso",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "legume",
                "expected": "other",
                "match": false
              }
            }
          },
          "metrics": {
            "input_tokens": 2501,
            "output_tokens": 607,
            "latency_ms": 7608.5,
            "usd": 0.009196,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.2887,
        "sem": 0.1667,
        "ci95_lo": 0.0067,
        "ci95_hi": 0.66,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2501.0,
        "output_tokens_mean": 628.0,
        "latency_ms_mean": 8323.067,
        "usd_sum": 0.028218
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-008",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "hard",
      "cuisine_region": "turkish",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "791ac1beebd95233",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'Urfa chili' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"Urfa chili\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"urfa-biber\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:urfa-biber",
          "ogs_version": "0.1.0",
          "category": "spice"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:urfa-biber\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"tr\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"Urfa chili\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Urfa biber\",\n        \"language\": \"tr\",\n        \"role\": \"alternate\"\n      },\n      {\n        \"value\": \"isot pepper\",\n        \"language\": \"en\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"A sun-dried and cured Turkish chili pepper from the Urfa region, known for its deep burgundy color, moderate heat, and raisin-like, smoky, earthy flavor.\"\n  },\n  \"category\": \"spice\",\n  \"subcategory\": \"chili pepper\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 1.0,\n      \"salty\": 0.0,\n      \"bitter\": 1.5,\n      \"umami\": 1.0\n    },\n    \"structural\": {\n      \"body\": 3.5,\n      \"acidity_perception\": 1.0,\n      \"tannin\": 0.0,\n      \"astringency\": 1.5,\n      \"alcohol_heat\": 0.0,\n      \"effervescence\": 0.0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"dried fruit\",\n        \"item\": \"raisin\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"soil\",\n        \"item\": \"earth\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"pepper\",\n        \"item\": \"chili\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"smoke\",\n        \"item\": \"smoke\",\n        \"intensity\": 5.0\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"cocoa\",\n        \"item\": \"dark chocolate\",\n        \"intensity\": 2.5\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"pepper\",\n        \"item\": \"red pepper\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4.5\n      }\n    ],\n    \"texture\": [\"flake\", \"slightly oily\", \"supple\", \"dry\"],\n    \"finish_length\": 7.0,\n    \"complexity\": 7.5\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Urfa, southeastern Türkiye\"\n    },\n    \"season\": {\n      \"en\": \"Harvested in late summer and traditionally sun-dried.\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Its character is shaped by hot, arid growing conditions and a curing process that develops dark fruit, earthy, and smoky notes.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Compared with many red chili flakes, Urfa chili is milder in apparent heat and more focused on savory, fruity depth.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS ingredient generation\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:urfa-biber",
                "expected": "ogs:core:ingredient:urfa-biber",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2500,
            "output_tokens": 797,
            "latency_ms": 10450.4,
            "usd": 0.011095,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:urfa-biber\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"tr\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"Urfa chili\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Urfa biber\",\n        \"language\": \"tr\",\n        \"role\": \"alternate\"\n      },\n      {\n        \"value\": \"İsot biberi\",\n        \"language\": \"tr\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"A sun-dried and cured Turkish chili pepper from the Urfa region, known for its deep red-black color, moderate heat, raisined fruit character, earthy depth, and subtle smoky, chocolate-like notes.\"\n  },\n  \"category\": \"spice\",\n  \"subcategory\": \"chili pepper\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 2,\n      \"umami\": 1.5\n    },\n    \"structural\": {\n      \"body\": 4,\n      \"acidity_perception\": 1.5,\n      \"tannin\": 0,\n      \"astringency\": 2,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"dried fruit\",\n        \"item\": \"raisin\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"dry earth\",\n        \"item\": \"earth\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"pepper\",\n        \"item\": \"chili\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"cocoa\",\n        \"item\": \"dark chocolate\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"smoke\",\n        \"item\": \"smoke\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"cured\",\n        \"item\": \"fermented pepper\",\n        \"qualifier\": \"fermented\",\n        \"intensity\": 3\n      }\n    ],\n    \"texture\": [\"flaky\", \"slightly oily\", \"supple\"],\n    \"finish_length\": 6.5,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Şanlıurfa, Türkiye\"\n    },\n    \"season\": {\n      \"en\": \"Late summer to autumn harvest; traditionally sun-dried and cured after picking\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Hot, dry southeastern Anatolian conditions contribute to concentrated fruit, dark color, and the distinctive cured, raisined profile.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Often used as flakes rather than powder, with a softer, rounder heat than many red chili varieties.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS generated entry\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:urfa-biber",
                "expected": "ogs:core:ingredient:urfa-biber",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2500,
            "output_tokens": 792,
            "latency_ms": 11180.0,
            "usd": 0.011045,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:urfa-biber\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"tr\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"Urfa chili\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Urfa biber\",\n        \"language\": \"tr\",\n        \"role\": \"alternate\"\n      },\n      {\n        \"value\": \"Isot pepper\",\n        \"language\": \"en\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"A Turkish chili pepper product from the Urfa region, typically sun-dried and cured to develop a dark color, mild-to-moderate heat, raisin-like fruitiness, earthy depth, and subtle smoky character.\"\n  },\n  \"category\": \"spice\",\n  \"subcategory\": \"chili pepper\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 1,\n      \"salty\": 0,\n      \"bitter\": 1.5,\n      \"umami\": 1\n    },\n    \"structural\": {\n      \"body\": 3.5,\n      \"acidity_perception\": 1,\n      \"tannin\": 0,\n      \"astringency\": 1.5,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"dried fruit\",\n        \"item\": \"raisin\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"soil\",\n        \"item\": \"earth\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"pepper\",\n        \"item\": \"chili\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"smoke\",\n        \"item\": \"smoke\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"cocoa\",\n        \"item\": \"cocoa\",\n        \"intensity\": 2.5\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"pepper\",\n        \"item\": \"red pepper\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\"flaky\", \"oily\", \"supple\"],\n    \"finish_length\": 6.5,\n    \"complexity\": 7\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Şanlıurfa, Türkiye\"\n    },\n    \"season\": {\n      \"en\": \"Late summer to autumn harvest; commonly dried and cured after harvest.\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Traditionally associated with the hot, dry climate of southeastern Türkiye, which supports ripening and post-harvest drying and curing.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Known for deep burgundy-to-purple color and a flavor profile that is more fruity and earthy than sharply hot.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS generated ingredient entry\",\n    \"license\": \"CC-BY-4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:urfa-biber",
                "expected": "ogs:core:ingredient:urfa-biber",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "spice",
                "expected": "spice",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2500,
            "output_tokens": 794,
            "latency_ms": 10401.9,
            "usd": 0.011065,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2500.0,
        "output_tokens_mean": 794.333,
        "latency_ms_mean": 10677.433,
        "usd_sum": 0.033205
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-009",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "easy",
      "cuisine_region": "american",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "a6a51cef64e43626",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'Meyer lemon' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"Meyer lemon\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"meyer-lemon\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:meyer-lemon",
          "ogs_version": "0.1.0",
          "category": "fruit"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:ingredient:meyer-lemon\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\"],\"name\":\"Meyer lemon\",\"description\":\"A hybrid citrus fruit known for its sweeter, less acidic profile than common lemon, with floral, mandarin-like aromatics.\",\"category\":\"fruit\",\"subcategory\":\"citrus\",\"sensory_profile\":{\"basic_tastes\":{\"sweet\":4,\"sour\":7,\"salty\":0,\"bitter\":2.5,\"umami\":0},\"structural\":{\"body\":3,\"acidity_perception\":7.5,\"tannin\":0,\"astringency\":1.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lemon\",\"qualifier\":\"fresh\",\"intensity\":8},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"mandarin\",\"qualifier\":\"fresh\",\"intensity\":6.5},{\"family\":\"floral\",\"group\":\"blossom\",\"item\":\"orange blossom\",\"qualifier\":\"fresh\",\"intensity\":4.5},{\"family\":\"vegetal\",\"group\":\"citrus peel\",\"item\":\"lemon zest\",\"qualifier\":\"zest\",\"intensity\":6}],\"texture\":[\"juicy\",\"tender\",\"thin-skinned\"],\"finish_length\":5.5,\"complexity\":6},\"origin\":{\"region\":\"Likely originating in China and popularized in California\",\"season\":\"Late fall through early spring\",\"terroir_notes\":\"Warmer growing conditions can emphasize sweetness and floral citrus character.\"},\"notes\":\"Commonly used in desserts, preserves, vinaigrettes, and cocktails where a softer lemon character is desired.\",\"metadata\":{\"source\":\"OpenAI culinary profile synthesis\",\"license\":\"CC-BY-4.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:meyer-lemon",
                "expected": "ogs:core:ingredient:meyer-lemon",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fruit",
                "expected": "fruit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2500,
            "output_tokens": 379,
            "latency_ms": 6019.3,
            "usd": 0.006915,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:meyer-lemon\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"Meyer lemon\",\n  \"description\": \"A thin-skinned citrus fruit, sweeter and less acidic than a common lemon, with floral, orange-like aromatic character.\",\n  \"category\": \"fruit\",\n  \"subcategory\": \"citrus\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 4.5,\n      \"sour\": 7.5,\n      \"salty\": 0,\n      \"bitter\": 2.5,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 2,\n      \"acidity_perception\": 7.5,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"orange\",\n        \"qualifier\": \"zest\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"blossom\",\n        \"item\": \"citrus blossom\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"sweet\",\n        \"item\": \"honey\",\n        \"intensity\": 2.5\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"peel\",\n        \"item\": \"citrus pith\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\"juicy\", \"tender\", \"thin-skinned\"],\n    \"finish_length\": 5.5,\n    \"complexity\": 6.5\n  },\n  \"origin\": {\n    \"region\": \"China\",\n    \"season\": \"Late fall through spring\",\n    \"terroir_notes\": \"Often cultivated in warm temperate to subtropical climates; aromatic expression increases with full ripeness.\"\n  },\n  \"notes\": \"A natural hybrid associated with lemon and mandarin/orange lineage; valued for balanced acidity and fragrant peel.\",\n  \"metadata\": {\n    \"source\": \"OGS generated entry\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:meyer-lemon",
                "expected": "ogs:core:ingredient:meyer-lemon",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fruit",
                "expected": "fruit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2500,
            "output_tokens": 582,
            "latency_ms": 7660.6,
            "usd": 0.008945,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:meyer-lemon\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\n    \"en\"\n  ],\n  \"name\": \"Meyer lemon\",\n  \"description\": \"A fragrant citrus fruit, sweeter and less acidic than common lemon, with floral and mandarin-like aromatic qualities.\",\n  \"category\": \"fruit\",\n  \"subcategory\": \"citrus\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3.5,\n      \"sour\": 7.5,\n      \"salty\": 0,\n      \"bitter\": 2.5,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 2,\n      \"acidity_perception\": 7.5,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"lemon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"mandarin\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"blossom\",\n        \"item\": \"orange blossom\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"citrus peel\",\n        \"item\": \"zest\",\n        \"qualifier\": \"zest\",\n        \"intensity\": 6\n      }\n    ],\n    \"texture\": [\n      \"juicy\",\n      \"tender\",\n      \"smooth-skinned\"\n    ],\n    \"finish_length\": 5.5,\n    \"complexity\": 6.5\n  },\n  \"origin\": {\n    \"region\": \"China, later popularized in California\",\n    \"season\": \"Late fall through spring\",\n    \"terroir_notes\": \"Warm, temperate growing conditions encourage sweetness, low harsh acidity, and pronounced floral-citrus perfume.\"\n  },\n  \"notes\": \"A hybrid citrus associated with lemon and mandarin parentage; valued for balanced juice, aromatic zest, and culinary versatility.\",\n  \"metadata\": {\n    \"source\": \"OGS generated entry\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:meyer-lemon",
                "expected": "ogs:core:ingredient:meyer-lemon",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fruit",
                "expected": "fruit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2500,
            "output_tokens": 570,
            "latency_ms": 7575.9,
            "usd": 0.008825,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2500.0,
        "output_tokens_mean": 510.333,
        "latency_ms_mean": 7085.267,
        "usd_sum": 0.024685
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-010",
      "category": "structured_output",
      "capability": "schema_valid_ingredient",
      "difficulty": "hard",
      "cuisine_region": "mexican",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.ingredient",
        "vocab.ingredient_categories",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "6d4d6d3cc23f3283",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS ingredient schema (`schema/ingredient.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/ingredient.schema.json\",\n  \"title\": \"OGS Ingredient\",\n  \"description\": \"An ingredient entity with sensory profile, category, and provenance metadata.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"category\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\",\n      \"description\": \"OGS identifier in the format ogs:<namespace>:ingredient:<id>\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\",\n      \"description\": \"OGS specification version (0.1.x or 0.2.x)\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\",\n      \"description\": \"Authoritative / endonym language of this entity's human-readable fields.\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      },\n      \"description\": \"Advisory list of languages covered by this entity's LString fields.\"\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"category\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"protein\",\n        \"vegetable\",\n        \"fruit\",\n        \"grain\",\n        \"dairy\",\n        \"fat\",\n        \"herb\",\n        \"spice\",\n        \"fungus\",\n        \"legume\",\n        \"nut\",\n        \"seed\",\n        \"sweetener\",\n        \"acid\",\n        \"salt\",\n        \"other\"\n      ]\n    },\n    \"subcategory\": {\n      \"type\": \"string\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"origin\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"region\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"season\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"terroir_notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Ingredient categories\n\nprotein, vegetable, fruit, grain, dairy, fat, herb, spice, fungus, legume, nut, seed, sweetener, acid, salt, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document that describes 'wild huitlacoche' as an OGS ingredient. It MUST validate against the OGS ingredient schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"ingredient_name\": \"wild huitlacoche\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"huitlacoche\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/ingredient.schema.json",
        "required_fields": {
          "id": "ogs:core:ingredient:huitlacoche",
          "ogs_version": "0.1.0",
          "category": "fungus"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:huitlacoche\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"es\"],\n  \"name\": {\n    \"primary_language\": \"es\",\n    \"labels\": [\n      {\n        \"value\": \"huitlacoche\",\n        \"language\": \"es\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"wild huitlacoche\",\n        \"language\": \"en\",\n        \"role\": \"alternate\"\n      },\n      {\n        \"value\": \"corn smut\",\n        \"language\": \"en\",\n        \"role\": \"descriptive\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"A wild edible corn fungus prized for its earthy, savory, and slightly sweet character.\"\n  },\n  \"category\": \"fungus\",\n  \"subcategory\": \"corn fungus\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 2.5,\n      \"sour\": 1,\n      \"salty\": 0.5,\n      \"bitter\": 1.5,\n      \"umami\": 7.5\n    },\n    \"structural\": {\n      \"body\": 5.5,\n      \"acidity_perception\": 1.5,\n      \"tannin\": 0,\n      \"astringency\": 1,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"forest floor\",\n        \"item\": \"mushroom\",\n        \"intensity\": 8\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"umami\",\n        \"item\": \"corn broth\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"corn\",\n        \"item\": \"roasted corn\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 6.5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"smoky\",\n        \"item\": \"char\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"lactic\",\n        \"item\": \"cream\",\n        \"intensity\": 2\n      }\n    ],\n    \"texture\": [\n      \"tender\",\n      \"silky\",\n      \"succulent\"\n    ],\n    \"finish_length\": 6.5,\n    \"complexity\": 7.5\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Mesoamerica\"\n    },\n    \"season\": {\n      \"en\": \"Typically harvested during the corn growing season.\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Develops on maize ears in field conditions, with expression influenced by corn variety, humidity, and maturity.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Often used in quesadillas, soups, tamales, and sautés.\"\n  },\n  \"metadata\": {\n    \"source\": \"Open Gastronomy generated entry\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:huitlacoche",
                "expected": "ogs:core:ingredient:huitlacoche",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fungus",
                "expected": "fungus",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2502,
            "output_tokens": 708,
            "latency_ms": 9559.1,
            "usd": 0.010208,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:ingredient:huitlacoche\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"es\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"wild huitlacoche\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"huitlacoche\",\n        \"language\": \"es\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"corn smut\",\n        \"language\": \"en\",\n        \"role\": \"alternate\"\n      }\n    ]\n  },\n  \"description\": {\n    \"en\": \"A wild fungus that grows on corn, prized in Mexican cuisine for its earthy, savory, mushroom-like flavor.\"\n  },\n  \"category\": \"fungus\",\n  \"subcategory\": \"corn fungus\",\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 1.0,\n      \"salty\": 0.5,\n      \"bitter\": 1.5,\n      \"umami\": 7.5\n    },\n    \"structural\": {\n      \"body\": 5.0,\n      \"acidity_perception\": 1.5,\n      \"tannin\": 0,\n      \"astringency\": 1.0,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"earth\",\n        \"group\": \"soil\",\n        \"item\": \"damp earth\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"mushroom\",\n        \"item\": \"black mushroom\",\n        \"intensity\": 8.0\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"corn\",\n        \"item\": \"roasted corn\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"smoke\",\n        \"item\": \"light smoke\",\n        \"intensity\": 2.5\n      }\n    ],\n    \"texture\": [\"tender\", \"silky\", \"slightly chewy\"],\n    \"finish_length\": 6.5,\n    \"complexity\": 7.5\n  },\n  \"origin\": {\n    \"region\": {\n      \"en\": \"Mexico\"\n    },\n    \"season\": {\n      \"en\": \"Rainy season, typically late summer to early autumn\"\n    },\n    \"terroir_notes\": {\n      \"en\": \"Develops on maize ears in humid field conditions; character varies with corn variety and maturity.\"\n    }\n  },\n  \"notes\": {\n    \"en\": \"Often treated as a delicacy, huitlacoche combines fungal depth with the vegetal sweetness of infected corn.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS generated entry\",\n    \"license\": \"CC BY 4.0\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:huitlacoche",
                "expected": "ogs:core:ingredient:huitlacoche",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fungus",
                "expected": "fungus",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2502,
            "output_tokens": 676,
            "latency_ms": 9688.8,
            "usd": 0.009888,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:ingredient:huitlacoche\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"es\",\"available_languages\":[\"es\",\"en\"],\"name\":{\"primary_language\":\"es\",\"labels\":[{\"value\":\"huitlacoche\",\"language\":\"es\",\"role\":\"preferred\"},{\"value\":\"wild huitlacoche\",\"language\":\"en\",\"role\":\"alternate\"},{\"value\":\"corn smut\",\"language\":\"en\",\"role\":\"alternate\"}]},\"description\":{\"en\":\"A wild edible fungus formed from infected maize kernels, valued for its earthy, savory, mushroom-like character.\",\"es\":\"Hongo comestible silvestre que se forma en granos de maíz infectados, apreciado por su carácter terroso, sabroso y similar al de los hongos.\"},\"category\":\"fungus\",\"subcategory\":\"corn smut\",\"sensory_profile\":{\"basic_tastes\":{\"sweet\":1,\"sour\":1,\"salty\":0,\"bitter\":1.5,\"umami\":7.5},\"structural\":{\"body\":5.5,\"acidity_perception\":1.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"earth\",\"group\":\"forest-floor\",\"item\":\"earth\",\"intensity\":7},{\"family\":\"savory\",\"group\":\"fungal\",\"item\":\"mushroom\",\"intensity\":8.5},{\"family\":\"vegetal\",\"group\":\"maize\",\"item\":\"corn husk\",\"qualifier\":\"fresh\",\"intensity\":5},{\"family\":\"savory\",\"group\":\"fermented\",\"item\":\"truffle-like\",\"intensity\":4.5}],\"texture\":[\"tender\",\"succulent\",\"slightly spongy\"],\"finish_length\":6.5,\"complexity\":7.5},\"origin\":{\"region\":{\"es\":\"México\",\"en\":\"Mexico\"},\"season\":{\"en\":\"Rainy season to early autumn\",\"es\":\"Temporada de lluvias a inicios de otoño\"},\"terroir_notes\":{\"en\":\"Typically gathered from maize fields; flavor varies with corn maturity and local growing conditions.\",\"es\":\"Suele recolectarse en milpas; el sabor varía según la madurez del maíz y las condiciones locales de cultivo.\"}},\"notes\":{\"en\":\"Often used in quesadillas, soups, tamales, and sautés.\",\"es\":\"Se utiliza con frecuencia en quesadillas, sopas, tamales y salteados.\"},\"metadata\":{\"source\":\"OpenAI gastronomic synthesis\",\"license\":\"CC BY 4.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:ingredient:huitlacoche",
                "expected": "ogs:core:ingredient:huitlacoche",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "category": {
                "predicted": "fungus",
                "expected": "fungus",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2502,
            "output_tokens": 528,
            "latency_ms": 9173.8,
            "usd": 0.008408,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2502.0,
        "output_tokens_mean": 637.333,
        "latency_ms_mean": 9473.9,
        "usd_sum": 0.028504
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-011",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "6615664480900680",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'Barolo' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"Barolo\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"barolo-docg\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:barolo-docg",
          "ogs_version": "0.1.0",
          "type": "wine"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:beverage:barolo-docg\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"it\",\"available_languages\":[\"it\",\"en\"],\"name\":{\"it\":\"Barolo\",\"en\":\"Barolo\"},\"description\":{\"it\":\"Vino rosso strutturato del Piemonte, prodotto principalmente da uve Nebbiolo.\",\"en\":\"Structured red wine from Piedmont, produced primarily from Nebbiolo grapes.\"},\"type\":\"wine\",\"subtype\":\"red wine\",\"grape_varieties\":[\"Nebbiolo\"],\"region\":{\"it\":\"Piemonte\",\"en\":\"Piedmont\"},\"appellation\":{\"it\":\"Barolo DOCG\",\"en\":\"Barolo DOCG\"},\"abv\":14,\"serve_temperature_c\":{\"min\":16,\"max\":18},\"sensory_profile\":{\"basic_tastes\":{\"sweet\":1,\"sour\":6.5,\"salty\":0,\"bitter\":4.5,\"umami\":0},\"structural\":{\"body\":8,\"acidity_perception\":7,\"tannin\":9,\"astringency\":7.5,\"alcohol_heat\":6,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"red fruit\",\"item\":\"cherry\",\"qualifier\":\"dried\",\"intensity\":6.5},{\"family\":\"fruit\",\"group\":\"red fruit\",\"item\":\"raspberry\",\"qualifier\":\"fresh\",\"intensity\":4.5},{\"family\":\"floral\",\"group\":\"flower\",\"item\":\"rose\",\"qualifier\":\"dried\",\"intensity\":6},{\"family\":\"spice\",\"group\":\"brown spice\",\"item\":\"anise\",\"intensity\":4.5},{\"family\":\"earth\",\"group\":\"soil\",\"item\":\"tar\",\"intensity\":5.5},{\"family\":\"wood\",\"group\":\"oak\",\"item\":\"cedar\",\"intensity\":4},{\"family\":\"herbal\",\"group\":\"herb\",\"item\":\"mint\",\"intensity\":3.5}],\"texture\":[\"firm\",\"dry\",\"gripping\"],\"finish_length\":8.5,\"complexity\":8.5},\"notes\":{\"en\":\"Typically powerful, ageworthy, and highly tannic, with notable aromatic complexity.\",\"it\":\"Tipicamente potente, longevo e molto tannico, con notevole complessità aromatica.\"},\"metadata\":{\"source\":\"OGS benchmark synthesis\",\"license\":\"CC-BY-4.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:barolo-docg",
                "expected": "ogs:core:beverage:barolo-docg",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2506,
            "output_tokens": 488,
            "latency_ms": 8593.2,
            "usd": 0.008012,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:beverage:barolo-docg\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"it\",\n  \"available_languages\": [\"it\", \"en\"],\n  \"name\": {\n    \"it\": \"Barolo\",\n    \"en\": \"Barolo\"\n  },\n  \"description\": {\n    \"it\": \"Vino rosso secco piemontese DOCG prodotto da uve Nebbiolo.\",\n    \"en\": \"A dry red DOCG wine from Piedmont made from Nebbiolo grapes.\"\n  },\n  \"type\": \"wine\",\n  \"subtype\": \"red wine\",\n  \"grape_varieties\": [\"Nebbiolo\"],\n  \"region\": {\n    \"it\": \"Piemonte\",\n    \"en\": \"Piedmont\"\n  },\n  \"appellation\": {\n    \"it\": \"Barolo DOCG\",\n    \"en\": \"Barolo DOCG\"\n  },\n  \"abv\": 14,\n  \"serve_temperature_c\": {\n    \"min\": 16,\n    \"max\": 18\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1,\n      \"sour\": 6,\n      \"salty\": 0,\n      \"bitter\": 4,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 8,\n      \"acidity_perception\": 7,\n      \"tannin\": 9,\n      \"astringency\": 8,\n      \"alcohol_heat\": 6,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"red fruit\",\n        \"item\": \"cherry\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"flowers\",\n        \"item\": \"rose\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"warm spice\",\n        \"item\": \"licorice\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"forest floor\",\n        \"item\": \"truffle\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"oak\",\n        \"item\": \"toast\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"dried herbs\",\n        \"item\": \"tar\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\"firm\", \"dry\", \"grippy\"],\n    \"finish_length\": 8,\n    \"complexity\": 9\n  },\n  \"notes\": {\n    \"en\": \"Typically powerful, structured, and age-worthy with pronounced tannin and high aromatic complexity.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS benchmark generation\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:barolo-docg",
                "expected": "ogs:core:beverage:barolo-docg",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2506,
            "output_tokens": 668,
            "latency_ms": 9274.9,
            "usd": 0.009813,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:beverage:barolo-docg\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"it\",\n  \"available_languages\": [\"it\", \"en\"],\n  \"name\": {\n    \"it\": \"Barolo\",\n    \"en\": \"Barolo\"\n  },\n  \"description\": {\n    \"en\": \"A prestigious Italian red wine from Piedmont made from Nebbiolo, known for high tannin, pronounced acidity, floral and tar-like aromatics, and long aging potential.\"\n  },\n  \"type\": \"wine\",\n  \"subtype\": \"red wine\",\n  \"grape_varieties\": [\"Nebbiolo\"],\n  \"region\": {\n    \"it\": \"Piemonte\",\n    \"en\": \"Piedmont\"\n  },\n  \"appellation\": {\n    \"it\": \"Barolo DOCG\",\n    \"en\": \"Barolo DOCG\"\n  },\n  \"abv\": 14,\n  \"serve_temperature_c\": {\n    \"min\": 16,\n    \"max\": 18\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1,\n      \"sour\": 7,\n      \"salty\": 0,\n      \"bitter\": 3,\n      \"umami\": 0\n    },\n    \"structural\": {\n      \"body\": 8,\n      \"acidity_perception\": 7.5,\n      \"tannin\": 9,\n      \"astringency\": 7.5,\n      \"alcohol_heat\": 5,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"floral\",\n        \"group\": \"flower\",\n        \"item\": \"rose\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"red fruit\",\n        \"item\": \"cherry\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"red fruit\",\n        \"item\": \"raspberry\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"brown spice\",\n        \"item\": \"licorice\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"earth\",\n        \"group\": \"mineral\",\n        \"item\": \"tar\",\n        \"intensity\": 7\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"dried herb\",\n        \"item\": \"mint\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"oak\",\n        \"item\": \"cedar\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\"firm\", \"structured\", \"grippy\"],\n    \"finish_length\": 9,\n    \"complexity\": 9\n  },\n  \"notes\": {\n    \"en\": \"Typically dry, powerful, and age-worthy, Barolo is often described as combining aromatic lift with strong tannic structure.\"\n  },\n  \"metadata\": {\n    \"source\": \"OGS benchmark synthesis\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:barolo-docg",
                "expected": "ogs:core:beverage:barolo-docg",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "wine",
                "expected": "wine",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2506,
            "output_tokens": 693,
            "latency_ms": 8829.4,
            "usd": 0.010063,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2506.0,
        "output_tokens_mean": 616.333,
        "latency_ms_mean": 8899.167,
        "usd_sum": 0.027888
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-012",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "hard",
      "cuisine_region": "scottish",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "bd330891a3e68b59",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'Islay single malt' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"Islay single malt\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"ardbeg-10-islay\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:ardbeg-10-islay",
          "ogs_version": "0.1.0",
          "type": "spirit"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:beverage:ardbeg-10-islay\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\"],\"name\":\"Islay single malt\",\"type\":\"spirit\",\"subtype\":\"single malt whisky\",\"region\":\"Islay, Scotland\",\"abv\":46,\"serve_temperature_c\":{\"min\":18,\"max\":22},\"sensory_profile\":{\"basic_tastes\":{\"sweet\":2.5,\"sour\":0.5,\"salty\":1.5,\"bitter\":2,\"umami\":0.5},\"structural\":{\"body\":6.5,\"acidity_perception\":0.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":6,\"effervescence\":0},\"aromatics\":[{\"family\":\"earth\",\"group\":\"smoke\",\"item\":\"peat\",\"intensity\":9},{\"family\":\"marine\",\"group\":\"coastal\",\"item\":\"sea spray\",\"intensity\":6.5},{\"family\":\"wood\",\"group\":\"oak\",\"item\":\"charred oak\",\"intensity\":5.5},{\"family\":\"spice\",\"group\":\"baking spice\",\"item\":\"black pepper\",\"intensity\":4.5},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lemon zest\",\"qualifier\":\"zest\",\"intensity\":4},{\"family\":\"confection\",\"group\":\"sweet\",\"item\":\"vanilla\",\"intensity\":3.5},{\"family\":\"herbal\",\"group\":\"medicinal\",\"item\":\"iodine\",\"intensity\":5.5}],\"texture\":[\"oily\",\"warming\"],\"finish_length\":8.5,\"complexity\":8},\"notes\":\"Classic Islay profile with pronounced peat smoke, maritime character, citrus lift, and a long warming finish.\",\"metadata\":{\"source\":\"OGS benchmark generation\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:ardbeg-10-islay",
                "expected": "ogs:core:beverage:ardbeg-10-islay",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "spirit",
                "expected": "spirit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2512,
            "output_tokens": 376,
            "latency_ms": 7405.4,
            "usd": 0.0069,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"id\":\"ogs:core:beverage:ardbeg-10-islay\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\"],\"name\":\"Islay single malt\",\"type\":\"spirit\",\"subtype\":\"single malt Scotch whisky\",\"region\":\"Islay, Scotland\",\"abv\":46,\"serve_temperature_c\":{\"min\":18,\"max\":22},\"sensory_profile\":{\"basic_tastes\":{\"sweet\":2.5,\"sour\":0.5,\"salty\":1.5,\"bitter\":2,\"umami\":0.5},\"structural\":{\"body\":6.5,\"acidity_perception\":0.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":6,\"effervescence\":0},\"aromatics\":[{\"family\":\"earth\",\"group\":\"smoke\",\"item\":\"peat smoke\",\"intensity\":9},{\"family\":\"marine\",\"group\":\"coastal\",\"item\":\"sea spray\",\"intensity\":6.5},{\"family\":\"wood\",\"group\":\"oak\",\"item\":\"charred oak\",\"intensity\":5.5},{\"family\":\"spice\",\"group\":\"baking spice\",\"item\":\"black pepper\",\"intensity\":4.5},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lemon zest\",\"qualifier\":\"zest\",\"intensity\":4},{\"family\":\"confection\",\"group\":\"sweet\",\"item\":\"vanilla\",\"intensity\":3.5},{\"family\":\"herbal\",\"group\":\"medicinal\",\"item\":\"iodine\",\"intensity\":5.5}],\"texture\":[\"oily\",\"warming\",\"smoky\"],\"finish_length\":8.5,\"complexity\":8},\"notes\":\"A strongly peated Islay single malt profile with maritime and smoky character.\",\"metadata\":{\"source\":\"OGS benchmark generated example\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:ardbeg-10-islay",
                "expected": "ogs:core:beverage:ardbeg-10-islay",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "spirit",
                "expected": "spirit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2512,
            "output_tokens": 377,
            "latency_ms": 5976.0,
            "usd": 0.00691,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:beverage:ardbeg-10-islay\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\"],\"name\":\"Islay single malt\",\"description\":\"Peated single malt Scotch whisky from Islay, known for smoky, maritime, medicinal, citrus, and malty characteristics.\",\"type\":\"spirit\",\"subtype\":\"single malt Scotch whisky\",\"region\":\"Islay, Scotland\",\"abv\":46,\"serve_temperature_c\":{\"min\":18,\"max\":22},\"sensory_profile\":{\"basic_tastes\":{\"sweet\":2.5,\"sour\":0.5,\"salty\":1.5,\"bitter\":2,\"umami\":1},\"structural\":{\"body\":6.5,\"acidity_perception\":0.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":6.5,\"effervescence\":0},\"aromatics\":[{\"family\":\"earth\",\"group\":\"smoke\",\"item\":\"peat smoke\",\"intensity\":9},{\"family\":\"marine\",\"group\":\"coastal\",\"item\":\"sea spray\",\"intensity\":6.5},{\"family\":\"herbal\",\"group\":\"medicinal\",\"item\":\"iodine\",\"intensity\":6},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lemon zest\",\"qualifier\":\"zest\",\"intensity\":4.5},{\"family\":\"wood\",\"group\":\"oak\",\"item\":\"charred oak\",\"intensity\":5},{\"family\":\"confection\",\"group\":\"sweet\",\"item\":\"vanilla\",\"intensity\":3.5},{\"family\":\"savory\",\"group\":\"malt\",\"item\":\"barley malt\",\"intensity\":5},{\"family\":\"spice\",\"group\":\"brown spice\",\"item\":\"black pepper\",\"intensity\":3.5}],\"texture\":[\"oily\",\"warming\",\"dry\"],\"finish_length\":8.5,\"complexity\":8},\"notes\":\"Representative profile of an Islay single malt with assertive peat smoke, maritime character, and firm alcoholic warmth.\",\"metadata\":{\"source\":\"OGS benchmark generation\",\"license\":\"CC-BY-4.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:ardbeg-10-islay",
                "expected": "ogs:core:beverage:ardbeg-10-islay",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "spirit",
                "expected": "spirit",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2512,
            "output_tokens": 436,
            "latency_ms": 6532.1,
            "usd": 0.0075,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2512.0,
        "output_tokens_mean": 396.333,
        "latency_ms_mean": 6637.833,
        "usd_sum": 0.02131
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-013",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "872d23f39307a374",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'junmai ginjo sake' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"junmai ginjo sake\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"junmai-ginjo-generic\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:junmai-ginjo-generic",
          "ogs_version": "0.1.0",
          "type": "sake"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:beverage:junmai-ginjo-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"junmai ginjo sake\",\n  \"type\": \"sake\",\n  \"subtype\": \"junmai ginjo\",\n  \"abv\": 15.5,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 14\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3,\n      \"sour\": 2.5,\n      \"salty\": 0.5,\n      \"bitter\": 1,\n      \"umami\": 4\n    },\n    \"structural\": {\n      \"body\": 4.5,\n      \"acidity_perception\": 3,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 2.5,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"orchard\",\n        \"item\": \"pear\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"tree fruit\",\n        \"item\": \"apple\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"white flower\",\n        \"item\": \"jasmine\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"sweet notes\",\n        \"item\": \"rice cream\",\n        \"intensity\": 2.5\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"cultured\",\n        \"item\": \"yogurt\",\n        \"intensity\": 1.5\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"grain\",\n        \"item\": \"steamed rice\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\n      \"silky\",\n      \"smooth\",\n      \"clean\"\n    ],\n    \"finish_length\": 5.5,\n    \"complexity\": 6\n  },\n  \"description\": \"A fragrant premium sake style made with rice, koji, water, and yeast, typically showing elegant fruit and floral aromatics, moderate umami, and a smooth, refined texture.\",\n  \"notes\": \"Generic profile for junmai ginjo sake.\",\n  \"metadata\": {\n    \"source\": \"OpenAI generated generic style profile\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:junmai-ginjo-generic",
                "expected": "ogs:core:beverage:junmai-ginjo-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "sake",
                "expected": "sake",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2515,
            "output_tokens": 601,
            "latency_ms": 8474.2,
            "usd": 0.009154,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:beverage:junmai-ginjo-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"junmai ginjo sake\",\n  \"description\": \"A premium junmai ginjo style sake made without added distilled alcohol, typically showing refined aromatics, light to medium body, and a clean, elegant finish.\",\n  \"type\": \"sake\",\n  \"subtype\": \"junmai ginjo\",\n  \"abv\": 15.5,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 14\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3.5,\n      \"sour\": 2.5,\n      \"salty\": 0.5,\n      \"bitter\": 1,\n      \"umami\": 4\n    },\n    \"structural\": {\n      \"body\": 4.5,\n      \"acidity_perception\": 3,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 2.5,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"orchard fruit\",\n        \"item\": \"pear\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"tree fruit\",\n        \"item\": \"apple\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"citrus\",\n        \"item\": \"melon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"white floral\",\n        \"item\": \"honeysuckle\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"blossom\",\n        \"item\": \"apple blossom\",\n        \"intensity\": 3\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"lactic\",\n        \"item\": \"cream\",\n        \"intensity\": 1.5\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"rice-derived\",\n        \"item\": \"steamed rice\",\n        \"intensity\": 4.5\n      }\n    ],\n    \"texture\": [\n      \"silky\",\n      \"clean\",\n      \"smooth\"\n    ],\n    \"finish_length\": 6,\n    \"complexity\": 6.5\n  },\n  \"notes\": \"Typically aromatic and polished in style, with fruity and floral notes balanced by gentle umami and a precise finish.\"\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:junmai-ginjo-generic",
                "expected": "ogs:core:beverage:junmai-ginjo-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "sake",
                "expected": "sake",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2515,
            "output_tokens": 640,
            "latency_ms": 9262.0,
            "usd": 0.009544,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:beverage:junmai-ginjo-generic\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"junmai ginjo sake\",\n  \"description\": \"A premium junmai ginjo style sake made without added distilled alcohol, typically fragrant, refined, and moderately light to medium-bodied.\",\n  \"type\": \"sake\",\n  \"subtype\": \"junmai ginjo\",\n  \"abv\": 15.5,\n  \"serve_temperature_c\": {\n    \"min\": 8,\n    \"max\": 14\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 3.5,\n      \"sour\": 2.5,\n      \"salty\": 0.5,\n      \"bitter\": 1,\n      \"umami\": 4.5\n    },\n    \"structural\": {\n      \"body\": 4.5,\n      \"acidity_perception\": 2.5,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 2.5,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"fruit\",\n        \"group\": \"orchard fruit\",\n        \"item\": \"pear\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5.5\n      },\n      {\n        \"family\": \"fruit\",\n        \"group\": \"tree fruit\",\n        \"item\": \"melon\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 5\n      },\n      {\n        \"family\": \"floral\",\n        \"group\": \"white flowers\",\n        \"item\": \"blossom\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"sweet notes\",\n        \"item\": \"rice candy\",\n        \"intensity\": 2.5\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"cultured dairy\",\n        \"item\": \"yogurt\",\n        \"intensity\": 1.5\n      },\n      {\n        \"family\": \"savory\",\n        \"group\": \"grain\",\n        \"item\": \"steamed rice\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\"silky\", \"smooth\"],\n    \"finish_length\": 5.5,\n    \"complexity\": 6\n  },\n  \"notes\": \"Commonly shows elegant ginjo aromatics with clean fruit, rice-driven umami, and a polished finish.\"\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:junmai-ginjo-generic",
                "expected": "ogs:core:beverage:junmai-ginjo-generic",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "sake",
                "expected": "sake",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2515,
            "output_tokens": 593,
            "latency_ms": 8339.1,
            "usd": 0.009074,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2515.0,
        "output_tokens_mean": 611.333,
        "latency_ms_mean": 8691.767,
        "usd_sum": 0.027772
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-014",
      "category": "structured_output",
      "capability": "schema_valid_beverage",
      "difficulty": "easy",
      "cuisine_region": "german",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.beverage",
        "vocab.beverage_types",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "3fe3fe74f8d80260",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS beverage schema (`schema/beverage.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/beverage.schema.json\",\n  \"title\": \"OGS Beverage\",\n  \"description\": \"A beverage entity with type, structural attributes, sensory profile, and provenance.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"type\",\n    \"sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"type\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"wine\",\n        \"beer\",\n        \"spirit\",\n        \"cider\",\n        \"sake\",\n        \"other\"\n      ]\n    },\n    \"subtype\": {\n      \"type\": \"string\"\n    },\n    \"grape_varieties\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"type\": \"string\"\n      }\n    },\n    \"vintage\": {\n      \"type\": \"integer\"\n    },\n    \"region\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"appellation\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"producer\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"abv\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"residual_sugar\": {\n      \"type\": \"number\",\n      \"minimum\": 0\n    },\n    \"serve_temperature_c\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"min\": {\n          \"type\": \"number\"\n        },\n        \"max\": {\n          \"type\": \"number\"\n        }\n      },\n      \"additionalProperties\": false,\n      \"description\": \"Recommended serving temperature range in degrees Celsius (SI canonical).\"\n    },\n    \"sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Beverage types\n\nwine, beer, spirit, cider, sake, other\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce a JSON document describing 'Oktoberfest Märzen lager' as an OGS beverage. It MUST validate against the OGS beverage schema. Respond with ONLY the JSON object.\n\nINPUT:\n{\n  \"beverage_name\": \"Oktoberfest Märzen lager\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"oktoberfest-marzen\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/beverage.schema.json",
        "required_fields": {
          "id": "ogs:core:beverage:oktoberfest-marzen",
          "ogs_version": "0.1.0",
          "type": "beer"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:beverage:oktoberfest-marzen\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\", \"de\"],\n  \"name\": {\n    \"labels\": [\n      {\n        \"value\": \"Oktoberfest Märzen lager\",\n        \"language\": \"en\",\n        \"role\": \"preferred\"\n      },\n      {\n        \"value\": \"Oktoberfest Märzen\",\n        \"language\": \"de\",\n        \"role\": \"alternate\"\n      }\n    ],\n    \"primary_language\": \"en\"\n  },\n  \"description\": {\n    \"en\": \"A traditional amber German lager associated with Oktoberfest, characterized by smooth malt richness, moderate bitterness, clean fermentation, and a rounded, medium body.\"\n  },\n  \"type\": \"beer\",\n  \"subtype\": \"märzen lager\",\n  \"abv\": 5.8,\n  \"serve_temperature_c\": {\n    \"min\": 6,\n    \"max\": 9\n  },\n  \"sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 4.5,\n      \"sour\": 1.5,\n      \"salty\": 0.5,\n      \"bitter\": 4,\n      \"umami\": 1\n    },\n    \"structural\": {\n      \"body\": 5.5,\n      \"acidity_perception\": 2,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 2,\n      \"effervescence\": 4.5\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"confection\",\n        \"group\": \"sweet bakery\",\n        \"item\": \"bread crust\",\n        \"intensity\": 6\n      },\n      {\n        \"family\": \"confection\",\n        \"group\": \"caramelized sugar\",\n        \"item\": \"toffee\",\n        \"intensity\": 4\n      },\n      {\n        \"family\": \"wood\",\n        \"group\": \"toasted\",\n        \"item\": \"toast\",\n        \"intensity\": 3.5\n      },\n      {\n        \"family\": \"spice\",\n        \"group\": \"hop spice\",\n        \"item\": \"pepper\",\n        \"intensity\": 2\n      },\n      {\n        \"family\": \"herbal\",\n        \"group\": \"noble hop\",\n        \"item\": \"herbal hop\",\n        \"intensity\": 2.5\n      }\n    ],\n    \"texture\": [\n      \"smooth\",\n      \"rounded\",\n      \"clean\"\n    ],\n    \"finish_length\": 5,\n    \"complexity\": 4.5\n  },\n  \"notes\": {\n    \"en\": \"Typically deep gold to amber, malt-forward but balanced, with a clean lager profile and restrained hop character.\"\n  },\n  \"metadata\": {\n    \"source\": \"Open Gastronomy Standard generated profile\"\n  }\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:oktoberfest-marzen",
                "expected": "ogs:core:beverage:oktoberfest-marzen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "beer",
                "expected": "beer",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2515,
            "output_tokens": 639,
            "latency_ms": 8731.2,
            "usd": 0.009534,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"id\":\"ogs:core:beverage:oktoberfest-marzen\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\",\"de\"],\"name\":{\"labels\":[{\"value\":\"Oktoberfest Märzen lager\",\"language\":\"en\",\"role\":\"preferred\"},{\"value\":\"Oktoberfest Märzen\",\"language\":\"de\",\"role\":\"alternate\"}],\"primary_language\":\"en\"},\"description\":{\"en\":\"A malt-forward amber lager associated with the Oktoberfest tradition, typically smooth, clean, and moderately rich with restrained bitterness.\"},\"type\":\"beer\",\"subtype\":\"lager\",\"abv\":5.8,\"serve_temperature_c\":{\"min\":6,\"max\":10},\"sensory_profile\":{\"basic_tastes\":{\"sweet\":4.5,\"sour\":1.5,\"salty\":0,\"bitter\":3,\"umami\":1},\"structural\":{\"body\":5.5,\"acidity_perception\":2,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":2,\"effervescence\":4.5},\"aromatics\":[{\"family\":\"confection\",\"group\":\"caramelized sugar\",\"item\":\"caramel\",\"intensity\":4},{\"family\":\"confection\",\"group\":\"baked goods\",\"item\":\"bread crust\",\"intensity\":5},{\"family\":\"wood\",\"group\":\"toast\",\"item\":\"toasted malt\",\"intensity\":4.5},{\"family\":\"spice\",\"group\":\"noble hop spice\",\"item\":\"peppery hop spice\",\"intensity\":2},{\"family\":\"herbal\",\"group\":\"hop herbal\",\"item\":\"herbal hop\",\"intensity\":2},{\"family\":\"fruit\",\"group\":\"orchard fruit\",\"item\":\"apple\",\"qualifier\":\"fresh\",\"intensity\":1.5}],\"texture\":[\"smooth\",\"round\",\"clean\"],\"finish_length\":5,\"complexity\":4.5},\"notes\":{\"en\":\"Commonly amber to deep gold in color, with a focus on elegant Munich-style malt character and high drinkability.\"},\"metadata\":{\"source\":\"OGS benchmark synthesis\",\"license\":\"CC-BY-4.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:oktoberfest-marzen",
                "expected": "ogs:core:beverage:oktoberfest-marzen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "beer",
                "expected": "beer",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2515,
            "output_tokens": 432,
            "latency_ms": 6876.8,
            "usd": 0.007464,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:beverage:oktoberfest-marzen\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"available_languages\":[\"en\",\"de\"],\"name\":{\"labels\":[{\"value\":\"Oktoberfest Märzen lager\",\"language\":\"en\",\"role\":\"preferred\"},{\"value\":\"Oktoberfest Märzen\",\"language\":\"de\",\"role\":\"alternate\"}],\"primary_language\":\"en\"},\"description\":{\"en\":\"A traditional amber German lager associated with Oktoberfest, showing smooth malt-forward character, moderate bitterness, clean fermentation, and restrained noble-hop aromatics.\"},\"type\":\"beer\",\"subtype\":\"märzen lager\",\"region\":{\"de\":\"Bayern\",\"en\":\"Bavaria\"},\"abv\":5.8,\"serve_temperature_c\":{\"min\":6,\"max\":9},\"sensory_profile\":{\"basic_tastes\":{\"sweet\":4.5,\"sour\":1.5,\"salty\":0.5,\"bitter\":4,\"umami\":1},\"structural\":{\"body\":5.5,\"acidity_perception\":2,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":2,\"effervescence\":4.5},\"aromatics\":[{\"family\":\"confection\",\"group\":\"caramelized sugar\",\"item\":\"toffee\",\"intensity\":3.5},{\"family\":\"confection\",\"group\":\"baked goods\",\"item\":\"bread crust\",\"intensity\":5},{\"family\":\"wood\",\"group\":\"toast\",\"item\":\"light toast\",\"intensity\":2.5},{\"family\":\"spice\",\"group\":\"noble hop spice\",\"item\":\"peppery spice\",\"intensity\":2},{\"family\":\"herbal\",\"group\":\"hops\",\"item\":\"herbal hop\",\"intensity\":2.5}],\"texture\":[\"smooth\",\"rounded\",\"clean\"],\"finish_length\":5,\"complexity\":4.5},\"notes\":{\"en\":\"Typical examples emphasize Munich malt richness, amber color, balanced bitterness, and a crisp lager finish.\"},\"metadata\":{\"source\":\"OGS generated profile\",\"license\":\"CC BY-SA 4.0\"}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:beverage:oktoberfest-marzen",
                "expected": "ogs:core:beverage:oktoberfest-marzen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "type": {
                "predicted": "beer",
                "expected": "beer",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 2515,
            "output_tokens": 427,
            "latency_ms": 6951.6,
            "usd": 0.007414,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 2515.0,
        "output_tokens_mean": 499.333,
        "latency_ms_mean": 7519.867,
        "usd_sum": 0.024412
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-015",
      "category": "structured_output",
      "capability": "schema_valid_pairing",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.pairing",
        "vocab.match_types",
        "vocab.explanation_codes",
        "vocab.grade_scale",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "f0c01fa20f35eb51",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS pairing schema (`schema/pairing.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/pairing.schema.json\",\n  \"title\": \"OGS Pairing\",\n  \"description\": \"A scored, explained pairing between a dish and a beverage.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"dish_ref\",\n    \"beverage_ref\",\n    \"match_types\",\n    \"overall_score\",\n    \"grade\",\n    \"explanation_codes\",\n    \"confidence\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:pairing:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"dish_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"beverage_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"match_types\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/match_type_entry\"\n      }\n    },\n    \"overall_score\": {\n      \"type\": \"integer\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"grade\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"A+\",\n        \"A\",\n        \"B+\",\n        \"B\",\n        \"C+\",\n        \"C\",\n        \"D\",\n        \"F\"\n      ]\n    },\n    \"explanation_codes\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/explanation_code_entry\"\n      }\n    },\n    \"risks\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/risk_entry\"\n      }\n    },\n    \"confidence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 1\n    },\n    \"evidence\": {\n      \"$ref\": \"#/$defs/evidence\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"match_type_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"type\",\n        \"score\"\n      ],\n      \"properties\": {\n        \"type\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"harmony\",\n            \"contrast\",\n            \"bridge\",\n            \"intensity_alignment\",\n            \"structural_balance\",\n            \"refresh\"\n          ]\n        },\n        \"score\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 100\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"primary\": {\n          \"type\": \"boolean\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"explanation_code_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"weight\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 1\n        },\n        \"detail\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"risk_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\",\n        \"severity\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"severity\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"low\",\n            \"medium\",\n            \"high\"\n          ]\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"mitigation\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"evidence\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"expert_panel\",\n            \"instrumental\",\n            \"computed\",\n            \"estimated\"\n          ]\n        },\n        \"sources\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Explanation Codes\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS pairing JSON document for the described match. It MUST validate against the OGS pairing schema. Respond with ONLY the JSON.\n\nINPUT:\n{\n  \"pairing_slug\": \"crottin-sancerre\",\n  \"dish_ref\": \"ogs:core:dish:crottin-de-chavignol\",\n  \"beverage_ref\": \"ogs:core:beverage:sancerre-blanc\",\n  \"intent\": \"Classic Loire regional pairing: Sancerre's acidity and citrus meet fresh goat cheese's tang.\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/pairing.schema.json",
        "required_fields": {
          "id": "ogs:core:pairing:crottin-sancerre",
          "ogs_version": "0.1.0",
          "dish_ref": "ogs:core:dish:crottin-de-chavignol",
          "beverage_ref": "ogs:core:beverage:sancerre-blanc"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\n  \"id\": \"ogs:core:pairing:crottin-sancerre\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"Crottin de Chavignol with Sancerre Blanc\",\n  \"description\": \"Classic Loire pairing of fresh goat cheese and Sancerre Blanc, driven by regional affinity, aligned intensity, and refreshing acidity.\",\n  \"dish_ref\": \"ogs:core:dish:crottin-de-chavignol\",\n  \"beverage_ref\": \"ogs:core:beverage:sancerre-blanc\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 92,\n      \"description\": \"Regional and flavor harmony make this a classic match.\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 93,\n      \"description\": \"The wine's acidity and body align closely with the cheese's tangy, creamy profile.\"\n    },\n    {\n      \"type\": \"refresh\",\n      \"score\": 94,\n      \"description\": \"Bright acidity keeps the palate fresh against the cheese's richness.\"\n    },\n    {\n      \"type\": \"intensity_alignment\",\n      \"score\": 90,\n      \"description\": \"Neither the cheese nor the wine overwhelms the other.\"\n    }\n  ],\n  \"overall_score\": 93,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"HARMONY_REGIONAL\",\n      \"weight\": 0.3,\n      \"detail\": \"Both originate from the Loire, forming a classic terroir-based pairing.\"\n    },\n    {\n      \"code\": \"HARMONY_FLAVOR_CONVERGE\",\n      \"weight\": 0.18,\n      \"detail\": \"Citrus and mineral notes in the wine align with the cheese's tangy freshness.\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.24,\n      \"detail\": \"The wine's acidity matches and lifts the cheese's lactic tang.\"\n    },\n    {\n      \"code\": \"STRUCT_BODY_ALIGN\",\n      \"weight\": 0.12,\n      \"detail\": \"Light-to-medium body suits the compact, creamy texture of fresh goat cheese.\"\n    },\n    {\n      \"code\": \"STRUCT_INTENSITY_MATCH\",\n      \"weight\": 0.08,\n      \"detail\": \"Both elements are vivid but not overpowering.\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.08,\n      \"detail\": \"Acidity refreshes the palate between bites.\"\n    }\n  ],\n  \"confidence\": 0.95,\n  \"evidence\": {\n    \"method\": \"estimated\",\n    \"notes\": \"Assessment based on canonical regional wine-and-cheese pairing logic and established sensory compatibility.\"\n  },\n  \"notes\": \"A benchmark classic: Sancerre Blanc's brisk acidity, citrus, and mineral profile complements the tangy, creamy character of Crottin de Chavignol.\"\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:crottin-sancerre",
                "expected": "ogs:core:pairing:crottin-sancerre",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:crottin-de-chavignol",
                "expected": "ogs:core:dish:crottin-de-chavignol",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:sancerre-blanc",
                "expected": "ogs:core:beverage:sancerre-blanc",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3381,
            "output_tokens": 665,
            "latency_ms": 9440.4,
            "usd": 0.010876,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"id\":\"ogs:core:pairing:crottin-sancerre\",\"ogs_version\":\"0.1.0\",\"primary_language\":\"en\",\"name\":\"Crottin de Chavignol with Sancerre Blanc\",\"dish_ref\":\"ogs:core:dish:crottin-de-chavignol\",\"beverage_ref\":\"ogs:core:beverage:sancerre-blanc\",\"match_types\":[{\"type\":\"harmony\",\"score\":92,\"primary\":true,\"description\":\"Classic regional affinity with converging tangy, mineral, and subtle herbal notes.\"},{\"type\":\"structural_balance\",\"score\":94,\"description\":\"Bright acidity and moderate body align well with the cheese's tang and creamy texture.\"},{\"type\":\"refresh\",\"score\":93,\"description\":\"The wine's acidity refreshes the palate and cuts through the cheese's richness.\"},{\"type\":\"intensity_alignment\",\"score\":90,\"description\":\"Both elements are high in flavor clarity but moderate in weight, so neither dominates.\"}],\"overall_score\":93,\"grade\":\"A\",\"explanation_codes\":[{\"code\":\"HARMONY_REGIONAL\",\"weight\":0.24,\"detail\":\"A benchmark Loire pairing, with both originating from the Sancerre/Chavignol area.\"},{\"code\":\"HARMONY_FLAVOR_CONVERGE\",\"weight\":0.2,\"detail\":\"Tangy, mineral, and citrus-driven profiles converge naturally.\"},{\"code\":\"STRUCT_ACIDITY_BALANCE\",\"weight\":0.24,\"detail\":\"The wine's acidity matches and lifts the cheese's tang while keeping the pairing fresh.\"},{\"code\":\"STRUCT_BODY_ALIGN\",\"weight\":0.14,\"detail\":\"The wine's medium-light body suits the cheese's fresh, creamy weight.\"},{\"code\":\"STRUCT_INTENSITY_MATCH\",\"weight\":0.08,\"detail\":\"Neither the cheese nor the wine overwhelms the other.\"},{\"code\":\"REFRESH_ACID\",\"weight\":0.1,\"detail\":\"Acidity cleanses the palate between bites of creamy goat cheese.\"}],\"confidence\":0.96,\"evidence\":{\"method\":\"estimated\",\"notes\":\"Assessment based on established classical regional pairing principles and typical sensory structure of fresh goat cheese and Sancerre Blanc.\"},\"notes\":\"A canonical regional match: fresh Crottin's lactic tang and light creaminess are sharpened and refreshed by Sancerre's citrusy acidity and mineral line.\"}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:crottin-sancerre",
                "expected": "ogs:core:pairing:crottin-sancerre",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:crottin-de-chavignol",
                "expected": "ogs:core:dish:crottin-de-chavignol",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:sancerre-blanc",
                "expected": "ogs:core:beverage:sancerre-blanc",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3381,
            "output_tokens": 487,
            "latency_ms": 7948.5,
            "usd": 0.009096,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\n  \"id\": \"ogs:core:pairing:crottin-sancerre\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"available_languages\": [\"en\"],\n  \"name\": \"Crottin de Chavignol with Sancerre Blanc\",\n  \"description\": \"Classic Loire regional pairing where bright, mineral Sancerre complements the tangy, fresh goat cheese.\",\n  \"dish_ref\": \"ogs:core:dish:crottin-de-chavignol\",\n  \"beverage_ref\": \"ogs:core:beverage:sancerre-blanc\",\n  \"match_types\": [\n    {\n      \"type\": \"harmony\",\n      \"score\": 93,\n      \"description\": \"Regional and flavor affinity create a classic benchmark pairing.\",\n      \"primary\": true\n    },\n    {\n      \"type\": \"structural_balance\",\n      \"score\": 92,\n      \"description\": \"The wine's acidity and light body balance the cheese's tangy richness.\"\n    },\n    {\n      \"type\": \"refresh\",\n      \"score\": 91,\n      \"description\": \"High acidity and mineral freshness cleanse the palate between bites.\"\n    },\n    {\n      \"type\": \"intensity_alignment\",\n      \"score\": 89,\n      \"description\": \"Both cheese and wine are moderately intense and finely detailed.\"\n    },\n    {\n      \"type\": \"bridge\",\n      \"score\": 86,\n      \"description\": \"Shared herbal and mineral nuances link the cheese and wine aromatically.\"\n    }\n  ],\n  \"overall_score\": 94,\n  \"grade\": \"A\",\n  \"explanation_codes\": [\n    {\n      \"code\": \"HARMONY_REGIONAL\",\n      \"weight\": 0.24,\n      \"detail\": \"Both come from the Loire, making this a canonical regional match.\"\n    },\n    {\n      \"code\": \"STRUCT_ACIDITY_BALANCE\",\n      \"weight\": 0.22,\n      \"detail\": \"Sancerre's acidity matches and lifts the cheese's tang while cutting through creaminess.\"\n    },\n    {\n      \"code\": \"REFRESH_ACID\",\n      \"weight\": 0.18,\n      \"detail\": \"The wine refreshes the palate after each bite.\"\n    },\n    {\n      \"code\": \"REFRESH_MINERAL\",\n      \"weight\": 0.14,\n      \"detail\": \"Mineral character adds a clean, stony reset.\"\n    },\n    {\n      \"code\": \"STRUCT_INTENSITY_MATCH\",\n      \"weight\": 0.12,\n      \"detail\": \"Neither element dominates; both sit at a similar intensity.\"\n    },\n    {\n      \"code\": \"AROMA_SHARED_HERBAL\",\n      \"weight\": 0.1,\n      \"detail\": \"Subtle herbal notes in the wine echo the cheese's rustic, grassy character.\"\n    }\n  ],\n  \"confidence\": 0.96,\n  \"evidence\": {\n    \"method\": \"estimated\",\n    \"notes\": \"Assessment based on classic Loire pairing conventions and standard structural behavior of fresh goat cheese with Sauvignon Blanc from Sancerre.\"\n  },\n  \"notes\": \"A textbook pairing, especially when the cheese is young to lightly matured.\"\n}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:crottin-sancerre",
                "expected": "ogs:core:pairing:crottin-sancerre",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:crottin-de-chavignol",
                "expected": "ogs:core:dish:crottin-de-chavignol",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:sancerre-blanc",
                "expected": "ogs:core:beverage:sancerre-blanc",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3381,
            "output_tokens": 678,
            "latency_ms": 9255.4,
            "usd": 0.011006,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3381.0,
        "output_tokens_mean": 610.0,
        "latency_ms_mean": 8881.433,
        "usd_sum": 0.030978
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-016",
      "category": "structured_output",
      "capability": "schema_valid_dish",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.dish",
        "vocab.component_roles",
        "vocab.cooking_methods",
        "vocab.courses",
        "vocab.service_temperatures",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "284b25ae8f0dc3ef",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS dish schema (`schema/dish.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/dish.schema.json\",\n  \"title\": \"OGS Dish\",\n  \"description\": \"A composed dish with components, cooking methods, and overall sensory profile.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"components\",\n    \"overall_sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"cuisine\": {\n      \"type\": \"string\",\n      \"description\": \"Free-text cuisine name (v0.1) OR a cuisine entity reference matching ogs:<ns>:cuisine:<id> (v0.2+).\"\n    },\n    \"course\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"amuse\",\n        \"appetizer\",\n        \"soup\",\n        \"salad\",\n        \"fish\",\n        \"main\",\n        \"cheese\",\n        \"dessert\",\n        \"petit_four\"\n      ]\n    },\n    \"service_temperature\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"frozen\",\n        \"cold\",\n        \"cool\",\n        \"room\",\n        \"warm\",\n        \"hot\",\n        \"very_hot\"\n      ]\n    },\n    \"components\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/component\"\n      }\n    },\n    \"overall_sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"component\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"role\",\n        \"ingredient_ref\"\n      ],\n      \"properties\": {\n        \"role\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"main_protein\",\n            \"main_vegetable\",\n            \"starch\",\n            \"sauce\",\n            \"fat_component\",\n            \"acid_component\",\n            \"aromatic_garnish\",\n            \"seasoning\",\n            \"textural_element\"\n          ]\n        },\n        \"ingredient_ref\": {\n          \"type\": \"string\",\n          \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\"\n        },\n        \"cooking_method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"raw\",\n            \"seared\",\n            \"grilled\",\n            \"roasted\",\n            \"braised\",\n            \"poached\",\n            \"steamed\",\n            \"fried\",\n            \"confit\",\n            \"smoked\",\n            \"fermented\",\n            \"cured\",\n            \"pickled\",\n            \"dehydrated\",\n            \"sous_vide\",\n            \"blanched\",\n            \"sauteed\",\n            \"baked\",\n            \"charred\",\n            \"emulsified\"\n          ]\n        },\n        \"maillard_index\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"reduction_level\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"proportion\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"dominant\",\n            \"significant\",\n            \"accent\",\n            \"trace\"\n          ]\n        },\n        \"sensory_contribution\": {\n          \"type\": \"object\",\n          \"properties\": {\n            \"basic_tastes\": {\n              \"$ref\": \"#/$defs/basic_tastes\"\n            },\n            \"aromatics\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"$ref\": \"#/$defs/aromatic_entry\"\n              }\n            },\n            \"texture\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"type\": \"string\"\n              }\n            }\n          },\n          \"additionalProperties\": true\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS dish JSON for tonkotsu ramen (pork-bone broth, wheat noodles, chashu, soft egg, nori, scallion). Respond with ONLY the JSON.\n\nINPUT:\n{\n  \"dish_name\": \"tonkotsu ramen\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"tonkotsu-ramen\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/dish.schema.json",
        "required_fields": {
          "id": "ogs:core:dish:tonkotsu-ramen",
          "ogs_version": "0.1.0",
          "course": "main"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:dish:tonkotsu-ramen\",\"ogs_version\":\"0.1.0\",\"name\":\"tonkotsu ramen\",\"cuisine\":\"Japanese\",\"course\":\"main\",\"service_temperature\":\"hot\",\"components\":[{\"role\":\"sauce\",\"ingredient_ref\":\"ogs:core:ingredient:pork-bone-broth\",\"cooking_method\":\"braised\",\"proportion\":\"dominant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0.5,\"salty\":5.5,\"bitter\":0.5,\"umami\":8.5},\"aromatics\":[{\"family\":\"savory\",\"group\":\"meat\",\"item\":\"pork\",\"qualifier\":\"cooked\",\"intensity\":8.5},{\"family\":\"dairy\",\"group\":\"cream\",\"item\":\"rich-fat\",\"intensity\":6.5}],\"texture\":[\"brothy\",\"silky\",\"rich\"]},\"notes\":\"Opaque long-simmered pork-bone broth.\"},{\"role\":\"starch\",\"ingredient_ref\":\"ogs:core:ingredient:wheat-noodles\",\"cooking_method\":\"boiled\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0,\"salty\":0.5,\"bitter\":0,\"umami\":0.5},\"aromatics\":[{\"family\":\"confection\",\"group\":\"grain\",\"item\":\"wheat\",\"intensity\":2}],\"texture\":[\"springy\",\"chewy\",\"slippery\"]}},{\"role\":\"main_protein\",\"ingredient_ref\":\"ogs:core:ingredient:chashu-pork\",\"cooking_method\":\"braised\",\"maillard_index\":1,\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":2,\"sour\":0,\"salty\":4.5,\"bitter\":0.5,\"umami\":7},\"aromatics\":[{\"family\":\"savory\",\"group\":\"meat\",\"item\":\"pork\",\"qualifier\":\"cooked\",\"intensity\":7},{\"family\":\"spice\",\"group\":\"allium\",\"item\":\"soy-braise\",\"intensity\":3}],\"texture\":[\"tender\",\"succulent\"]}},{\"role\":\"main_protein\",\"ingredient_ref\":\"ogs:core:ingredient:soft-boiled-egg\",\"cooking_method\":\"boiled\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0,\"salty\":1.5,\"bitter\":0,\"umami\":3.5},\"aromatics\":[{\"family\":\"dairy\",\"group\":\"custard\",\"item\":\"egg-yolk\",\"intensity\":4}],\"texture\":[\"jammy\",\"soft\",\"creamy\"]}},{\"role\":\"aromatic_garnish\",\"ingredient_ref\":\"ogs:core:ingredient:nori\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":0,\"sour\":0,\"salty\":2,\"bitter\":1,\"umami\":4},\"aromatics\":[{\"family\":\"marine\",\"group\":\"seaweed\",\"item\":\"nori\",\"intensity\":5.5}],\"texture\":[\"papery\",\"softened\"]}},{\"role\":\"aromatic_garnish\",\"ingredient_ref\":\"ogs:core:ingredient:scallion\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":0.5,\"sour\":0,\"salty\":0,\"bitter\":0.5,\"umami\":0.5},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"scallion\",\"qualifier\":\"fresh\",\"intensity\":4.5}],\"texture\":[\"crisp\",\"fresh\"]}}],\"overall_sensory_profile\":{\"basic_tastes\":{\"sweet\":1.5,\"sour\":0.5,\"salty\":6,\"bitter\":0.5,\"umami\":8.5},\"structural\":{\"body\":8.5,\"acidity_perception\":1,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"savory\",\"group\":\"meat\",\"item\":\"pork\",\"qualifier\":\"cooked\",\"intensity\":8.5},{\"family\":\"marine\",\"group\":\"seaweed\",\"item\":\"nori\",\"intensity\":4.5},{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"scallion\",\"qualifier\":\"fresh\",\"intensity\":4},{\"family\":\"dairy\",\"group\":\"custard\",\"item\":\"egg-yolk\",\"intensity\":3.5},{\"family\":\"confection\",\"group\":\"grain\",\"item\":\"wheat\",\"intensity\":2.5}],\"texture\":[\"brothy\",\"silky\",\"rich\",\"springy\",\"chewy\"],\"finish_length\":7.5,\"complexity\":7}}",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "'boiled' is not one of ['raw', 'seared', 'grilled', 'roasted', 'braised', 'poached', 'steamed', 'fried', 'confit', 'smoked', 'fermented', 'cured', 'pickled', 'dehydrated', 'sous_vide', 'blanched', 'sauteed', 'baked', 'charred', 'emulsified']"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:tonkotsu-ramen",
                "expected": "ogs:core:dish:tonkotsu-ramen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "course": {
                "predicted": "main",
                "expected": "main",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3077,
            "output_tokens": 1010,
            "latency_ms": 13407.0,
            "usd": 0.013946,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\n  \"id\": \"ogs:core:dish:tonkotsu-ramen\",\n  \"ogs_version\": \"0.1.0\",\n  \"primary_language\": \"en\",\n  \"name\": \"tonkotsu ramen\",\n  \"course\": \"main\",\n  \"service_temperature\": \"hot\",\n  \"components\": [\n    {\n      \"role\": \"sauce\",\n      \"ingredient_ref\": \"ogs:core:ingredient:pork-bone-broth\",\n      \"cooking_method\": \"braised\",\n      \"proportion\": \"dominant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 6.5,\n          \"umami\": 8.5,\n          \"sweet\": 1.5,\n          \"sour\": 0.5,\n          \"bitter\": 0.5\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"meat\",\n            \"item\": \"pork stock\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 9\n          },\n          {\n            \"family\": \"dairy\",\n            \"group\": \"fat\",\n            \"item\": \"creamy fat\",\n            \"intensity\": 7.5\n          }\n        ],\n        \"texture\": [\n          \"rich\",\n          \"silky\",\n          \"coating\"\n        ]\n      },\n      \"notes\": \"Opaque emulsified pork-bone broth\"\n    },\n    {\n      \"role\": \"starch\",\n      \"ingredient_ref\": \"ogs:core:ingredient:wheat-noodles\",\n      \"cooking_method\": \"boiled\",\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"salty\": 0.5,\n          \"sour\": 0,\n          \"bitter\": 0,\n          \"umami\": 0.5\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"confection\",\n            \"group\": \"grain\",\n            \"item\": \"wheat\",\n            \"intensity\": 2\n          }\n        ],\n        \"texture\": [\n          \"springy\",\n          \"chewy\",\n          \"slippery\"\n        ]\n      }\n    },\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:chashu-pork\",\n      \"cooking_method\": \"braised\",\n      \"maillard_index\": 1,\n      \"proportion\": \"significant\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"salty\": 5,\n          \"sweet\": 2,\n          \"sour\": 0,\n          \"bitter\": 0.5,\n          \"umami\": 7\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"savory\",\n            \"group\": \"meat\",\n            \"item\": \"pork\",\n            \"qualifier\": \"cooked\",\n            \"intensity\": 7.5\n          }\n        ],\n        \"texture\": [\n          \"tender\",\n          \"fatty\"\n        ]\n      }\n    },\n    {\n      \"role\": \"main_protein\",\n      \"ingredient_ref\": \"ogs:core:ingredient:soft-boiled-egg\",\n      \"cooking_method\": \"poached\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 1,\n          \"sour\": 0,\n          \"salty\": 1,\n          \"bitter\": 0,\n          \"umami\": 3.5\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"dairy\",\n            \"group\": \"egg\",\n            \"item\": \"egg yolk\",\n            \"intensity\": 4\n          }\n        ],\n        \"texture\": [\n          \"jammy\",\n          \"soft\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:nori\",\n      \"cooking_method\": \"dried\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 0,\n          \"sour\": 0,\n          \"salty\": 2,\n          \"bitter\": 1,\n          \"umami\": 4\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"marine\",\n            \"group\": \"seaweed\",\n            \"item\": \"nori\",\n            \"qualifier\": \"dried\",\n            \"intensity\": 5\n          }\n        ],\n        \"texture\": [\n          \"papery\",\n          \"softening\"\n        ]\n      }\n    },\n    {\n      \"role\": \"aromatic_garnish\",\n      \"ingredient_ref\": \"ogs:core:ingredient:scallion\",\n      \"cooking_method\": \"raw\",\n      \"proportion\": \"accent\",\n      \"sensory_contribution\": {\n        \"basic_tastes\": {\n          \"sweet\": 0.5,\n          \"sour\": 0,\n          \"salty\": 0,\n          \"bitter\": 0.5,\n          \"umami\": 0.5\n        },\n        \"aromatics\": [\n          {\n            \"family\": \"vegetal\",\n            \"group\": \"allium\",\n            \"item\": \"scallion\",\n            \"qualifier\": \"fresh\",\n            \"intensity\": 4.5\n          }\n        ],\n        \"texture\": [\n          \"crisp\",\n          \"fresh\"\n        ]\n      }\n    }\n  ],\n  \"overall_sensory_profile\": {\n    \"basic_tastes\": {\n      \"sweet\": 1.5,\n      \"sour\": 0.5,\n      \"salty\": 6.5,\n      \"bitter\": 0.8,\n      \"umami\": 8.5\n    },\n    \"structural\": {\n      \"body\": 8.5,\n      \"acidity_perception\": 0.5,\n      \"tannin\": 0,\n      \"astringency\": 0.5,\n      \"alcohol_heat\": 0,\n      \"effervescence\": 0\n    },\n    \"aromatics\": [\n      {\n        \"family\": \"savory\",\n        \"group\": \"meat\",\n        \"item\": \"pork broth\",\n        \"qualifier\": \"cooked\",\n        \"intensity\": 9\n      },\n      {\n        \"family\": \"dairy\",\n        \"group\": \"fat\",\n        \"item\": \"creamy richness\",\n        \"intensity\": 7.5\n      },\n      {\n        \"family\": \"marine\",\n        \"group\": \"seaweed\",\n        \"item\": \"nori\",\n        \"qualifier\": \"dried\",\n        \"intensity\": 4.5\n      },\n      {\n        \"family\": \"vegetal\",\n        \"group\": \"allium\",\n        \"item\": \"scallion\",\n        \"qualifier\": \"fresh\",\n        \"intensity\": 4\n      }\n    ],\n    \"texture\": [\n      \"rich\",\n      \"silky\",\n      \"springy\",\n      \"chewy\",\n      \"coating\"\n    ],\n    \"finish_length\": 8,\n    \"complexity\": 7.5\n  },\n  \"description\": \"Japanese noodle soup featuring rich emulsified pork-bone broth, wheat noodles, braised chashu, soft egg, nori, and scallion.\"\n}",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "'dried' is not one of ['raw', 'seared', 'grilled', 'roasted', 'braised', 'poached', 'steamed', 'fried', 'confit', 'smoked', 'fermented', 'cured', 'pickled', 'dehydrated', 'sous_vide', 'blanched', 'sauteed', 'baked', 'charred', 'emulsified']"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:tonkotsu-ramen",
                "expected": "ogs:core:dish:tonkotsu-ramen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "course": {
                "predicted": "main",
                "expected": "main",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3077,
            "output_tokens": 1624,
            "latency_ms": 18252.6,
            "usd": 0.020086,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:dish:tonkotsu-ramen\",\"ogs_version\":\"0.1.0\",\"name\":\"tonkotsu ramen\",\"cuisine\":\"Japanese\",\"course\":\"main\",\"service_temperature\":\"hot\",\"components\":[{\"role\":\"sauce\",\"ingredient_ref\":\"ogs:core:ingredient:tonkotsu-broth\",\"cooking_method\":\"braised\",\"proportion\":\"dominant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0.5,\"salty\":5.5,\"bitter\":0.5,\"umami\":8.5},\"aromatics\":[{\"family\":\"savory\",\"group\":\"meat\",\"item\":\"pork\",\"qualifier\":\"cooked\",\"intensity\":8.5},{\"family\":\"dairy\",\"group\":\"fat\",\"item\":\"creamy\",\"intensity\":6.5}],\"texture\":[\"rich\",\"silky\"]},\"notes\":\"Pork-bone broth base.\"},{\"role\":\"starch\",\"ingredient_ref\":\"ogs:core:ingredient:wheat-noodles\",\"cooking_method\":\"boiled\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0,\"salty\":0.5,\"bitter\":0,\"umami\":0.5},\"aromatics\":[{\"family\":\"confection\",\"group\":\"grain\",\"item\":\"wheat\",\"intensity\":2}],\"texture\":[\"springy\",\"chewy\"]},\"notes\":\"Alkaline wheat ramen noodles.\"},{\"role\":\"main_protein\",\"ingredient_ref\":\"ogs:core:ingredient:chashu-pork\",\"cooking_method\":\"braised\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":2,\"sour\":0.5,\"salty\":4,\"bitter\":0,\"umami\":7},\"aromatics\":[{\"family\":\"savory\",\"group\":\"meat\",\"item\":\"pork\",\"qualifier\":\"cooked\",\"intensity\":7},{\"family\":\"spice\",\"group\":\"allium\",\"item\":\"garlic\",\"intensity\":2}],\"texture\":[\"tender\",\"succulent\"]},\"notes\":\"Braised pork belly slices.\"},{\"role\":\"main_protein\",\"ingredient_ref\":\"ogs:core:ingredient:soft-boiled-egg\",\"cooking_method\":\"boiled\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0,\"salty\":1.5,\"bitter\":0,\"umami\":3.5},\"aromatics\":[{\"family\":\"dairy\",\"group\":\"custard\",\"item\":\"egg-yolk\",\"intensity\":3}],\"texture\":[\"jammy\",\"creamy\",\"tender\"]},\"notes\":\"Soft egg, often marinated.\"},{\"role\":\"aromatic_garnish\",\"ingredient_ref\":\"ogs:core:ingredient:nori\",\"cooking_method\":\"dried\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":0,\"sour\":0,\"salty\":2,\"bitter\":1,\"umami\":4},\"aromatics\":[{\"family\":\"marine\",\"group\":\"seaweed\",\"item\":\"nori\",\"qualifier\":\"dried\",\"intensity\":5}],\"texture\":[\"papery\",\"softened\"]}},{\"role\":\"aromatic_garnish\",\"ingredient_ref\":\"ogs:core:ingredient:scallion\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":0.5,\"sour\":0,\"salty\":0,\"bitter\":0.5,\"umami\":0.5},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"scallion\",\"qualifier\":\"fresh\",\"intensity\":4},{\"family\":\"herbal\",\"group\":\"green-herb\",\"item\":\"green-onion-top\",\"qualifier\":\"fresh\",\"intensity\":2.5}],\"texture\":[\"crisp\",\"fresh\"]}}],\"overall_sensory_profile\":{\"basic_tastes\":{\"sweet\":1.5,\"sour\":1,\"salty\":6,\"bitter\":0.5,\"umami\":9},\"structural\":{\"body\":9,\"acidity_perception\":1,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"savory\",\"group\":\"meat\",\"item\":\"pork\",\"qualifier\":\"cooked\",\"intensity\":8.5},{\"family\":\"dairy\",\"group\":\"fat\",\"item\":\"creamy\",\"intensity\":6.5},{\"family\":\"marine\",\"group\":\"seaweed\",\"item\":\"nori\",\"qualifier\":\"dried\",\"intensity\":4.5},{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"scallion\",\"qualifier\":\"fresh\",\"intensity\":4},{\"family\":\"confection\",\"group\":\"grain\",\"item\":\"wheat\",\"intensity\":2.5}],\"texture\":[\"rich\",\"silky\",\"springy\",\"chewy\",\"tender\"],\"finish_length\":8,\"complexity\":7.5},\"notes\":\"Classic tonkotsu ramen with pork-bone broth, wheat noodles, chashu, soft egg, nori, and scallion.\"}",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "schema_valid": false,
            "schema_errors": [
              "'dried' is not one of ['raw', 'seared', 'grilled', 'roasted', 'braised', 'poached', 'steamed', 'fried', 'confit', 'smoked', 'fermented', 'cured', 'pickled', 'dehydrated', 'sous_vide', 'blanched', 'sauteed', 'baked', 'charred', 'emulsified']"
            ],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:tonkotsu-ramen",
                "expected": "ogs:core:dish:tonkotsu-ramen",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "course": {
                "predicted": "main",
                "expected": "main",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3077,
            "output_tokens": 1088,
            "latency_ms": 12581.0,
            "usd": 0.014726,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3077.0,
        "output_tokens_mean": 1240.667,
        "latency_ms_mean": 14746.867,
        "usd_sum": 0.048758
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-017",
      "category": "structured_output",
      "capability": "schema_valid_dish",
      "difficulty": "medium",
      "cuisine_region": "mexican",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.dish",
        "vocab.component_roles",
        "vocab.cooking_methods",
        "vocab.courses",
        "vocab.service_temperatures",
        "vocab.aromatic_families",
        "scales.basic_tastes",
        "scales.structural",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "534d8590759d6dc3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS dish schema (`schema/dish.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/dish.schema.json\",\n  \"title\": \"OGS Dish\",\n  \"description\": \"A composed dish with components, cooking methods, and overall sensory profile.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"components\",\n    \"overall_sensory_profile\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"cuisine\": {\n      \"type\": \"string\",\n      \"description\": \"Free-text cuisine name (v0.1) OR a cuisine entity reference matching ogs:<ns>:cuisine:<id> (v0.2+).\"\n    },\n    \"course\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"amuse\",\n        \"appetizer\",\n        \"soup\",\n        \"salad\",\n        \"fish\",\n        \"main\",\n        \"cheese\",\n        \"dessert\",\n        \"petit_four\"\n      ]\n    },\n    \"service_temperature\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"frozen\",\n        \"cold\",\n        \"cool\",\n        \"room\",\n        \"warm\",\n        \"hot\",\n        \"very_hot\"\n      ]\n    },\n    \"components\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/component\"\n      }\n    },\n    \"overall_sensory_profile\": {\n      \"$ref\": \"#/$defs/sensory_profile\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"component\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"role\",\n        \"ingredient_ref\"\n      ],\n      \"properties\": {\n        \"role\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"main_protein\",\n            \"main_vegetable\",\n            \"starch\",\n            \"sauce\",\n            \"fat_component\",\n            \"acid_component\",\n            \"aromatic_garnish\",\n            \"seasoning\",\n            \"textural_element\"\n          ]\n        },\n        \"ingredient_ref\": {\n          \"type\": \"string\",\n          \"pattern\": \"^ogs:[a-z0-9-]+:ingredient:[a-z0-9-]+$\"\n        },\n        \"cooking_method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"raw\",\n            \"seared\",\n            \"grilled\",\n            \"roasted\",\n            \"braised\",\n            \"poached\",\n            \"steamed\",\n            \"fried\",\n            \"confit\",\n            \"smoked\",\n            \"fermented\",\n            \"cured\",\n            \"pickled\",\n            \"dehydrated\",\n            \"sous_vide\",\n            \"blanched\",\n            \"sauteed\",\n            \"baked\",\n            \"charred\",\n            \"emulsified\"\n          ]\n        },\n        \"maillard_index\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"reduction_level\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 5\n        },\n        \"proportion\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"dominant\",\n            \"significant\",\n            \"accent\",\n            \"trace\"\n          ]\n        },\n        \"sensory_contribution\": {\n          \"type\": \"object\",\n          \"properties\": {\n            \"basic_tastes\": {\n              \"$ref\": \"#/$defs/basic_tastes\"\n            },\n            \"aromatics\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"$ref\": \"#/$defs/aromatic_entry\"\n              }\n            },\n            \"texture\": {\n              \"type\": \"array\",\n              \"items\": {\n                \"type\": \"string\"\n              }\n            }\n          },\n          \"additionalProperties\": true\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"sensory_profile\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"basic_tastes\": {\n          \"$ref\": \"#/$defs/basic_tastes\"\n        },\n        \"structural\": {\n          \"$ref\": \"#/$defs/structural\"\n        },\n        \"aromatics\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"$ref\": \"#/$defs/aromatic_entry\"\n          }\n        },\n        \"texture\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"finish_length\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        },\n        \"complexity\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 10\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"basic_tastes\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"sweet\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"sour\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"salty\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"bitter\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"umami\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"structural\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"body\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"acidity_perception\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"tannin\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"astringency\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"alcohol_heat\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        },\n        \"effervescence\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"intensity_value\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 10\n    },\n    \"aromatic_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"family\",\n        \"group\",\n        \"item\",\n        \"intensity\"\n      ],\n      \"properties\": {\n        \"family\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fruit\",\n            \"floral\",\n            \"herbal\",\n            \"spice\",\n            \"earth\",\n            \"wood\",\n            \"dairy\",\n            \"savory\",\n            \"confection\",\n            \"vegetal\",\n            \"marine\"\n          ]\n        },\n        \"group\": {\n          \"type\": \"string\"\n        },\n        \"item\": {\n          \"type\": \"string\"\n        },\n        \"qualifier\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"fresh\",\n            \"dried\",\n            \"cooked\",\n            \"candied\",\n            \"fermented\",\n            \"zest\",\n            \"overripe\"\n          ]\n        },\n        \"intensity\": {\n          \"$ref\": \"#/$defs/intensity_value\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n### Dish courses\n\namuse, appetizer, soup, salad, fish, main, cheese, dessert, petit_four\n\n### Service temperatures\n\nfrozen, cold, cool, room, warm, hot, very_hot\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS dish JSON for pico de gallo (tomato, onion, cilantro, jalapeño, lime). JSON only.\n\nINPUT:\n{\n  \"dish_name\": \"pico de gallo\",\n  \"namespace\": \"core\",\n  \"id_slug\": \"pico-de-gallo\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/dish.schema.json",
        "required_fields": {
          "id": "ogs:core:dish:pico-de-gallo",
          "ogs_version": "0.1.0"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:dish:pico-de-gallo\",\"ogs_version\":\"0.1.0\",\"name\":\"pico de gallo\",\"course\":\"appetizer\",\"service_temperature\":\"cool\",\"components\":[{\"role\":\"main_vegetable\",\"ingredient_ref\":\"ogs:core:ingredient:tomato\",\"cooking_method\":\"raw\",\"proportion\":\"dominant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":2.5,\"sour\":2,\"salty\":0,\"bitter\":0.5,\"umami\":2},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"nightshade\",\"item\":\"tomato\",\"qualifier\":\"fresh\",\"intensity\":6}],\"texture\":[\"juicy\",\"soft\",\"seeded\"]}},{\"role\":\"main_vegetable\",\"ingredient_ref\":\"ogs:core:ingredient:onion\",\"cooking_method\":\"raw\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0,\"salty\":0,\"bitter\":0.5,\"umami\":0.5},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"onion\",\"qualifier\":\"fresh\",\"intensity\":6}],\"texture\":[\"crisp\",\"watery\",\"pungent\"]}},{\"role\":\"aromatic_garnish\",\"ingredient_ref\":\"ogs:core:ingredient:cilantro\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":0,\"sour\":0,\"salty\":0,\"bitter\":0.5,\"umami\":0},\"aromatics\":[{\"family\":\"herbal\",\"group\":\"leafy-herb\",\"item\":\"cilantro\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"leafy\",\"fresh\"]}},{\"role\":\"seasoning\",\"ingredient_ref\":\"ogs:core:ingredient:jalapeno\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":0,\"sour\":0,\"salty\":0,\"bitter\":0.5,\"umami\":0},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"chile\",\"item\":\"jalapeno\",\"qualifier\":\"fresh\",\"intensity\":5}],\"texture\":[\"crisp\",\"juicy\",\"peppery\"]}},{\"role\":\"acid_component\",\"ingredient_ref\":\"ogs:core:ingredient:lime\",\"cooking_method\":\"raw\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":0.5,\"sour\":8,\"salty\":0,\"bitter\":0.5,\"umami\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime\",\"qualifier\":\"zest\",\"intensity\":6},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"juicy\"]}}],\"overall_sensory_profile\":{\"basic_tastes\":{\"sweet\":2,\"sour\":5.5,\"salty\":0.5,\"bitter\":0.5,\"umami\":1.5},\"structural\":{\"body\":3,\"acidity_perception\":6.5,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"nightshade\",\"item\":\"tomato\",\"qualifier\":\"fresh\",\"intensity\":6.5},{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"onion\",\"qualifier\":\"fresh\",\"intensity\":6},{\"family\":\"herbal\",\"group\":\"leafy-herb\",\"item\":\"cilantro\",\"qualifier\":\"fresh\",\"intensity\":7},{\"family\":\"vegetal\",\"group\":\"chile\",\"item\":\"jalapeno\",\"qualifier\":\"fresh\",\"intensity\":5.5},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"juicy\",\"chunky\",\"crisp\",\"fresh\"],\"finish_length\":4.5,\"complexity\":4.5},\"description\":\"Fresh chopped salsa of tomato, onion, cilantro, jalapeño, and lime.\"}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:pico-de-gallo",
                "expected": "ogs:core:dish:pico-de-gallo",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3062,
            "output_tokens": 884,
            "latency_ms": 10544.0,
            "usd": 0.012668,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"id\":\"ogs:core:dish:pico-de-gallo\",\"ogs_version\":\"0.1.0\",\"name\":\"pico de gallo\",\"components\":[{\"role\":\"main_vegetable\",\"ingredient_ref\":\"ogs:core:ingredient:tomato\",\"cooking_method\":\"raw\",\"proportion\":\"dominant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":2.5,\"sour\":2,\"umami\":2},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"tomato\",\"item\":\"tomato\",\"qualifier\":\"fresh\",\"intensity\":6},{\"family\":\"vegetal\",\"group\":\"green\",\"item\":\"fresh tomato leaf\",\"qualifier\":\"fresh\",\"intensity\":2}],\"texture\":[\"juicy\",\"soft\",\"seeded\"]}},{\"role\":\"main_vegetable\",\"ingredient_ref\":\"ogs:core:ingredient:onion\",\"cooking_method\":\"raw\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0.5,\"bitter\":0.5},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"onion\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"crisp\",\"watery\",\"pungent\"]}},{\"role\":\"aromatic_garnish\",\"ingredient_ref\":\"ogs:core:ingredient:cilantro\",\"cooking_method\":\"raw\",\"proportion\":\"significant\",\"sensory_contribution\":{\"aromatics\":[{\"family\":\"herbal\",\"group\":\"leafy herb\",\"item\":\"cilantro\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"leafy\",\"fresh\"]}},{\"role\":\"seasoning\",\"ingredient_ref\":\"ogs:core:ingredient:jalapeno\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"bitter\":0.5},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"chile pepper\",\"item\":\"jalapeno\",\"qualifier\":\"fresh\",\"intensity\":6}],\"texture\":[\"crisp\",\"juicy\",\"piquant\"]}},{\"role\":\"acid_component\",\"ingredient_ref\":\"ogs:core:ingredient:lime\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sour\":8,\"bitter\":1},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime\",\"qualifier\":\"zest\",\"intensity\":6},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime juice\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"juicy\"]}}],\"overall_sensory_profile\":{\"basic_tastes\":{\"sweet\":2,\"sour\":5.5,\"salty\":0,\"bitter\":1,\"umami\":2},\"structural\":{\"body\":3,\"acidity_perception\":6,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"tomato\",\"item\":\"tomato\",\"qualifier\":\"fresh\",\"intensity\":6.5},{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"onion\",\"qualifier\":\"fresh\",\"intensity\":6.5},{\"family\":\"herbal\",\"group\":\"leafy herb\",\"item\":\"cilantro\",\"qualifier\":\"fresh\",\"intensity\":7},{\"family\":\"vegetal\",\"group\":\"chile pepper\",\"item\":\"jalapeno\",\"qualifier\":\"fresh\",\"intensity\":5.5},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime\",\"qualifier\":\"fresh\",\"intensity\":6.5}],\"texture\":[\"chunky\",\"juicy\",\"crisp\",\"fresh\"],\"finish_length\":4.5,\"complexity\":4}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:pico-de-gallo",
                "expected": "ogs:core:dish:pico-de-gallo",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3062,
            "output_tokens": 781,
            "latency_ms": 9404.5,
            "usd": 0.011638,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:dish:pico-de-gallo\",\"ogs_version\":\"0.1.0\",\"name\":\"pico de gallo\",\"course\":\"appetizer\",\"service_temperature\":\"cold\",\"components\":[{\"role\":\"main_vegetable\",\"ingredient_ref\":\"ogs:core:ingredient:tomato\",\"cooking_method\":\"raw\",\"proportion\":\"dominant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":2.5,\"sour\":2,\"umami\":2},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"nightshade\",\"item\":\"tomato\",\"qualifier\":\"fresh\",\"intensity\":6}],\"texture\":[\"juicy\",\"soft\"]}},{\"role\":\"main_vegetable\",\"ingredient_ref\":\"ogs:core:ingredient:onion\",\"cooking_method\":\"raw\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"sweet\":1,\"sour\":0.5,\"bitter\":0.5},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"onion\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"crisp\",\"pungent\"]}},{\"role\":\"aromatic_garnish\",\"ingredient_ref\":\"ogs:core:ingredient:cilantro\",\"cooking_method\":\"raw\",\"proportion\":\"significant\",\"sensory_contribution\":{\"basic_tastes\":{\"bitter\":0.5},\"aromatics\":[{\"family\":\"herbal\",\"group\":\"leafy-herb\",\"item\":\"cilantro\",\"qualifier\":\"fresh\",\"intensity\":7}],\"texture\":[\"leafy\",\"fresh\"]}},{\"role\":\"seasoning\",\"ingredient_ref\":\"ogs:core:ingredient:jalapeno\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"bitter\":0.5},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"chile\",\"item\":\"jalapeno\",\"qualifier\":\"fresh\",\"intensity\":6}],\"texture\":[\"crisp\"]}},{\"role\":\"acid_component\",\"ingredient_ref\":\"ogs:core:ingredient:lime\",\"cooking_method\":\"raw\",\"proportion\":\"accent\",\"sensory_contribution\":{\"basic_tastes\":{\"sour\":8,\"bitter\":0.5},\"aromatics\":[{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime\",\"qualifier\":\"zest\",\"intensity\":5}],\"texture\":[\"juicy\"]}}],\"overall_sensory_profile\":{\"basic_tastes\":{\"sweet\":2,\"sour\":5.5,\"salty\":0,\"bitter\":1,\"umami\":2},\"structural\":{\"body\":3,\"acidity_perception\":6,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"vegetal\",\"group\":\"nightshade\",\"item\":\"tomato\",\"qualifier\":\"fresh\",\"intensity\":6},{\"family\":\"vegetal\",\"group\":\"allium\",\"item\":\"onion\",\"qualifier\":\"fresh\",\"intensity\":6.5},{\"family\":\"herbal\",\"group\":\"leafy-herb\",\"item\":\"cilantro\",\"qualifier\":\"fresh\",\"intensity\":6.5},{\"family\":\"vegetal\",\"group\":\"chile\",\"item\":\"jalapeno\",\"qualifier\":\"fresh\",\"intensity\":5.5},{\"family\":\"fruit\",\"group\":\"citrus\",\"item\":\"lime\",\"qualifier\":\"fresh\",\"intensity\":5.5}],\"texture\":[\"juicy\",\"chunky\",\"crisp\",\"fresh\"],\"finish_length\":4.5,\"complexity\":4}}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:dish:pico-de-gallo",
                "expected": "ogs:core:dish:pico-de-gallo",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3062,
            "output_tokens": 745,
            "latency_ms": 13215.4,
            "usd": 0.011277,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3062.0,
        "output_tokens_mean": 803.333,
        "latency_ms_mean": 11054.633,
        "usd_sum": 0.035583
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-018",
      "category": "structured_output",
      "capability": "sensory_profile_json",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "json_match",
      "context_needs": [
        "scales.basic_tastes",
        "scales.structural",
        "vocab.aromatic_families",
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "1470a28b8e9ef028",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nProduce a JSON sensory profile for 24-month aged Comté cheese. Keys: basic_tastes, structural, aromatics. All intensities on 0-10. JSON only.\n\nINPUT:\n{\n  \"item\": \"24-month aged Comté\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "value": {
          "basic_tastes": {
            "sweet": 2.0,
            "sour": 2.0,
            "salty": 5.0,
            "bitter": 1.5,
            "umami": 8.0
          },
          "structural": {
            "body": 7.0,
            "acidity_perception": 3.0,
            "tannin": 0,
            "astringency": 0,
            "alcohol_heat": 0,
            "effervescence": 0
          }
        },
        "required_keys": [
          "basic_tastes.umami",
          "basic_tastes.salty",
          "structural.body"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"basic_tastes\":{\"sweet\":1.5,\"sour\":1,\"salty\":5.5,\"bitter\":1,\"umami\":7.5},\"structural\":{\"body\":8,\"acidity_perception\":1.5,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"dairy\",\"qualifier\":\"fermented\",\"intensity\":8.5},{\"family\":\"savory\",\"qualifier\":\"cooked\",\"intensity\":7.5},{\"family\":\"wood\",\"qualifier\":\"dried\",\"intensity\":3.5},{\"family\":\"earth\",\"qualifier\":\"dried\",\"intensity\":3},{\"family\":\"confection\",\"qualifier\":\"candied\",\"intensity\":2.5},{\"family\":\"fruit\",\"qualifier\":\"dried\",\"intensity\":2}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.umami": {
                "predicted": 7.5,
                "expected": 8.0,
                "match": true
              },
              "basic_tastes.salty": {
                "predicted": 5.5,
                "expected": 5.0,
                "match": true
              },
              "structural.body": {
                "predicted": 8,
                "expected": 7.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 453,
            "output_tokens": 186,
            "latency_ms": 2892.8,
            "usd": 0.002426,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"basic_tastes\":{\"sweet\":1.5,\"sour\":1,\"salty\":5.5,\"bitter\":1,\"umami\":7.5},\"structural\":{\"body\":8,\"acidity_perception\":1.5,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"dairy\",\"qualifier\":\"fermented\",\"intensity\":8.5},{\"family\":\"dairy\",\"qualifier\":\"cooked\",\"intensity\":6.5},{\"family\":\"confection\",\"qualifier\":\"candied\",\"intensity\":3},{\"family\":\"savory\",\"qualifier\":\"fresh\",\"intensity\":5.5},{\"family\":\"wood\",\"qualifier\":\"dried\",\"intensity\":2.5},{\"family\":\"earth\",\"qualifier\":\"dried\",\"intensity\":2},{\"family\":\"fruit\",\"qualifier\":\"dried\",\"intensity\":2}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.umami": {
                "predicted": 7.5,
                "expected": 8.0,
                "match": true
              },
              "basic_tastes.salty": {
                "predicted": 5.5,
                "expected": 5.0,
                "match": true
              },
              "structural.body": {
                "predicted": 8,
                "expected": 7.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 453,
            "output_tokens": 201,
            "latency_ms": 3532.0,
            "usd": 0.002576,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"basic_tastes\":{\"sweet\":1.5,\"sour\":1.5,\"salty\":5.5,\"bitter\":1,\"umami\":7.5},\"structural\":{\"body\":7.5,\"acidity_perception\":2,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"dairy\",\"qualifier\":\"fermented\",\"intensity\":8},{\"family\":\"savory\",\"qualifier\":\"cooked\",\"intensity\":7.5},{\"family\":\"wood\",\"qualifier\":\"fresh\",\"intensity\":2},{\"family\":\"earth\",\"qualifier\":\"dried\",\"intensity\":3},{\"family\":\"confection\",\"qualifier\":\"candied\",\"intensity\":2},{\"family\":\"fruit\",\"qualifier\":\"dried\",\"intensity\":2},{\"family\":\"spice\",\"qualifier\":\"fresh\",\"intensity\":2}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.umami": {
                "predicted": 7.5,
                "expected": 8.0,
                "match": true
              },
              "basic_tastes.salty": {
                "predicted": 5.5,
                "expected": 5.0,
                "match": true
              },
              "structural.body": {
                "predicted": 7.5,
                "expected": 7.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 453,
            "output_tokens": 196,
            "latency_ms": 2957.0,
            "usd": 0.002526,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 453.0,
        "output_tokens_mean": 194.333,
        "latency_ms_mean": 3127.267,
        "usd_sum": 0.007528
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-019",
      "category": "structured_output",
      "capability": "sensory_profile_json",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "json_match",
      "context_needs": [
        "scales.basic_tastes",
        "scales.structural",
        "vocab.aromatic_families",
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "aa2726470aa7cfb8",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Basic tastes scale (0-10)\n\nEach of `sweet`, `sour`, `salty`, `bitter`, `umami` is rated on a\n0-10 intensity scale where 0 = not detectable, 5 = clearly present,\n10 = overwhelming. Use decimals when precision helps (e.g. 8.5).\n\n### Structural attributes scale (0-10)\n\n`body`, `acidity_perception`, `tannin`, `astringency`, `alcohol_heat`,\n`effervescence` are each rated on the same 0-10 scale as basic tastes.\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nProduce a JSON sensory profile for blistered shishito peppers lightly salted. Keys: basic_tastes, structural, aromatics.\n\nINPUT:\n{\n  \"item\": \"blistered shishito peppers with flaky salt\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "value": {
          "basic_tastes": {
            "sweet": 2.0,
            "sour": 1.0,
            "salty": 5.0,
            "bitter": 3.0,
            "umami": 2.0
          },
          "structural": {
            "body": 3.0,
            "acidity_perception": 2.0,
            "tannin": 0,
            "astringency": 1.0,
            "alcohol_heat": 0,
            "effervescence": 0
          }
        },
        "required_keys": [
          "basic_tastes.salty",
          "basic_tastes.bitter",
          "structural.body"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"basic_tastes\":{\"sweet\":1,\"sour\":1,\"salty\":4.5,\"bitter\":2.5,\"umami\":2},\"structural\":{\"body\":2.5,\"acidity_perception\":1.5,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"vegetal\",\"qualifier\":\"fresh\",\"intensity\":6},{\"family\":\"vegetal\",\"qualifier\":\"cooked\",\"intensity\":5.5},{\"family\":\"earth\",\"qualifier\":\"fresh\",\"intensity\":2},{\"family\":\"savory\",\"qualifier\":\"cooked\",\"intensity\":2.5}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.salty": {
                "predicted": 4.5,
                "expected": 5.0,
                "match": true
              },
              "basic_tastes.bitter": {
                "predicted": 2.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 2.5,
                "expected": 3.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 447,
            "output_tokens": 148,
            "latency_ms": 2463.3,
            "usd": 0.002039,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"basic_tastes\":{\"sweet\":1,\"sour\":0.5,\"salty\":4.5,\"bitter\":1.5,\"umami\":1},\"structural\":{\"body\":2,\"acidity_perception\":1,\"tannin\":0,\"astringency\":0.5,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"vegetal\",\"qualifier\":\"fresh\",\"intensity\":5.5},{\"family\":\"vegetal\",\"qualifier\":\"cooked\",\"intensity\":4.5},{\"family\":\"earth\",\"qualifier\":\"fresh\",\"intensity\":1.5},{\"family\":\"savory\",\"qualifier\":\"cooked\",\"intensity\":2},{\"family\":\"spice\",\"qualifier\":\"fresh\",\"intensity\":1}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.salty": {
                "predicted": 4.5,
                "expected": 5.0,
                "match": true
              },
              "basic_tastes.bitter": {
                "predicted": 1.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 2,
                "expected": 3.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 447,
            "output_tokens": 165,
            "latency_ms": 2436.2,
            "usd": 0.002209,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"basic_tastes\":{\"sweet\":1.5,\"sour\":0.5,\"salty\":4.5,\"bitter\":1.5,\"umami\":2},\"structural\":{\"body\":2.5,\"acidity_perception\":1,\"tannin\":0,\"astringency\":1,\"alcohol_heat\":0,\"effervescence\":0},\"aromatics\":[{\"family\":\"vegetal\",\"qualifier\":\"fresh\"},{\"family\":\"vegetal\",\"qualifier\":\"cooked\"},{\"family\":\"earth\",\"qualifier\":\"fresh\"},{\"family\":\"savory\",\"qualifier\":\"cooked\"}]}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "per_key": {
              "basic_tastes.salty": {
                "predicted": 4.5,
                "expected": 5.0,
                "match": true
              },
              "basic_tastes.bitter": {
                "predicted": 1.5,
                "expected": 3.0,
                "match": true
              },
              "structural.body": {
                "predicted": 2.5,
                "expected": 3.0,
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 447,
            "output_tokens": 126,
            "latency_ms": 2320.7,
            "usd": 0.001819,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 447.0,
        "output_tokens_mean": 146.333,
        "latency_ms_mean": 2406.733,
        "usd_sum": 0.006067
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:struct-020",
      "category": "structured_output",
      "capability": "schema_valid_pairing",
      "difficulty": "expert",
      "cuisine_region": "mexican",
      "scorer": "schema_validation",
      "context_needs": [
        "schema.pairing",
        "vocab.match_types",
        "vocab.explanation_codes",
        "vocab.grade_scale",
        "id.format",
        "ogs_version"
      ],
      "prompt_template_hash": "5b115600e8dad71f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS pairing schema (`schema/pairing.schema.json`)\n\n```json\n{\n  \"$schema\": \"https://json-schema.org/draft/2020-12/schema\",\n  \"$id\": \"https://opengastronomy.org/schema/v0.2/pairing.schema.json\",\n  \"title\": \"OGS Pairing\",\n  \"description\": \"A scored, explained pairing between a dish and a beverage.\",\n  \"type\": \"object\",\n  \"required\": [\n    \"id\",\n    \"ogs_version\",\n    \"name\",\n    \"dish_ref\",\n    \"beverage_ref\",\n    \"match_types\",\n    \"overall_score\",\n    \"grade\",\n    \"explanation_codes\",\n    \"confidence\"\n  ],\n  \"properties\": {\n    \"id\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:pairing:[a-z0-9-]+$\"\n    },\n    \"ogs_version\": {\n      \"type\": \"string\",\n      \"pattern\": \"^0\\\\.(1|2)\\\\.\\\\d+$\"\n    },\n    \"primary_language\": {\n      \"$ref\": \"#/$defs/bcp47_tag\"\n    },\n    \"available_languages\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/bcp47_tag\"\n      }\n    },\n    \"name\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"description\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"dish_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:dish:[a-z0-9-]+$\"\n    },\n    \"beverage_ref\": {\n      \"type\": \"string\",\n      \"pattern\": \"^ogs:[a-z0-9-]+:beverage:[a-z0-9-]+$\"\n    },\n    \"match_types\": {\n      \"type\": \"array\",\n      \"minItems\": 1,\n      \"items\": {\n        \"$ref\": \"#/$defs/match_type_entry\"\n      }\n    },\n    \"overall_score\": {\n      \"type\": \"integer\",\n      \"minimum\": 0,\n      \"maximum\": 100\n    },\n    \"grade\": {\n      \"type\": \"string\",\n      \"enum\": [\n        \"A+\",\n        \"A\",\n        \"B+\",\n        \"B\",\n        \"C+\",\n        \"C\",\n        \"D\",\n        \"F\"\n      ]\n    },\n    \"explanation_codes\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/explanation_code_entry\"\n      }\n    },\n    \"risks\": {\n      \"type\": \"array\",\n      \"items\": {\n        \"$ref\": \"#/$defs/risk_entry\"\n      }\n    },\n    \"confidence\": {\n      \"type\": \"number\",\n      \"minimum\": 0,\n      \"maximum\": 1\n    },\n    \"evidence\": {\n      \"$ref\": \"#/$defs/evidence\"\n    },\n    \"notes\": {\n      \"$ref\": \"#/$defs/lstring\"\n    },\n    \"metadata\": {\n      \"$ref\": \"#/$defs/metadata\"\n    }\n  },\n  \"additionalProperties\": true,\n  \"$defs\": {\n    \"bcp47_tag\": {\n      \"type\": \"string\",\n      \"pattern\": \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\"\n    },\n    \"lstring\": {\n      \"oneOf\": [\n        {\n          \"type\": \"string\",\n          \"minLength\": 1\n        },\n        {\n          \"type\": \"object\",\n          \"minProperties\": 1,\n          \"patternProperties\": {\n            \"^[A-Za-z]{2,3}(-[A-Za-z0-9]{2,8})*$\": {\n              \"type\": \"string\",\n              \"minLength\": 1\n            }\n          },\n          \"additionalProperties\": false\n        },\n        {\n          \"type\": \"object\",\n          \"required\": [\n            \"labels\"\n          ],\n          \"properties\": {\n            \"primary_language\": {\n              \"$ref\": \"#/$defs/bcp47_tag\"\n            },\n            \"labels\": {\n              \"type\": \"array\",\n              \"minItems\": 1,\n              \"items\": {\n                \"type\": \"object\",\n                \"required\": [\n                  \"value\",\n                  \"language\"\n                ],\n                \"properties\": {\n                  \"value\": {\n                    \"type\": \"string\",\n                    \"minLength\": 1\n                  },\n                  \"language\": {\n                    \"$ref\": \"#/$defs/bcp47_tag\"\n                  },\n                  \"script\": {\n                    \"type\": \"string\",\n                    \"pattern\": \"^[A-Z][a-z]{3}$\"\n                  },\n                  \"role\": {\n                    \"type\": \"string\",\n                    \"enum\": [\n                      \"preferred\",\n                      \"alternate\",\n                      \"transliteration\",\n                      \"descriptive\",\n                      \"loan\",\n                      \"historical\",\n                      \"deprecated\",\n                      \"ipa\"\n                    ]\n                  },\n                  \"transliteration_scheme\": {\n                    \"type\": \"string\"\n                  },\n                  \"translation\": {\n                    \"type\": \"object\",\n                    \"additionalProperties\": true\n                  }\n                },\n                \"additionalProperties\": false\n              }\n            }\n          },\n          \"additionalProperties\": false\n        }\n      ]\n    },\n    \"match_type_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"type\",\n        \"score\"\n      ],\n      \"properties\": {\n        \"type\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"harmony\",\n            \"contrast\",\n            \"bridge\",\n            \"intensity_alignment\",\n            \"structural_balance\",\n            \"refresh\"\n          ]\n        },\n        \"score\": {\n          \"type\": \"integer\",\n          \"minimum\": 0,\n          \"maximum\": 100\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"primary\": {\n          \"type\": \"boolean\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"explanation_code_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"weight\": {\n          \"type\": \"number\",\n          \"minimum\": 0,\n          \"maximum\": 1\n        },\n        \"detail\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"risk_entry\": {\n      \"type\": \"object\",\n      \"required\": [\n        \"code\",\n        \"severity\"\n      ],\n      \"properties\": {\n        \"code\": {\n          \"type\": \"string\"\n        },\n        \"severity\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"low\",\n            \"medium\",\n            \"high\"\n          ]\n        },\n        \"description\": {\n          \"$ref\": \"#/$defs/lstring\"\n        },\n        \"mitigation\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": false\n    },\n    \"evidence\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"method\": {\n          \"type\": \"string\",\n          \"enum\": [\n            \"expert_panel\",\n            \"instrumental\",\n            \"computed\",\n            \"estimated\"\n          ]\n        },\n        \"sources\": {\n          \"type\": \"array\",\n          \"items\": {\n            \"type\": \"string\"\n          }\n        },\n        \"notes\": {\n          \"$ref\": \"#/$defs/lstring\"\n        }\n      },\n      \"additionalProperties\": true\n    },\n    \"metadata\": {\n      \"type\": \"object\",\n      \"properties\": {\n        \"created\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"updated\": {\n          \"type\": \"string\",\n          \"format\": \"date-time\"\n        },\n        \"source\": {\n          \"type\": \"string\"\n        },\n        \"license\": {\n          \"type\": \"string\"\n        }\n      },\n      \"additionalProperties\": true\n    }\n  }\n}\n```\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n### OGS Explanation Codes\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**negative**\n- `RISK_TANNIN_FISH` — Tannin-Fish Clash: Tannins interact with fish oils to produce metallic, unpleasant off-flavors.\n- `RISK_ALCOHOL_OVERPOWERS` — Alcohol Overpowers Dish: High alcohol content overwhelms the dish's delicate flavors with heat and bitterness.\n- `RISK_BITTER_AMPLIFY` — Bitterness Amplification: Bitterness in both dish and beverage compounds to an unpleasant level.\n- `RISK_OAK_MASKS` — Oak Masks Dish: Heavy oak influence in the beverage masks or overwhelms the dish's subtle flavors.\n- `RISK_INTENSITY_MISMATCH` — Intensity Mismatch: One element is significantly more intense than the other, causing the weaker to be lost.\n- `RISK_ACID_CLASH` — Acid Clash: Excessive acidity from both dish and beverage creates an unpleasantly sharp experience.\n- `RISK_SWEETNESS_CONFLICT` — Sweetness Conflict: The dish makes the beverage taste less sweet, thin, or overly acidic.\n- `RISK_SPICE_ALCOHOL` — Spice-Alcohol Amplification: High alcohol amplifies the heat of spicy dishes to an uncomfortable level.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n### OGS ID format\n\nAll OGS entity IDs follow the pattern:\n\n    ogs:<namespace>:<entity-type>:<slug>\n\n- `namespace` is lowercase alphanumeric with hyphens, e.g. `core`.\n- `entity-type` is one of `ingredient`, `dish`, `beverage`, `pairing`,\n  `benchmark-task`, etc.\n- `slug` is lowercase alphanumeric with hyphens.\n\nExample: `ogs:core:ingredient:fleur-de-sel`.\n\n### OGS version\n\nAll produced documents must set `ogs_version` to **\"0.1.0\"**.\n\n---\n\nProduce an OGS pairing JSON for mole-negro enchiladas with Aragonese Garnacha. JSON only.\n\nINPUT:\n{\n  \"pairing_slug\": \"mole-negro-garnacha\",\n  \"dish_ref\": \"ogs:core:dish:mole-negro-enchiladas\",\n  \"beverage_ref\": \"ogs:core:beverage:garnacha-aragon\",\n  \"intent\": \"Cross-cultural bridge pairing: ripe fruit and body handle chocolate-chile sauce.\"\n}\n\nANSWER FORMAT: json",
      "expected": {
        "schema_ref": "schema/pairing.schema.json",
        "required_fields": {
          "id": "ogs:core:pairing:mole-negro-garnacha",
          "ogs_version": "0.1.0",
          "dish_ref": "ogs:core:dish:mole-negro-enchiladas",
          "beverage_ref": "ogs:core:beverage:garnacha-aragon"
        }
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "{\"id\":\"ogs:core:pairing:mole-negro-garnacha\",\"ogs_version\":\"0.1.0\",\"name\":\"Mole Negro Enchiladas with Aragonese Garnacha\",\"dish_ref\":\"ogs:core:dish:mole-negro-enchiladas\",\"beverage_ref\":\"ogs:core:beverage:garnacha-aragon\",\"match_types\":[{\"type\":\"bridge\",\"score\":90,\"primary\":true,\"description\":\"Ripe fruit, spice, and earthy notes connect the wine to the complex chocolate-chile mole.\"},{\"type\":\"structural_balance\",\"score\":86,\"description\":\"The wine's body and soft tannin suit the dense sauce and filled enchiladas.\"},{\"type\":\"intensity_alignment\",\"score\":88,\"description\":\"Both dish and beverage have enough concentration to meet without either disappearing.\"},{\"type\":\"harmony\",\"score\":84,\"description\":\"Dark fruit, spice, and savory depth create a broadly convergent flavor experience.\"},{\"type\":\"refresh\",\"score\":72,\"description\":\"Moderate acidity provides some relief from the richness, though less than a higher-acid option.\"}],\"overall_score\":87,\"grade\":\"B+\",\"explanation_codes\":[{\"code\":\"AROMA_SHARED_FRUIT\",\"weight\":0.18,\"detail\":\"Ripe red and dark fruit notes in Garnacha echo the fruity dimensions often present in complex mole.\"},{\"code\":\"AROMA_SHARED_SPICE\",\"weight\":0.17,\"detail\":\"Baking spice and peppery accents in the wine bridge with chile and spice elements in the sauce.\"},{\"code\":\"AROMA_SHARED_EARTH\",\"weight\":0.11,\"detail\":\"Earthy undertones in both wine and mole help create a cross-cultural aromatic link.\"},{\"code\":\"HARMONY_FLAVOR_CONVERGE\",\"weight\":0.13,\"detail\":\"The wine's generous fruit and spice converge with the mole's sweet-savory, chocolate-chile profile.\"},{\"code\":\"HARMONY_COMPLEXITY_MATCH\",\"weight\":0.08,\"detail\":\"A robust Garnacha can stand up to the layered complexity of mole negro.\"},{\"code\":\"STRUCT_INTENSITY_MATCH\",\"weight\":0.12,\"detail\":\"Both elements are full-flavored and similarly assertive.\"},{\"code\":\"STRUCT_BODY_ALIGN\",\"weight\":0.08,\"detail\":\"The wine's round body matches the dish's weight and sauce density.\"},{\"code\":\"STRUCT_TANNIN_PROTEIN\",\"weight\":0.05,\"detail\":\"Moderate tannin is softened by the dish's protein and cheese components.\"},{\"code\":\"STRUCT_UMAMI_ENHANCE\",\"weight\":0.04,\"detail\":\"Savory depth in the enchiladas can enhance the wine's perceived fruit and body.\"},{\"code\":\"REFRESH_ACID\",\"weight\":0.04,\"detail\":\"The wine's acidity offers some palate lift against the rich mole.\"}],\"risks\":[{\"code\":\"RISK_SPICE_ALCOHOL\",\"severity\":\"medium\",\"description\":\"If the Garnacha is notably high in alcohol, it may amplify chile heat.\",\"mitigation\":\"Prefer a fresher, less alcoholic expression and avoid serving too warm.\"},{\"code\":\"RISK_INTENSITY_MISMATCH\",\"severity\":\"low\",\"description\":\"A lighter, simple Garnacha may seem insufficient against a deeply concentrated mole.\",\"mitigation\":\"Choose a ripe, full-bodied Aragonese example with ample fruit concentration.\"}],\"confidence\":0.85,\"evidence\":{\"method\":\"estimated\",\"notes\":\"Assessment based on the classic compatibility of ripe, spicy, medium-full reds with complex mole sauces and the stated cross-cultural bridge intent.\"},\"notes\":\"Best with fruit-forward Garnacha showing moderate tannin, generous body, and restrained oak.\"}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:mole-negro-garnacha",
                "expected": "ogs:core:pairing:mole-negro-garnacha",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:mole-negro-enchiladas",
                "expected": "ogs:core:dish:mole-negro-enchiladas",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:garnacha-aragon",
                "expected": "ogs:core:beverage:garnacha-aragon",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3371,
            "output_tokens": 757,
            "latency_ms": 12279.3,
            "usd": 0.011784,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "{\"id\":\"ogs:core:pairing:mole-negro-garnacha\",\"ogs_version\":\"0.1.0\",\"name\":\"Mole Negro Enchiladas with Aragonese Garnacha\",\"dish_ref\":\"ogs:core:dish:mole-negro-enchiladas\",\"beverage_ref\":\"ogs:core:beverage:garnacha-aragon\",\"match_types\":[{\"type\":\"bridge\",\"score\":90,\"description\":\"Ripe dark-fruit notes and warm spice create a bridge to the mole's chile-chocolate complexity.\",\"primary\":true},{\"type\":\"structural_balance\",\"score\":87,\"description\":\"The wine's body and fruit concentration stand up to the dense, savory sauce without collapsing.\"},{\"type\":\"harmony\",\"score\":84,\"description\":\"Shared spice and earthy tones help the pairing feel integrated despite the cross-cultural frame.\"},{\"type\":\"intensity_alignment\",\"score\":86,\"description\":\"Both dish and wine are full-flavored and persistent, keeping the pairing balanced.\"}],\"overall_score\":88,\"grade\":\"B+\",\"explanation_codes\":[{\"code\":\"AROMA_SHARED_FRUIT\",\"weight\":0.2,\"detail\":\"Ripe fruit in the Garnacha links with the dried-fruit and dark-fruit impressions often present in mole negro.\"},{\"code\":\"AROMA_SHARED_SPICE\",\"weight\":0.2,\"detail\":\"Warm spice notes in the wine resonate with the layered chile and spice profile of the sauce.\"},{\"code\":\"AROMA_SHARED_EARTH\",\"weight\":0.12,\"detail\":\"Earthy undertones in both elements support a savory bridge.\"},{\"code\":\"STRUCT_BODY_ALIGN\",\"weight\":0.16,\"detail\":\"A generous, medium-full to full body matches the weight of enchiladas coated in dense mole.\"},{\"code\":\"STRUCT_INTENSITY_MATCH\",\"weight\":0.14,\"detail\":\"Neither the dish nor the wine is easily overwhelmed; intensity is well matched.\"},{\"code\":\"STRUCT_UMAMI_ENHANCE\",\"weight\":0.1,\"detail\":\"Savory depth in the dish can make the Garnacha seem rounder and fruitier.\"},{\"code\":\"HARMONY_FLAVOR_CONVERGE\",\"weight\":0.08,\"detail\":\"Fruit, spice, and earth converge into a coherent overall profile.\"}],\"risks\":[{\"code\":\"RISK_SPICE_ALCOHOL\",\"severity\":\"medium\",\"description\":\"If the Garnacha is notably high in alcohol, chile heat may feel amplified.\",\"mitigation\":\"Serve slightly cool and prefer a fruit-forward, not excessively alcoholic example.\"},{\"code\":\"RISK_BITTER_AMPLIFY\",\"severity\":\"low\",\"description\":\"Bitter cocoa and toasted elements in mole can accentuate any bitter edge in the wine.\",\"mitigation\":\"Choose a ripe, soft-textured Garnacha with modest oak and gentle tannin.\"}],\"confidence\":0.85,\"evidence\":{\"method\":\"estimated\",\"notes\":\"Assessment based on classic structural pairing logic for rich mole sauces and ripe Mediterranean red wines, emphasizing fruit, body, and spice bridge.\"},\"notes\":\"Cross-cultural bridge pairing: ripe fruit and body handle chocolate-chile sauce.\"}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:mole-negro-garnacha",
                "expected": "ogs:core:pairing:mole-negro-garnacha",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:mole-negro-enchiladas",
                "expected": "ogs:core:dish:mole-negro-enchiladas",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:garnacha-aragon",
                "expected": "ogs:core:beverage:garnacha-aragon",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3371,
            "output_tokens": 633,
            "latency_ms": 10107.0,
            "usd": 0.010544,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "{\"id\":\"ogs:core:pairing:mole-negro-garnacha\",\"ogs_version\":\"0.1.0\",\"name\":\"Mole Negro Enchiladas with Aragonese Garnacha\",\"dish_ref\":\"ogs:core:dish:mole-negro-enchiladas\",\"beverage_ref\":\"ogs:core:beverage:garnacha-aragon\",\"match_types\":[{\"type\":\"bridge\",\"score\":91,\"primary\":true,\"description\":\"Ripe dark-fruit character and warm spice notes bridge the chocolate-chile complexity of mole negro.\"},{\"type\":\"structural_balance\",\"score\":88,\"description\":\"The wine's body and softness can stand up to the dense sauce while avoiding a harsh clash.\"},{\"type\":\"intensity_alignment\",\"score\":86,\"description\":\"Both dish and beverage are full-flavored, allowing neither to disappear.\"},{\"type\":\"harmony\",\"score\":84,\"description\":\"Shared spice and dark, earthy flavor directions create a cohesive impression.\"}],\"overall_score\":89,\"grade\":\"B+\",\"explanation_codes\":[{\"code\":\"AROMA_SHARED_FRUIT\",\"weight\":0.19,\"detail\":\"Ripe berry and dark-fruit aromas in Garnacha connect with the dried-fruit and cocoa-fruit impressions often present in mole negro.\"},{\"code\":\"AROMA_SHARED_SPICE\",\"weight\":0.2,\"detail\":\"Baking-spice and peppery notes in the wine echo the layered chile-spice profile of the sauce.\"},{\"code\":\"AROMA_SHARED_EARTH\",\"weight\":0.1,\"detail\":\"Earthy undertones in both wine and mole help create a cross-cultural aromatic bridge.\"},{\"code\":\"STRUCT_INTENSITY_MATCH\",\"weight\":0.18,\"detail\":\"The concentration of mole negro is met by the wine's generous fruit and alcohol presence.\"},{\"code\":\"STRUCT_BODY_ALIGN\",\"weight\":0.12,\"detail\":\"A medium-full Garnacha has enough body for sauce-coated enchiladas without feeling too heavy.\"},{\"code\":\"STRUCT_UMAMI_ENHANCE\",\"weight\":0.08,\"detail\":\"Savory depth in the dish can make the wine seem rounder and more fruit-forward.\"},{\"code\":\"HARMONY_FLAVOR_CONVERGE\",\"weight\":0.08,\"detail\":\"Dark fruit, spice, and earthy tones converge into a coherent overall pairing.\"},{\"code\":\"HARMONY_COMPLEXITY_MATCH\",\"weight\":0.05,\"detail\":\"Both elements offer layered flavors, supporting a balanced, interesting pairing.\"}],\"risks\":[{\"code\":\"RISK_SPICE_ALCOHOL\",\"severity\":\"medium\",\"description\":\"If the enchiladas are especially hot, alcohol can amplify chile heat.\",\"mitigation\":\"Serve the wine slightly cool and favor a fruit-forward, moderate-alcohol expression of Garnacha.\"},{\"code\":\"RISK_BITTER_AMPLIFY\",\"severity\":\"low\",\"description\":\"Charred chile or bitter chocolate notes in the mole can emphasize bitterness in the wine.\",\"mitigation\":\"Choose a plush, ripe style with soft tannin and avoid heavily extracted examples.\"}],\"confidence\":0.86,\"evidence\":{\"method\":\"estimated\",\"notes\":\"Assessment based on classic structural pairing logic for rich mole sauces and ripe, medium-full red wines.\"},\"notes\":\"Cross-cultural bridge pairing: ripe fruit and body handle chocolate-chile sauce.\"}",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "schema_valid": true,
            "schema_errors": [],
            "consistent": true,
            "required_fields": {
              "id": {
                "predicted": "ogs:core:pairing:mole-negro-garnacha",
                "expected": "ogs:core:pairing:mole-negro-garnacha",
                "match": true
              },
              "ogs_version": {
                "predicted": "0.1.0",
                "expected": "0.1.0",
                "match": true
              },
              "dish_ref": {
                "predicted": "ogs:core:dish:mole-negro-enchiladas",
                "expected": "ogs:core:dish:mole-negro-enchiladas",
                "match": true
              },
              "beverage_ref": {
                "predicted": "ogs:core:beverage:garnacha-aragon",
                "expected": "ogs:core:beverage:garnacha-aragon",
                "match": true
              }
            }
          },
          "metrics": {
            "input_tokens": 3371,
            "output_tokens": 672,
            "latency_ms": 11661.8,
            "usd": 0.010934,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 3371.0,
        "output_tokens_mean": 687.333,
        "latency_ms_mean": 11349.367,
        "usd_sum": 0.033262
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-001",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "e52270d817a9dc7f",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family (from the controlled vocabulary) that best classifies the aromatic item below. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"lemon zest\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "fruit"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 332,
            "output_tokens": 4,
            "latency_ms": 633.0,
            "usd": 0.000455,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 332,
            "output_tokens": 4,
            "latency_ms": 678.8,
            "usd": 0.000455,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "fruit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fruit",
            "expected": "fruit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 332,
            "output_tokens": 4,
            "latency_ms": 607.3,
            "usd": 0.000455,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 332.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 639.7,
        "usd_sum": 0.001365
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-002",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "e2096242953ff882",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family that best classifies the item below. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"wet stone\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "earth"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "earth",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "earth",
            "expected": "earth",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 324,
            "output_tokens": 4,
            "latency_ms": 711.4,
            "usd": 0.000445,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "earth",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "earth",
            "expected": "earth",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 324,
            "output_tokens": 4,
            "latency_ms": 810.7,
            "usd": 0.000445,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "earth",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "earth",
            "expected": "earth",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 324,
            "output_tokens": 4,
            "latency_ms": 658.6,
            "usd": 0.000445,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 324.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 726.9,
        "usd_sum": 0.001335
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-003",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "fa5fa32bfe275dfb",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family that best classifies the item below. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"cooked beef fat\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "savory"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "savory",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "savory",
            "expected": "savory",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 326,
            "output_tokens": 5,
            "latency_ms": 698.8,
            "usd": 0.000458,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "savory",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "savory",
            "expected": "savory",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 326,
            "output_tokens": 5,
            "latency_ms": 578.8,
            "usd": 0.000458,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "savory",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "savory",
            "expected": "savory",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 326,
            "output_tokens": 5,
            "latency_ms": 535.9,
            "usd": 0.000458,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 326.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 604.5,
        "usd_sum": 0.001374
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-004",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "e0631034b8b8ece4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best describes a protein slowly cooked while fully submerged in fat at a low temperature. Respond with just the method name from the OGS vocabulary.\n\nINPUT:\n{\n  \"description\": \"duck legs slow-cooked submerged in their own rendered fat at 80 C\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "confit"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "confit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "confit",
            "expected": "confit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 5,
            "latency_ms": 649.9,
            "usd": 0.000558,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "confit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "confit",
            "expected": "confit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 5,
            "latency_ms": 777.5,
            "usd": 0.000558,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "confit",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "confit",
            "expected": "confit",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 406,
            "output_tokens": 5,
            "latency_ms": 1011.8,
            "usd": 0.000558,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 406.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 813.067,
        "usd_sum": 0.001674
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-005",
      "category": "vocabulary",
      "capability": "match_type_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "bbbd81bb3c8b8ea4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nSelect the single OGS match type that best describes a pairing whose strength is driven by a specific shared aromatic compound or family between dish and beverage.\n\nINPUT:\n{\n  \"description\": \"dish and wine are linked by a shared earthy, forest-floor aromatic\"\n}\n\nANSWER FORMAT: one_of:harmony,contrast,bridge,intensity_alignment,structural_balance,refresh",
      "expected": {
        "value": "bridge"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "bridge",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "bridge",
            "expected": "bridge",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 316,
            "output_tokens": 4,
            "latency_ms": 523.8,
            "usd": 0.000435,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "bridge",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "bridge",
            "expected": "bridge",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 316,
            "output_tokens": 4,
            "latency_ms": 800.3,
            "usd": 0.000435,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "bridge",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "bridge",
            "expected": "bridge",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 316,
            "output_tokens": 4,
            "latency_ms": 654.3,
            "usd": 0.000435,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 316.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 659.467,
        "usd_sum": 0.001305
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-006",
      "category": "vocabulary",
      "capability": "component_role_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "6c6ae6f82f354be3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits the described element. Use only roles from the OGS controlled vocabulary.\n\nINPUT:\n{\n  \"element\": \"a soy-dashi based liquid ladled over the rice bowl, providing salt, umami and moisture\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "sauce"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauce",
            "expected": "sauce",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 350,
            "output_tokens": 5,
            "latency_ms": 987.2,
            "usd": 0.000488,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauce",
            "expected": "sauce",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 350,
            "output_tokens": 5,
            "latency_ms": 667.2,
            "usd": 0.000488,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "sauce",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "sauce",
            "expected": "sauce",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 350,
            "output_tokens": 5,
            "latency_ms": 674.7,
            "usd": 0.000488,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 350.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 776.367,
        "usd_sum": 0.001464
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-007",
      "category": "vocabulary",
      "capability": "qualifier_vocabulary",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "0adac8c06022acf5",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nSelect the OGS aromatic qualifier that best describes how the item is expressed.\n\nINPUT:\n{\n  \"item\": \"candied ginger\"\n}\n\nANSWER FORMAT: one_of:fresh,dried,cooked,candied,fermented,zest,overripe",
      "expected": {
        "value": "candied"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "candied",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "candied",
            "expected": "candied",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 298,
            "output_tokens": 5,
            "latency_ms": 630.0,
            "usd": 0.000423,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "candied",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "candied",
            "expected": "candied",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 298,
            "output_tokens": 5,
            "latency_ms": 739.9,
            "usd": 0.000423,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "candied",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "candied",
            "expected": "candied",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 298,
            "output_tokens": 5,
            "latency_ms": 628.6,
            "usd": 0.000423,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 298.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 666.167,
        "usd_sum": 0.001269
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-008",
      "category": "vocabulary",
      "capability": "grade_scale",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "e58319dc1276cdb2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nUsing the OGS pairing grade scale (ogs-match.md §5), return the letter grade for the given overall score. Respond with just the grade token, e.g. 'A+'.\n\nINPUT:\n{\n  \"overall_score\": 92\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "A"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "A",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "A",
            "expected": "A",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 385,
            "output_tokens": 4,
            "latency_ms": 627.3,
            "usd": 0.000521,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "A",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "A",
            "expected": "A",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 385,
            "output_tokens": 4,
            "latency_ms": 906.2,
            "usd": 0.000521,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "A",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "A",
            "expected": "A",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 385,
            "output_tokens": 4,
            "latency_ms": 828.3,
            "usd": 0.000521,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 385.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 787.267,
        "usd_sum": 0.001563
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-009",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "easy",
      "cuisine_region": "spanish",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "d51864ab27ecfa64",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single aromatic family that best characterises smoked paprika. Respond with just the family name.\n\nINPUT:\n{\n  \"item\": \"smoked paprika (pimentón de la Vera)\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "spice"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "spice",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spice",
            "expected": "spice",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 331,
            "output_tokens": 5,
            "latency_ms": 711.0,
            "usd": 0.000464,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "spice",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spice",
            "expected": "spice",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 331,
            "output_tokens": 5,
            "latency_ms": 616.4,
            "usd": 0.000464,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "spice",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "spice",
            "expected": "spice",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 331,
            "output_tokens": 5,
            "latency_ms": 787.2,
            "usd": 0.000464,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 331.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 704.867,
        "usd_sum": 0.001392
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-010",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "medium",
      "cuisine_region": "japanese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "1748069ad0ebdde7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single dominant aromatic family of a classic kombu-katsuo dashi.\n\nINPUT:\n{\n  \"item\": \"classic kombu-katsuobushi dashi\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "marine"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "marine",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "marine",
            "expected": "marine",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 4,
            "latency_ms": 586.4,
            "usd": 0.00045,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "savory",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "savory",
            "expected": "marine",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 5,
            "latency_ms": 727.1,
            "usd": 0.00046,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "savory",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "savory",
            "expected": "marine",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 328,
            "output_tokens": 5,
            "latency_ms": 736.6,
            "usd": 0.00046,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.3333,
      "score_summary": {
        "mean": 0.3333,
        "stdev": 0.5774,
        "sem": 0.3333,
        "ci95_lo": -0.32,
        "ci95_hi": 0.9867,
        "n": 3
      },
      "score_if_valid": 0.3333,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 328.0,
        "output_tokens_mean": 4.667,
        "latency_ms_mean": 683.367,
        "usd_sum": 0.00137
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-011",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "easy",
      "cuisine_region": "indian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "85de6add5c1ebfba",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best matches tandoori chicken cooked at 450°C in a clay oven.\n\nINPUT:\n{\n  \"description\": \"tandoori chicken, marinated and cooked in a clay tandoor at ~450°C\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "roasted"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 398,
            "output_tokens": 5,
            "latency_ms": 721.4,
            "usd": 0.000548,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 398,
            "output_tokens": 5,
            "latency_ms": 776.1,
            "usd": 0.000548,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "roasted",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "roasted",
            "expected": "roasted",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 398,
            "output_tokens": 5,
            "latency_ms": 811.6,
            "usd": 0.000548,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 398.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 769.7,
        "usd_sum": 0.001644
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-012",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "peruvian",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "44c9b26dc70c4557",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best describes Peruvian ceviche's treatment of white fish in lime juice.\n\nINPUT:\n{\n  \"description\": \"diced white fish denatured in freshly squeezed lime juice for 10 minutes with red onion and aji limo\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "cured"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 401,
            "output_tokens": 5,
            "latency_ms": 847.6,
            "usd": 0.000551,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 401,
            "output_tokens": 5,
            "latency_ms": 746.3,
            "usd": 0.000551,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "cured",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "cured",
            "expected": "cured",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 401,
            "output_tokens": 5,
            "latency_ms": 777.2,
            "usd": 0.000551,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 401.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 790.367,
        "usd_sum": 0.001653
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-013",
      "category": "vocabulary",
      "capability": "component_role_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "moroccan",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "dc3e62daf8bdf21a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits harissa stirred into a lamb tagine at service.\n\nINPUT:\n{\n  \"description\": \"a spoonful of fiery harissa stirred into a lamb-and-apricot tagine at service\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "seasoning"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "seasoning",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "seasoning",
            "expected": "seasoning",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 5,
            "latency_ms": 672.4,
            "usd": 0.000483,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "seasoning",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "seasoning",
            "expected": "seasoning",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 5,
            "latency_ms": 684.7,
            "usd": 0.000483,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "seasoning",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "seasoning",
            "expected": "seasoning",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 5,
            "latency_ms": 855.0,
            "usd": 0.000483,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 346.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 737.367,
        "usd_sum": 0.001449
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-014",
      "category": "vocabulary",
      "capability": "qualifier_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "chinese",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.qualifiers"
      ],
      "prompt_template_hash": "3980456a47e9b7c6",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic qualifiers\n\nfresh, dried, cooked, candied, fermented, zest, overripe\n\n---\n\nSelect the OGS aromatic qualifier that best describes the state of Sichuan peppercorn as it is commonly used — whole, toasted, and ground over finished mapo tofu.\n\nINPUT:\n{\n  \"item\": \"Sichuan peppercorn (Zanthoxylum), whole, pan-toasted just before grinding over the dish\"\n}\n\nANSWER FORMAT: one_of:fresh,dried,cooked,candied,fermented,zest,overripe",
      "expected": {
        "value": "dried"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "cooked",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "cooked",
            "expected": "dried",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 336,
            "output_tokens": 5,
            "latency_ms": 751.1,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "cooked",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "cooked",
            "expected": "dried",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 336,
            "output_tokens": 5,
            "latency_ms": 727.5,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "cooked",
          "compliant": true,
          "score": 0.0,
          "detail": {
            "predicted": "cooked",
            "expected": "dried",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 336,
            "output_tokens": 5,
            "latency_ms": 743.9,
            "usd": 0.00047,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.0,
      "score_summary": {
        "mean": 0.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.0,
        "ci95_hi": 0.0,
        "n": 3
      },
      "score_if_valid": 0.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 336.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 740.833,
        "usd_sum": 0.00141
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-015",
      "category": "vocabulary",
      "capability": "grade_scale",
      "difficulty": "easy",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "ad313a98974d64f7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nUsing the OGS pairing grade scale, return the letter grade for a pairing whose `overall_score` is 78.\n\nINPUT:\n{\n  \"overall_score\": 78\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "C+"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "C+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c+",
            "expected": "c+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 5,
            "latency_ms": 1149.6,
            "usd": 0.000514,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "C+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c+",
            "expected": "c+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 5,
            "latency_ms": 680.8,
            "usd": 0.000514,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "C+",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "c+",
            "expected": "c+",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 5,
            "latency_ms": 684.1,
            "usd": 0.000514,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 371.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 838.167,
        "usd_sum": 0.001542
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-016",
      "category": "vocabulary",
      "capability": "match_type_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "british",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.match_types"
      ],
      "prompt_template_hash": "95e41ab82cba029a",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Match types\n\nharmony, contrast, bridge, intensity_alignment, structural_balance, refresh\n\n---\n\nSelect the single OGS match type that best describes a beverage whose carbonation physically scrubs the palate between bites of a rich fried dish.\n\nINPUT:\n{\n  \"description\": \"Champagne cleanses the palate between bites of fish and chips\"\n}\n\nANSWER FORMAT: one_of:harmony,contrast,bridge,intensity_alignment,structural_balance,refresh",
      "expected": {
        "value": "refresh"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "refresh",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "refresh",
            "expected": "refresh",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 313,
            "output_tokens": 4,
            "latency_ms": 4014.9,
            "usd": 0.000431,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "refresh",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "refresh",
            "expected": "refresh",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 313,
            "output_tokens": 4,
            "latency_ms": 790.0,
            "usd": 0.000431,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "refresh",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "refresh",
            "expected": "refresh",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 313,
            "output_tokens": 4,
            "latency_ms": 716.8,
            "usd": 0.000431,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 313.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 1840.567,
        "usd_sum": 0.001293
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-017",
      "category": "vocabulary",
      "capability": "aromatic_family_classification",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.aromatic_families"
      ],
      "prompt_template_hash": "8070fb687fd2f108",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Aromatic families\n\nfruit, floral, herbal, spice, earth, wood, dairy, savory, confection, vegetal, marine\n\n---\n\nUsing the OGS sensory model, return the single dominant aromatic family of fresh cilantro leaves.\n\nINPUT:\n{\n  \"item\": \"fresh cilantro / coriander leaves\"\n}\n\nANSWER FORMAT: one_of:fruit,floral,herbal,spice,earth,wood,dairy,savory,confection,vegetal,marine",
      "expected": {
        "value": "herbal"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 318,
            "output_tokens": 5,
            "latency_ms": 705.7,
            "usd": 0.000448,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 318,
            "output_tokens": 5,
            "latency_ms": 625.3,
            "usd": 0.000448,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "herbal",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "herbal",
            "expected": "herbal",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 318,
            "output_tokens": 5,
            "latency_ms": 534.3,
            "usd": 0.000448,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 318.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 621.767,
        "usd_sum": 0.001344
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-018",
      "category": "vocabulary",
      "capability": "component_role_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "levantine",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.component_roles"
      ],
      "prompt_template_hash": "9e97b59361dbdf0e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Component roles\n\nmain_protein, main_vegetable, starch, sauce, fat_component, acid_component, aromatic_garnish, seasoning, textural_element\n\n---\n\nReturn the single OGS component role that best fits a drizzle of pomegranate molasses dressed onto a fattoush salad.\n\nINPUT:\n{\n  \"description\": \"pomegranate molasses drizzled over a fattoush salad at service\"\n}\n\nANSWER FORMAT: one_of:main_protein,main_vegetable,starch,sauce,fat_component,acid_component,aromatic_garnish,seasoning,textural_element",
      "expected": {
        "value": "acid_component"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "acid_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "acid_component",
            "expected": "acid_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 5,
            "latency_ms": 773.5,
            "usd": 0.000483,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "acid_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "acid_component",
            "expected": "acid_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 5,
            "latency_ms": 560.2,
            "usd": 0.000483,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "acid_component",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "acid_component",
            "expected": "acid_component",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 346,
            "output_tokens": 5,
            "latency_ms": 593.4,
            "usd": 0.000483,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 346.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 642.367,
        "usd_sum": 0.001449
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-019",
      "category": "vocabulary",
      "capability": "cooking_method_vocabulary",
      "difficulty": "medium",
      "cuisine_region": "filipino",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.cooking_methods"
      ],
      "prompt_template_hash": "fc9a763682f1225b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Cooking methods\n\nraw, seared, grilled, roasted, braised, poached, steamed, fried, confit, smoked, fermented, cured, pickled, dehydrated, sous_vide, blanched, sauteed, baked, charred, emulsified\n\n---\n\nPick the single OGS cooking method that best describes bagoong alamang (Philippine fermented salt-cured shrimp paste).\n\nINPUT:\n{\n  \"description\": \"bagoong alamang: salt-cured tiny shrimp left to ferment for 3-6 months\"\n}\n\nANSWER FORMAT: one_of:raw,seared,grilled,roasted,braised,poached,steamed,fried,confit,smoked,fermented,cured,pickled,dehydrated,sous_vide,blanched,sauteed,baked,charred,emulsified",
      "expected": {
        "value": "fermented"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "fermented",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fermented",
            "expected": "fermented",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 400,
            "output_tokens": 5,
            "latency_ms": 993.0,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "fermented",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fermented",
            "expected": "fermented",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 400,
            "output_tokens": 5,
            "latency_ms": 720.4,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "fermented",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "fermented",
            "expected": "fermented",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 400,
            "output_tokens": 5,
            "latency_ms": 1218.9,
            "usd": 0.00055,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 400.0,
        "output_tokens_mean": 5.0,
        "latency_ms_mean": 977.433,
        "usd_sum": 0.00165
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:vocab-020",
      "category": "vocabulary",
      "capability": "grade_scale",
      "difficulty": "medium",
      "cuisine_region": "global",
      "scorer": "exact_match",
      "context_needs": [
        "vocab.grade_scale"
      ],
      "prompt_template_hash": "56ad4f9c58bc1eb7",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### Grade scale (derived from 0-100 overall_score)\n\n- **A+**: 95-100\n- **A**: 90-94\n- **B+**: 85-89\n- **B**: 80-84\n- **C+**: 75-79\n- **C**: 70-74\n- **D**: 60-69\n- **F**: <60\n\n---\n\nUsing the OGS pairing grade scale, return the letter grade for a pairing whose `overall_score` is 64.\n\nINPUT:\n{\n  \"overall_score\": 64\n}\n\nANSWER FORMAT: one_of:A+,A,B+,B,C+,C,D,F",
      "expected": {
        "value": "D"
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "d",
            "expected": "d",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 4,
            "latency_ms": 749.0,
            "usd": 0.000504,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "d",
            "expected": "d",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 4,
            "latency_ms": 602.4,
            "usd": 0.000504,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "D",
          "compliant": true,
          "score": 1.0,
          "detail": {
            "predicted": "d",
            "expected": "d",
            "aliases": []
          },
          "metrics": {
            "input_tokens": 371,
            "output_tokens": 4,
            "latency_ms": 688.8,
            "usd": 0.000504,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 1.0,
      "score_summary": {
        "mean": 1.0,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 1.0,
        "ci95_hi": 1.0,
        "n": 3
      },
      "score_if_valid": 1.0,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 371.0,
        "output_tokens_mean": 4.0,
        "latency_ms_mean": 680.067,
        "usd_sum": 0.001512
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-001",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "a75ad0ee2a385794",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nFrom the OGS explanation-code vocabulary, return the set of codes that best explain WHY the pairing below is excellent. Respond as a JSON array of code IDs. Pick 3 to 6 most relevant positive codes.\n\nINPUT:\n{\n  \"dish\": \"A5 wagyu with black truffle jus (very rich, high umami, earthy aromatics, full body)\",\n  \"beverage\": \"Barolo 2018 (Nebbiolo, high tannin, high acidity, full body, tar/forest-floor)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_FAT",
          "STRUCT_INTENSITY_MATCH",
          "AROMA_SHARED_EARTH",
          "STRUCT_BODY_ALIGN",
          "HARMONY_FLAVOR_CONVERGE"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_TANNIN_FAT\",\"STRUCT_TANNIN_PROTEIN\",\"REFRESH_ACID\",\"AROMA_SHARED_EARTH\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.7273,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.6667,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1158,
            "output_tokens": 41,
            "latency_ms": 1076.9,
            "usd": 0.001857,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_TANNIN_FAT\",\"STRUCT_TANNIN_PROTEIN\",\"REFRESH_ACID\",\"AROMA_SHARED_EARTH\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.7273,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.6667,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1158,
            "output_tokens": 41,
            "latency_ms": 1137.4,
            "usd": 0.001857,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_TANNIN_FAT\",\"STRUCT_TANNIN_PROTEIN\",\"REFRESH_ACID\",\"AROMA_SHARED_EARTH\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.7273,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.6667,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1158,
            "output_tokens": 41,
            "latency_ms": 1199.6,
            "usd": 0.001857,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.7273,
      "score_summary": {
        "mean": 0.7273,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.7273,
        "ci95_hi": 0.7273,
        "n": 3
      },
      "score_if_valid": 0.7273,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1158.0,
        "output_tokens_mean": 41.0,
        "latency_ms_mean": 1137.967,
        "usd_sum": 0.005571
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-002",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "bdc5d8d74c85e9ab",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain the success of the pairing below. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"fresh oysters on the half shell with lemon\",\n  \"beverage\": \"Chablis Premier Cru (Chardonnay, high acid, mineral, lean body)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_ACIDITY_BALANCE",
          "REFRESH_ACID",
          "REFRESH_MINERAL",
          "CONTRAST_RICH_CRISP",
          "HARMONY_REGIONAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"REFRESH_MINERAL\",\"REFRESH_ACID\",\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_BODY_ALIGN\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "HARMONY_FLAVOR_CONVERGE",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_BODY_ALIGN"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1123,
            "output_tokens": 34,
            "latency_ms": 1296.0,
            "usd": 0.001744,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"REFRESH_MINERAL\",\"REFRESH_ACID\",\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_BODY_ALIGN\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "HARMONY_FLAVOR_CONVERGE",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_BODY_ALIGN"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1123,
            "output_tokens": 34,
            "latency_ms": 993.1,
            "usd": 0.001744,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"REFRESH_MINERAL\",\"STRUCT_ACIDITY_BALANCE\",\"CONTRAST_RICH_CRISP\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_FLAVOR_CONVERGE",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.75,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1123,
            "output_tokens": 33,
            "latency_ms": 1347.1,
            "usd": 0.001734,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6222,
      "score_summary": {
        "mean": 0.6222,
        "stdev": 0.0385,
        "sem": 0.0222,
        "ci95_lo": 0.5787,
        "ci95_hi": 0.6658,
        "n": 3
      },
      "score_if_valid": 0.6222,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1123.0,
        "output_tokens_mean": 33.667,
        "latency_ms_mean": 1212.067,
        "usd_sum": 0.005222
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-003",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "3440bcf3978a1c58",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (2 to 4 codes).\n\nINPUT:\n{\n  \"dish\": \"crispy buttermilk fried chicken\",\n  \"beverage\": \"non-vintage Brut Champagne\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "REFRESH_EFFERVESCE",
          "REFRESH_ACID",
          "CONTRAST_RICH_CRISP"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"REFRESH_EFFERVESCE\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.8571,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE"
            ],
            "precision": 0.75,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 31,
            "latency_ms": 946.9,
            "usd": 0.001698,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"REFRESH_EFFERVESCE\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.8571,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE"
            ],
            "precision": 0.75,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 31,
            "latency_ms": 1023.8,
            "usd": 0.001698,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"REFRESH_EFFERVESCE\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"STRUCT_BODY_ALIGN\"]",
          "compliant": true,
          "score": 0.8571,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE",
              "STRUCT_BODY_ALIGN"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE"
            ],
            "precision": 0.75,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 29,
            "latency_ms": 1055.0,
            "usd": 0.001678,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.8571,
      "score_summary": {
        "mean": 0.8571,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8571,
        "ci95_hi": 0.8571,
        "n": 3
      },
      "score_if_valid": 0.8571,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1110.0,
        "output_tokens_mean": 30.333,
        "latency_ms_mean": 1008.567,
        "usd_sum": 0.005074
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-004",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "b3249dafda966f70",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain the success of the pairing below. Respond as a JSON array (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"seared foie gras on brioche (very rich, fatty, slightly sweet-savoury)\",\n  \"beverage\": \"Sauternes (botrytised sweet white, high residual sugar, high acidity)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "STRUCT_ACIDITY_BALANCE",
          "CONTRAST_RICH_CRISP",
          "HARMONY_REGIONAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"STRUCT_ACIDITY_BALANCE\",\"CONTRAST_SWEET_SALTY\",\"CONTRAST_RICH_CRISP\",\"STRUCT_BODY_ALIGN\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "CONTRAST_SWEET_SALTY",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1130,
            "output_tokens": 39,
            "latency_ms": 1132.3,
            "usd": 0.001802,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"REFRESH_ACID\",\"CONTRAST_SWEET_SALTY\",\"STRUCT_BODY_ALIGN\"]",
          "compliant": true,
          "score": 0.25,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.25,
            "recall": 0.25
          },
          "metrics": {
            "input_tokens": 1130,
            "output_tokens": 29,
            "latency_ms": 1047.4,
            "usd": 0.001703,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"REFRESH_ACID\",\"CONTRAST_SWEET_SALTY\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1130,
            "output_tokens": 39,
            "latency_ms": 1110.2,
            "usd": 0.001802,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.4537,
      "score_summary": {
        "mean": 0.4537,
        "stdev": 0.2085,
        "sem": 0.1204,
        "ci95_lo": 0.2178,
        "ci95_hi": 0.6896,
        "n": 3
      },
      "score_if_valid": 0.4537,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1130.0,
        "output_tokens_mean": 35.667,
        "latency_ms_mean": 1096.633,
        "usd_sum": 0.005307
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-005",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "0cb2f1946943a0e2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS explanation codes that best explain why the pairing works. Respond as a JSON array (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"charcoal-grilled lamb rack with rosemary, black pepper crust, charred edges\",\n  \"beverage\": \"Northern Rhone Syrah (Cote-Rotie) — black pepper, violet, smoked meat, moderate-high tannin\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "AROMA_SHARED_SPICE",
          "AROMA_SHARED_WOOD",
          "AROMA_SHARED_HERBAL",
          "STRUCT_TANNIN_FAT",
          "STRUCT_TANNIN_PROTEIN"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"AROMA_SHARED_SPICE\",\"AROMA_SHARED_WOOD\",\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1135,
            "output_tokens": 39,
            "latency_ms": 1142.6,
            "usd": 0.001809,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"AROMA_SHARED_SPICE\",\"AROMA_SHARED_WOOD\",\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.8,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.8,
            "recall": 0.8
          },
          "metrics": {
            "input_tokens": 1135,
            "output_tokens": 37,
            "latency_ms": 1222.2,
            "usd": 0.001789,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"AROMA_SHARED_SPICE\",\"AROMA_SHARED_WOOD\",\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "AROMA_SHARED_SPICE",
              "AROMA_SHARED_WOOD",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1135,
            "output_tokens": 39,
            "latency_ms": 1240.5,
            "usd": 0.001809,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.1155,
        "sem": 0.0667,
        "ci95_lo": 0.536,
        "ci95_hi": 0.7973,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1135.0,
        "output_tokens_mean": 38.333,
        "latency_ms_mean": 1201.767,
        "usd_sum": 0.005407
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-006",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "9a46b0d141442ce3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Pizza Margherita (tomato, mozzarella, basil)\",\n  \"beverage\": \"Chianti Classico (Sangiovese, high acid, medium tannin)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_ACIDITY_BALANCE",
          "HARMONY_REGIONAL",
          "STRUCT_INTENSITY_MATCH",
          "AROMA_SHARED_HERBAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"AROMA_SHARED_HERBAL\",\"STRUCT_TANNIN_PROTEIN\",\"HARMONY_REGIONAL\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1126,
            "output_tokens": 38,
            "latency_ms": 1534.6,
            "usd": 0.001788,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"AROMA_SHARED_HERBAL\",\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_REGIONAL\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1126,
            "output_tokens": 38,
            "latency_ms": 1063.2,
            "usd": 0.001788,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"AROMA_SHARED_HERBAL\",\"STRUCT_TANNIN_FAT\",\"HARMONY_REGIONAL\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1126,
            "output_tokens": 37,
            "latency_ms": 1195.6,
            "usd": 0.001778,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.8889,
      "score_summary": {
        "mean": 0.8889,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8889,
        "ci95_hi": 0.8889,
        "n": 3
      },
      "score_if_valid": 0.8889,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1126.0,
        "output_tokens_mean": 37.667,
        "latency_ms_mean": 1264.467,
        "usd_sum": 0.005354
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-007",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "chinese",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "497adc64f3303eb4",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"mapo tofu with Sichuan peppercorn\",\n  \"beverage\": \"off-dry German Spätlese Riesling\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "REFRESH_ACID",
          "STRUCT_INTENSITY_MATCH",
          "CONTRAST_SWEET_SALTY"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\",\"AROMA_SHARED_FLORAL\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FLORAL",
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1117,
            "output_tokens": 36,
            "latency_ms": 1038.6,
            "usd": 0.001756,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1117,
            "output_tokens": 33,
            "latency_ms": 1023.6,
            "usd": 0.001726,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\",\"AROMA_SHARED_FLORAL\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FLORAL",
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1117,
            "output_tokens": 36,
            "latency_ms": 999.2,
            "usd": 0.001756,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1117.0,
        "output_tokens_mean": 35.0,
        "latency_ms_mean": 1020.467,
        "usd_sum": 0.005238
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-008",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "2120f96a0444a50e",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"herb-crusted rack of lamb, medium-rare\",\n  \"beverage\": \"Northern Rhône Syrah\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_PROTEIN",
          "AROMA_SHARED_HERBAL",
          "STRUCT_BODY_ALIGN",
          "HARMONY_FLAVOR_CONVERGE"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_BODY_ALIGN\",\"AROMA_SHARED_HERBAL\",\"AROMA_SHARED_EARTH\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "AROMA_SHARED_HERBAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1114,
            "output_tokens": 36,
            "latency_ms": 1373.4,
            "usd": 0.001753,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"AROMA_SHARED_HERBAL\",\"AROMA_SHARED_EARTH\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "AROMA_SHARED_HERBAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1114,
            "output_tokens": 39,
            "latency_ms": 1338.1,
            "usd": 0.001783,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"AROMA_SHARED_HERBAL\",\"AROMA_SHARED_EARTH\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "AROMA_SHARED_HERBAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1114,
            "output_tokens": 39,
            "latency_ms": 1442.9,
            "usd": 0.001783,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.5185,
      "score_summary": {
        "mean": 0.5185,
        "stdev": 0.1283,
        "sem": 0.0741,
        "ci95_lo": 0.3733,
        "ci95_hi": 0.6637,
        "n": 3
      },
      "score_if_valid": 0.5185,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1114.0,
        "output_tokens_mean": 38.0,
        "latency_ms_mean": 1384.8,
        "usd_sum": 0.005319
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-009",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "thai",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "c077c498ceed5f73",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Thai green curry with chicken and Thai basil\",\n  \"beverage\": \"off-dry New Zealand Sauvignon Blanc\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "AROMA_SHARED_HERBAL",
          "REFRESH_ACID",
          "STRUCT_INTENSITY_MATCH"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_SWEET_OFFSET\",\"AROMA_SHARED_HERBAL\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "CONTRAST_RICH_CRISP",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1114,
            "output_tokens": 38,
            "latency_ms": 1270.7,
            "usd": 0.001773,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_SWEET_OFFSET\",\"AROMA_SHARED_HERBAL\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "CONTRAST_RICH_CRISP",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1114,
            "output_tokens": 38,
            "latency_ms": 1432.0,
            "usd": 0.001773,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_SWEET_OFFSET\",\"AROMA_SHARED_HERBAL\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "CONTRAST_RICH_CRISP",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1114,
            "output_tokens": 38,
            "latency_ms": 1096.1,
            "usd": 0.001773,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1114.0,
        "output_tokens_mean": 38.0,
        "latency_ms_mean": 1266.267,
        "usd_sum": 0.005319
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-010",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "956c6ce654c2e297",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"moules marinières\",\n  \"beverage\": \"Muscadet sur lie\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_ACIDITY_BALANCE",
          "REFRESH_ACID",
          "AROMA_SHARED_HERBAL",
          "HARMONY_REGIONAL",
          "REFRESH_MINERAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"HARMONY_REGIONAL\",\"STRUCT_ACIDITY_BALANCE\",\"REFRESH_MINERAL\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "HARMONY_FLAVOR_CONVERGE",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 37,
            "latency_ms": 1171.3,
            "usd": 0.001757,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"HARMONY_REGIONAL\",\"STRUCT_ACIDITY_BALANCE\",\"REFRESH_MINERAL\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "HARMONY_FLAVOR_CONVERGE",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 37,
            "latency_ms": 1006.4,
            "usd": 0.001757,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"HARMONY_REGIONAL\",\"STRUCT_ACIDITY_BALANCE\",\"REFRESH_MINERAL\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6,
          "detail": {
            "predicted": [
              "HARMONY_FLAVOR_CONVERGE",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "precision": 0.6,
            "recall": 0.6
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 37,
            "latency_ms": 1177.1,
            "usd": 0.001757,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6,
      "score_summary": {
        "mean": 0.6,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6,
        "ci95_hi": 0.6,
        "n": 3
      },
      "score_if_valid": 0.6,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1110.0,
        "output_tokens_mean": 37.0,
        "latency_ms_mean": 1118.267,
        "usd_sum": 0.005271
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-011",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "mexican",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "1dfeb782059a4814",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"tacos al pastor with pineapple and chili\",\n  \"beverage\": \"dry sparkling rosé (Cava rosado)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "REFRESH_EFFERVESCE",
          "AROMA_SHARED_FRUIT",
          "CONTRAST_RICH_CRISP"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"REFRESH_EFFERVESCE\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"AROMA_SHARED_FRUIT\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1116,
            "output_tokens": 37,
            "latency_ms": 1043.3,
            "usd": 0.001765,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"REFRESH_EFFERVESCE\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\",\"AROMA_SHARED_FRUIT\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1116,
            "output_tokens": 37,
            "latency_ms": 1086.1,
            "usd": 0.001765,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"REFRESH_EFFERVESCE\",\"REFRESH_ACID\",\"CONTRAST_RICH_CRISP\",\"AROMA_SHARED_FRUIT\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_ACID",
              "REFRESH_EFFERVESCE",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "REFRESH_EFFERVESCE",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1116,
            "output_tokens": 37,
            "latency_ms": 1064.1,
            "usd": 0.001765,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1116.0,
        "output_tokens_mean": 37.0,
        "latency_ms_mean": 1064.5,
        "usd_sum": 0.005295
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-012",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "28df36c8ae155697",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"duck confit, crispy skin\",\n  \"beverage\": \"mature Burgundy Pinot Noir (10 years)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_FAT",
          "AROMA_SHARED_EARTH",
          "HARMONY_REGIONAL",
          "HARMONY_COMPLEXITY_MATCH"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_TANNIN_FAT\",\"STRUCT_TANNIN_PROTEIN\",\"REFRESH_ACID\",\"AROMA_SHARED_EARTH\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_COMPLEXITY_MATCH",
              "HARMONY_REGIONAL",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1113,
            "output_tokens": 37,
            "latency_ms": 1009.1,
            "usd": 0.001761,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_TANNIN_FAT\",\"STRUCT_TANNIN_PROTEIN\",\"REFRESH_ACID\",\"AROMA_SHARED_EARTH\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_COMPLEXITY_MATCH",
              "HARMONY_REGIONAL",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1113,
            "output_tokens": 37,
            "latency_ms": 1066.7,
            "usd": 0.001761,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_TANNIN_FAT\",\"STRUCT_TANNIN_PROTEIN\",\"REFRESH_ACID\",\"AROMA_SHARED_EARTH\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_EARTH",
              "HARMONY_COMPLEXITY_MATCH",
              "HARMONY_REGIONAL",
              "STRUCT_TANNIN_FAT"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1113,
            "output_tokens": 37,
            "latency_ms": 1183.4,
            "usd": 0.001761,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.4444,
      "score_summary": {
        "mean": 0.4444,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.4444,
        "ci95_hi": 0.4444,
        "n": 3
      },
      "score_if_valid": 0.4444,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1113.0,
        "output_tokens_mean": 37.0,
        "latency_ms_mean": 1086.4,
        "usd_sum": 0.005283
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-013",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "spanish",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "18a20e7e30844c41",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"jamón ibérico de bellota, sliced thin\",\n  \"beverage\": \"chilled Fino sherry (Manzanilla)\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_UMAMI_ENHANCE",
          "HARMONY_REGIONAL",
          "AROMA_SHARED_SAVORY",
          "REFRESH_MINERAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\",\"REFRESH_ACID\",\"REFRESH_MINERAL\",\"HARMONY_REGIONAL\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_SAVORY",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1119,
            "output_tokens": 30,
            "latency_ms": 971.3,
            "usd": 0.001699,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\",\"REFRESH_ACID\",\"REFRESH_MINERAL\",\"HARMONY_REGIONAL\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_SAVORY",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1119,
            "output_tokens": 30,
            "latency_ms": 1180.6,
            "usd": 0.001699,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\",\"STRUCT_UMAMI_ENHANCE\",\"REFRESH_MINERAL\",\"HARMONY_REGIONAL\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "expected": [
              "AROMA_SHARED_SAVORY",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1119,
            "output_tokens": 33,
            "latency_ms": 839.5,
            "usd": 0.001729,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.5185,
      "score_summary": {
        "mean": 0.5185,
        "stdev": 0.1283,
        "sem": 0.0741,
        "ci95_lo": 0.3733,
        "ci95_hi": 0.6637,
        "n": 3
      },
      "score_if_valid": 0.5185,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1119.0,
        "output_tokens_mean": 31.0,
        "latency_ms_mean": 997.133,
        "usd_sum": 0.005127
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-014",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "korean",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "f9be4ef2853d67b3",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"bulgogi: soy-pear-garlic marinated grilled beef\",\n  \"beverage\": \"medium-bodied unoaked Merlot\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_PROTEIN",
          "STRUCT_SWEET_OFFSET",
          "AROMA_SHARED_FRUIT",
          "STRUCT_BODY_ALIGN"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"AROMA_SHARED_FRUIT\",\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1119,
            "output_tokens": 35,
            "latency_ms": 1120.5,
            "usd": 0.001749,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"AROMA_SHARED_FRUIT\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1119,
            "output_tokens": 35,
            "latency_ms": 1125.5,
            "usd": 0.001749,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"AROMA_SHARED_FRUIT\",\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1119,
            "output_tokens": 35,
            "latency_ms": 1120.5,
            "usd": 0.001749,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6667,
      "score_summary": {
        "mean": 0.6667,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.6667,
        "ci95_hi": 0.6667,
        "n": 3
      },
      "score_if_valid": 0.6667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1119.0,
        "output_tokens_mean": 35.0,
        "latency_ms_mean": 1122.167,
        "usd_sum": 0.005247
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-015",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "easy",
      "cuisine_region": "british",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "5474f731e5d9751c",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Stilton blue cheese\",\n  \"beverage\": \"Vintage Port\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "CONTRAST_SWEET_SALTY",
          "STRUCT_INTENSITY_MATCH",
          "HARMONY_COMPLEXITY_MATCH",
          "STRUCT_SWEET_OFFSET"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\",\"AROMA_SHARED_EARTH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_EARTH",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_COMPLEXITY_MATCH",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1106,
            "output_tokens": 36,
            "latency_ms": 1019.2,
            "usd": 0.001743,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"CONTRAST_SWEET_SALTY\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_COMPLEXITY_MATCH",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1106,
            "output_tokens": 39,
            "latency_ms": 1089.1,
            "usd": 0.001772,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\",\"AROMA_SHARED_FRUIT\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_COMPLEXITY_MATCH",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1106,
            "output_tokens": 35,
            "latency_ms": 1076.8,
            "usd": 0.001733,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.5185,
      "score_summary": {
        "mean": 0.5185,
        "stdev": 0.1283,
        "sem": 0.0741,
        "ci95_lo": 0.3733,
        "ci95_hi": 0.6637,
        "n": 3
      },
      "score_if_valid": 0.5185,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1106.0,
        "output_tokens_mean": 36.667,
        "latency_ms_mean": 1061.7,
        "usd_sum": 0.005248
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-016",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "japanese",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "fc7f227aebed6756",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"shrimp tempura, light fry\",\n  \"beverage\": \"chilled junmai ginjo sake\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_UMAMI_ENHANCE",
          "REFRESH_MINERAL",
          "AROMA_SHARED_FLORAL",
          "STRUCT_INTENSITY_MATCH",
          "HARMONY_REGIONAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\",\"HARMONY_TEXTURE_MATCH\",\"REFRESH_MINERAL\",\"CONTRAST_RICH_CRISP\"]",
          "compliant": true,
          "score": 0.4,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_TEXTURE_MATCH",
              "REFRESH_MINERAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FLORAL",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.4,
            "recall": 0.4
          },
          "metrics": {
            "input_tokens": 1113,
            "output_tokens": 33,
            "latency_ms": 923.9,
            "usd": 0.001721,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\",\"HARMONY_TEXTURE_MATCH\",\"CONTRAST_RICH_CRISP\",\"REFRESH_ACID\"]",
          "compliant": true,
          "score": 0.2,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_TEXTURE_MATCH",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FLORAL",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.2,
            "recall": 0.2
          },
          "metrics": {
            "input_tokens": 1113,
            "output_tokens": 33,
            "latency_ms": 1025.5,
            "usd": 0.001721,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_INTENSITY_MATCH\",\"STRUCT_BODY_ALIGN\",\"HARMONY_COMPLEXITY_MATCH\",\"CONTRAST_RICH_CRISP\",\"REFRESH_ACID\"]",
          "compliant": true,
          "score": 0.2,
          "detail": {
            "predicted": [
              "CONTRAST_RICH_CRISP",
              "HARMONY_COMPLEXITY_MATCH",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH"
            ],
            "expected": [
              "AROMA_SHARED_FLORAL",
              "HARMONY_REGIONAL",
              "REFRESH_MINERAL",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_UMAMI_ENHANCE"
            ],
            "precision": 0.2,
            "recall": 0.2
          },
          "metrics": {
            "input_tokens": 1113,
            "output_tokens": 35,
            "latency_ms": 1012.7,
            "usd": 0.001741,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.2667,
      "score_summary": {
        "mean": 0.2667,
        "stdev": 0.1155,
        "sem": 0.0667,
        "ci95_lo": 0.136,
        "ci95_hi": 0.3973,
        "n": 3
      },
      "score_if_valid": 0.2667,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1113.0,
        "output_tokens_mean": 33.667,
        "latency_ms_mean": 987.367,
        "usd_sum": 0.005183
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-017",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "greek",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "024198497e327bb1",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"horiatiki salad with feta, tomato, olive, oregano\",\n  \"beverage\": \"Santorini Assyrtiko\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "REFRESH_ACID",
          "REFRESH_MINERAL",
          "HARMONY_REGIONAL",
          "AROMA_SHARED_HERBAL"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"HARMONY_REGIONAL\",\"STRUCT_ACIDITY_BALANCE\",\"REFRESH_ACID\",\"REFRESH_MINERAL\",\"AROMA_SHARED_HERBAL\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1118,
            "output_tokens": 34,
            "latency_ms": 1025.0,
            "usd": 0.001738,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"HARMONY_REGIONAL\",\"STRUCT_ACIDITY_BALANCE\",\"REFRESH_ACID\",\"REFRESH_MINERAL\",\"AROMA_SHARED_HERBAL\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1118,
            "output_tokens": 34,
            "latency_ms": 998.1,
            "usd": 0.001738,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"HARMONY_REGIONAL\",\"STRUCT_ACIDITY_BALANCE\",\"REFRESH_ACID\",\"REFRESH_MINERAL\",\"AROMA_SHARED_HERBAL\"]",
          "compliant": true,
          "score": 0.8889,
          "detail": {
            "predicted": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL",
              "STRUCT_ACIDITY_BALANCE"
            ],
            "expected": [
              "AROMA_SHARED_HERBAL",
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "REFRESH_MINERAL"
            ],
            "precision": 0.8,
            "recall": 1.0
          },
          "metrics": {
            "input_tokens": 1118,
            "output_tokens": 34,
            "latency_ms": 1036.1,
            "usd": 0.001738,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.8889,
      "score_summary": {
        "mean": 0.8889,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.8889,
        "ci95_hi": 0.8889,
        "n": 3
      },
      "score_if_valid": 0.8889,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1118.0,
        "output_tokens_mean": 34.0,
        "latency_ms_mean": 1019.733,
        "usd_sum": 0.005214
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-018",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "hard",
      "cuisine_region": "chinese",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "283bbb9e703e6e30",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Peking duck with pancake, scallion, hoisin\",\n  \"beverage\": \"10-year-old off-dry Mosel Riesling\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "AROMA_SHARED_FRUIT",
          "REFRESH_ACID",
          "HARMONY_COMPLEXITY_MATCH"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_SWEET_OFFSET\",\"CONTRAST_RICH_CRISP\",\"AROMA_SHARED_FRUIT\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_COMPLEXITY_MATCH",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1120,
            "output_tokens": 37,
            "latency_ms": 1200.6,
            "usd": 0.00177,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_SWEET_OFFSET\",\"AROMA_SHARED_FRUIT\",\"CONTRAST_RICH_CRISP\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_COMPLEXITY_MATCH",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1120,
            "output_tokens": 37,
            "latency_ms": 1167.5,
            "usd": 0.00177,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_ACIDITY_BALANCE\",\"STRUCT_SWEET_OFFSET\",\"CONTRAST_RICH_CRISP\",\"AROMA_SHARED_FRUIT\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "AROMA_SHARED_FRUIT",
              "CONTRAST_RICH_CRISP",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "AROMA_SHARED_FRUIT",
              "HARMONY_COMPLEXITY_MATCH",
              "REFRESH_ACID",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1120,
            "output_tokens": 37,
            "latency_ms": 1156.1,
            "usd": 0.00177,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.4444,
      "score_summary": {
        "mean": 0.4444,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.4444,
        "ci95_hi": 0.4444,
        "n": 3
      },
      "score_if_valid": 0.4444,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1120.0,
        "output_tokens_mean": 37.0,
        "latency_ms_mean": 1174.733,
        "usd_sum": 0.00531
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-019",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "medium",
      "cuisine_region": "italian",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "9ea6cd79384f9ee2",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"Sunday ragù napoletano (long-cooked beef and pork)\",\n  \"beverage\": \"Campanian Aglianico, 5 years\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_TANNIN_PROTEIN",
          "HARMONY_REGIONAL",
          "STRUCT_BODY_ALIGN",
          "AROMA_SHARED_WOOD"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"STRUCT_ACIDITY_BALANCE\",\"HARMONY_REGIONAL\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "STRUCT_ACIDITY_BALANCE",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_WOOD",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1123,
            "output_tokens": 38,
            "latency_ms": 1061.5,
            "usd": 0.001784,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"REFRESH_ACID\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_REGIONAL\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_WOOD",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1123,
            "output_tokens": 36,
            "latency_ms": 1100.8,
            "usd": 0.001764,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_TANNIN_PROTEIN\",\"STRUCT_TANNIN_FAT\",\"REFRESH_ACID\",\"HARMONY_REGIONAL\",\"STRUCT_INTENSITY_MATCH\"]",
          "compliant": true,
          "score": 0.4444,
          "detail": {
            "predicted": [
              "HARMONY_REGIONAL",
              "REFRESH_ACID",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_TANNIN_FAT",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "expected": [
              "AROMA_SHARED_WOOD",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_TANNIN_PROTEIN"
            ],
            "precision": 0.4,
            "recall": 0.5
          },
          "metrics": {
            "input_tokens": 1123,
            "output_tokens": 36,
            "latency_ms": 1257.4,
            "usd": 0.001764,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.4444,
      "score_summary": {
        "mean": 0.4444,
        "stdev": 0.0,
        "sem": 0.0,
        "ci95_lo": 0.4444,
        "ci95_hi": 0.4444,
        "n": 3
      },
      "score_if_valid": 0.4444,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1123.0,
        "output_tokens_mean": 36.667,
        "latency_ms_mean": 1139.9,
        "usd_sum": 0.005312
      }
    },
    {
      "task_id": "ogs:core:benchmark-task:xplain-020",
      "category": "explanation",
      "capability": "explanation_code_selection",
      "difficulty": "easy",
      "cuisine_region": "french",
      "scorer": "set_f1",
      "context_needs": [
        "vocab.explanation_codes.positive"
      ],
      "prompt_template_hash": "43005f44c870684b",
      "prompt": "You are a gastronomy expert evaluated by the Open Gastronomy Standard (OGS) Benchmark.\n\nYou will be given:\n- The relevant slices of the OGS specification (controlled vocabularies, scales,\n  JSON schemas) needed to answer the task correctly.\n- A task instruction, an input payload, and an answer format.\n\nFollow these rules:\n- Use ONLY the OGS controlled vocabularies supplied below. Do not invent new\n  enum values, tag names, or code IDs.\n- When the task asks for JSON, respond with a single JSON document and nothing\n  else. Do NOT wrap JSON in markdown code fences.\n- When the task asks for a single token (e.g. 'a', '3', 'fruit'), respond with\n  just the token and no surrounding prose.\n- Intensities use the OGS 0-10 scale unless otherwise stated\n  (0 = not detectable, 10 = overwhelming).\n- If the task supplies a list of candidates, your answer must come from that\n  list exactly.\n\n## OGS Context\n\n### OGS Explanation Codes (positive only)\n\nEach code has a fixed `id` (UPPER_SNAKE_CASE). Use the exact `id` string.\n\n**aromatic_bridge**\n- `AROMA_SHARED_FRUIT` — Shared Fruit Aromatics: Both dish and beverage share fruit aromatic families, creating an aromatic bridge.\n- `AROMA_SHARED_SPICE` — Shared Spice Notes: Spice aromatics appear in both dish and beverage, linking them aromatically.\n- `AROMA_SHARED_EARTH` — Shared Earth Notes: Earthy, mineral, or fungal aromatics bridge the dish and beverage.\n- `AROMA_SHARED_HERBAL` — Shared Herbal Notes: Herbal aromatics present in both dish and beverage create continuity.\n- `AROMA_SHARED_FLORAL` — Shared Floral Notes: Floral aromatics appear in both elements, creating an elegant aromatic link.\n- `AROMA_SHARED_WOOD` — Shared Wood/Oak Notes: Wood, smoke, or oak aromatics bridge dish and beverage (e.g., grilled meat with oaked wine).\n\n**contrast**\n- `CONTRAST_SWEET_ACID` — Sweet-Acid Contrast: Sweetness in one element contrasts with acidity in the other, creating dynamic tension.\n- `CONTRAST_RICH_CRISP` — Rich-Crisp Contrast: A rich, heavy dish is contrasted by a crisp, lean beverage (or vice versa).\n- `CONTRAST_SWEET_SALTY` — Sweet-Salty Contrast: Sweetness in the beverage contrasts with saltiness in the dish, creating interest.\n- `CONTRAST_TEMP_TEXTURE` — Temperature-Texture Contrast: Temperature or textural differences between dish and beverage create pleasant contrast.\n\n**harmony**\n- `HARMONY_FLAVOR_CONVERGE` — Flavor Profile Convergence: The overall flavor profiles of dish and beverage converge, creating a unified taste experience.\n- `HARMONY_TEXTURE_MATCH` — Textural Harmony: The textures of dish and beverage complement each other harmoniously.\n- `HARMONY_REGIONAL` — Regional Affinity: Dish and beverage share a regional origin, reflecting terroir-driven compatibility.\n- `HARMONY_COMPLEXITY_MATCH` — Complexity Match: Both dish and beverage have similar complexity levels, so neither outshines the other.\n\n**positive_structural**\n- `STRUCT_INTENSITY_MATCH` — Intensity Match: The overall flavor intensity of the dish and beverage are well-aligned, so neither overpowers the other.\n- `STRUCT_BODY_ALIGN` — Body Alignment: The body/weight of the dish matches the body of the beverage, creating a balanced mouthfeel.\n- `STRUCT_ACIDITY_BALANCE` — Acidity Balance: The beverage's acidity cuts through richness or matches the dish's acidity, creating freshness.\n- `STRUCT_TANNIN_FAT` — Tannin-Fat Synergy: Tannins in the beverage are softened by fat in the dish, creating a smooth, integrated sensation.\n- `STRUCT_TANNIN_PROTEIN` — Tannin-Protein Synergy: Proteins in the dish bind with tannins, reducing astringency and enhancing mouthfeel.\n- `STRUCT_SWEET_OFFSET` — Sweetness Offset: Residual sweetness in the beverage balances salt, spice, or acidity in the dish.\n- `STRUCT_UMAMI_ENHANCE` — Umami Enhancement: Umami in the dish enhances the perceived fruit and body of the beverage.\n- `STRUCT_ALCOHOL_WARMTH` — Alcohol Warmth Complement: The warmth of alcohol complements rich, hearty dishes without overpowering.\n\n**refresh**\n- `REFRESH_EFFERVESCE` — Effervescence Cleanse: Carbonation in the beverage physically scrubs the palate between bites of rich food.\n- `REFRESH_ACID` — Acid Refresh: High acidity in the beverage refreshes the palate, cutting through fat and richness.\n- `REFRESH_MINERAL` — Mineral Reset: Mineral character in the beverage provides a clean, palate-resetting effect.\n\n---\n\nReturn the OGS positive explanation codes that best explain why the pairing below works. Respond as a JSON array of code IDs (3 to 5 codes).\n\nINPUT:\n{\n  \"dish\": \"foie gras torchon with brioche\",\n  \"beverage\": \"Sauternes\"\n}\n\nANSWER FORMAT: set_of_codes",
      "expected": {
        "values": [
          "STRUCT_SWEET_OFFSET",
          "CONTRAST_SWEET_SALTY",
          "HARMONY_REGIONAL",
          "STRUCT_BODY_ALIGN"
        ]
      },
      "samples": [
        {
          "sample": 0,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"REFRESH_ACID\",\"CONTRAST_SWEET_SALTY\",\"STRUCT_BODY_ALIGN\"]",
          "compliant": true,
          "score": 0.75,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_SALTY",
              "REFRESH_ACID",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.75,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 29,
            "latency_ms": 1006.6,
            "usd": 0.001678,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 1,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"CONTRAST_SWEET_SALTY\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 39,
            "latency_ms": 1178.9,
            "usd": 0.001778,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        },
        {
          "sample": 2,
          "raw_response": "[\"STRUCT_SWEET_OFFSET\",\"CONTRAST_SWEET_SALTY\",\"STRUCT_BODY_ALIGN\",\"STRUCT_INTENSITY_MATCH\",\"HARMONY_FLAVOR_CONVERGE\"]",
          "compliant": true,
          "score": 0.6667,
          "detail": {
            "predicted": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_FLAVOR_CONVERGE",
              "STRUCT_BODY_ALIGN",
              "STRUCT_INTENSITY_MATCH",
              "STRUCT_SWEET_OFFSET"
            ],
            "expected": [
              "CONTRAST_SWEET_SALTY",
              "HARMONY_REGIONAL",
              "STRUCT_BODY_ALIGN",
              "STRUCT_SWEET_OFFSET"
            ],
            "precision": 0.6,
            "recall": 0.75
          },
          "metrics": {
            "input_tokens": 1110,
            "output_tokens": 39,
            "latency_ms": 1294.6,
            "usd": 0.001778,
            "model_version": "gpt-5.4-2026-03-05",
            "finish_reason": "stop"
          }
        }
      ],
      "score": 0.6945,
      "score_summary": {
        "mean": 0.6945,
        "stdev": 0.0481,
        "sem": 0.0278,
        "ci95_lo": 0.64,
        "ci95_hi": 0.7489,
        "n": 3
      },
      "score_if_valid": 0.6945,
      "format_rate": 1.0,
      "metrics": {
        "input_tokens_mean": 1110.0,
        "output_tokens_mean": 35.667,
        "latency_ms_mean": 1160.033,
        "usd_sum": 0.005234
      }
    }
  ],
  "category_scores": {
    "composition": {
      "mean_score": 0.8752,
      "ci95_lo": 0.7385,
      "ci95_hi": 1.0119,
      "stdev": 0.3119,
      "task_count": 20,
      "weight": 0.12,
      "format_rate": 1.0,
      "score_if_valid": 0.8752
    },
    "dietary": {
      "mean_score": 0.8333,
      "ci95_lo": 0.6131,
      "ci95_hi": 1.0536,
      "stdev": 0.3892,
      "task_count": 12,
      "weight": 0.06,
      "format_rate": 1.0,
      "score_if_valid": 0.8333
    },
    "explanation": {
      "mean_score": 0.6111,
      "ci95_lo": 0.5397,
      "ci95_hi": 0.6824,
      "stdev": 0.1628,
      "task_count": 20,
      "weight": 0.1,
      "format_rate": 1.0,
      "score_if_valid": 0.6111
    },
    "menu_engineering": {
      "mean_score": 1.0,
      "ci95_lo": 1.0,
      "ci95_hi": 1.0,
      "stdev": 0.0,
      "task_count": 8,
      "weight": 0.03,
      "format_rate": 1.0,
      "score_if_valid": 1.0
    },
    "nutrition": {
      "mean_score": 0.8973,
      "ci95_lo": 0.7728,
      "ci95_hi": 1.0218,
      "stdev": 0.2009,
      "task_count": 10,
      "weight": 0.03,
      "format_rate": 1.0,
      "score_if_valid": 0.8973
    },
    "pairing": {
      "mean_score": 0.9742,
      "ci95_lo": 0.9238,
      "ci95_hi": 1.0247,
      "stdev": 0.1208,
      "task_count": 22,
      "weight": 0.15,
      "format_rate": 1.0,
      "score_if_valid": 0.9742
    },
    "regional": {
      "mean_score": 1.0,
      "ci95_lo": 1.0,
      "ci95_hi": 1.0,
      "stdev": 0.0,
      "task_count": 24,
      "weight": 0.05,
      "format_rate": 1.0,
      "score_if_valid": 1.0
    },
    "risk": {
      "mean_score": 0.85,
      "ci95_lo": 0.6894,
      "ci95_hi": 1.0106,
      "stdev": 0.3663,
      "task_count": 20,
      "weight": 0.08,
      "format_rate": 1.0,
      "score_if_valid": 0.85
    },
    "safety": {
      "mean_score": 0.9808,
      "ci95_lo": 0.9431,
      "ci95_hi": 1.0185,
      "stdev": 0.0666,
      "task_count": 12,
      "weight": 0.06,
      "format_rate": 1.0,
      "score_if_valid": 0.9808
    },
    "sensory": {
      "mean_score": 0.6366,
      "ci95_lo": 0.5056,
      "ci95_hi": 0.7677,
      "stdev": 0.3065,
      "task_count": 21,
      "weight": 0.12,
      "format_rate": 1.0,
      "score_if_valid": 0.6366
    },
    "structured_output": {
      "mean_score": 0.8833,
      "ci95_lo": 0.7555,
      "ci95_hi": 1.0112,
      "stdev": 0.2917,
      "task_count": 20,
      "weight": 0.12,
      "format_rate": 1.0,
      "score_if_valid": 0.8833
    },
    "vocabulary": {
      "mean_score": 0.9275,
      "ci95_lo": 0.8273,
      "ci95_hi": 1.0278,
      "stdev": 0.2453,
      "task_count": 23,
      "weight": 0.08,
      "format_rate": 1.0,
      "score_if_valid": 0.9275
    }
  },
  "aggregate": {
    "score": 85.26,
    "ci95_lo": 81.68,
    "ci95_hi": 88.84,
    "scale": "0-100",
    "partial": false,
    "missing_categories": []
  },
  "metrics": {
    "latency_ms_mean": 1611.455,
    "latency_ms_median": 787.267,
    "input_tokens_mean": 728.8,
    "output_tokens_mean": 67.2,
    "usd_total": 1.011608,
    "tasks_with_usd": 213
  },
  "weights": {
    "vocabulary": 0.08,
    "sensory": 0.12,
    "composition": 0.12,
    "pairing": 0.15,
    "explanation": 0.1,
    "risk": 0.08,
    "structured_output": 0.12,
    "regional": 0.05,
    "dietary": 0.06,
    "safety": 0.06,
    "nutrition": 0.03,
    "menu_engineering": 0.03
  },
  "judge": {
    "judge_model": "gpt-5.4",
    "judge_prompt_hash": "756db6a487714bca",
    "judge_validated": false,
    "excluded_task_ids": [
      "ogs:core:benchmark-task:menu-009"
    ],
    "note": "Rubric-judge tasks are scored and retained in `tasks`, but excluded from `aggregate` until the judge clears the human-validation protocol (r > 0.8 against a 50-task sample)."
  }
}